Compare commits

...
233 Commits
Author SHA1 Message Date
Will Pazner c11b2df36f Fix some CuDSS lifetime issues 2026-08-19 11:13:02 -07:00
Will Pazner 1299e13e65 Fix bug in GPU hybridization 2026-08-19 11:12:46 -07:00
Tzanio Kolev 907a629f82 Merge pull request #5232 from mfem/tuple-refactor
refactor tuple for generic size
2026-08-18 17:56:53 -07:00
Tzanio Kolev e032c15aef Merge pull request #5249 from mfem/multi-vector-dev
Add new array-of-Vectors class that supports separate memory allocations for the individual Vectors
2026-08-18 10:57:15 -07:00
Veselin Dobrev 10ceb3e66b Added CHANGELOG entry for class MultiVector 2026-08-18 10:47:01 -07:00
Tzanio Kolev efa30a4a62 Merge pull request #5400 from mfem/gpu_em
GPU improvements for electromagnetics
2026-08-18 10:41:44 -07:00
Tzanio Kolev 3ef9a5c668 Merge branch 'master' into gpu_em 2026-08-17 19:00:24 -07:00
Tzanio Kolev 7b85e1e9c1 Merge pull request #5440 from Sbozzolo/cuda-multi-arch-makefile
Makefile: support multiple CUDA architectures
2026-08-17 12:16:43 -07:00
Tzanio Kolev 775195b887 Merge pull request #5454 from mfem/umpire-cmake
Update Umpire CMake
2026-08-17 12:15:50 -07:00
Andrew Ho 89adf27a44 reduce max order since higher orders exceed the max dof/quad limits for HIP 2026-08-16 13:51:17 -07:00
Tzanio Kolev a7dbea190f Merge pull request #5435 from adamqc/fix-pncmesh-rebalance-attributes
Preserve element attributes during ParNCMesh rebalance
2026-08-15 13:35:42 -07:00
Tzanio Kolev 12e9b66eae Merge pull request #5412 from mfem/cuda-or-hip-in-c++-mode
Better support for using `mfem.hpp` in pure C++ sources when MFEM is built with CUDA or HIP
2026-08-15 13:27:42 -07:00
Tzanio Kolev 713edd670d Merge pull request #5399 from mfem/lor-mesh-connectivity
Support batched LOR assembly on highly connected meshes
2026-08-14 16:58:34 -07:00
Tzanio Kolev e2d6f5fb3b Merge pull request #5451 from mfem/shadow-warnings-take-2
Adjust default warnings
2026-08-14 16:58:16 -07:00
Tzanio Kolev 45b0e6e02c Merge pull request #5386 from mfem/curl_interp_pa
Curl Interpolator PA
2026-08-14 16:57:43 -07:00
Veselin Dobrev d37b7867ec In 'tuple.hpp':
* moved helper functions inside the namespace mfem::future::detail
* generalized functions using 'real_t' to any "scalar" type
* some formatting edits
2026-08-14 16:55:46 -07:00
Veselin Dobrev 73779b1de6 In the unit test 'test_tuple.cpp':
* fix for the case of debug + cuda/hip build
* add a gpu test for operator+ for tuples
2026-08-14 15:04:03 -07:00
Veselin DobrevandHugh Carson 8307a751db Apply suggestion from @hughcars
Co-authored-by: Hugh Carson <114775781+hughcars@users.noreply.github.com>
2026-08-13 17:15:12 -07:00
Andrew Ho 9f12aee475 review comments 2026-08-13 14:13:21 -07:00
Veselin Dobrev 366157036e Fix the test_tuple unit test for single-precision builds. 2026-08-13 14:10:19 -07:00
Andrew Ho 8afc1d1e36 Umpire also has moved to C++20 2026-08-13 14:03:03 -07:00
Veselin Dobrev 2bc734468d Fix the tuple unit test for serial build.
A few formatting edits.

Exclude the namespace mfem::future::detail from docs.
2026-08-13 13:34:03 -07:00
Tzanio Kolev c07c534f42 Merge pull request #5423 from adamqc/par-sesquilinear-device-diagonal-dev
Make complex system assembly device-safe
2026-08-13 07:27:54 -07:00
Will Pazner ccade73917 Move WARNING_FLAGS to the end of the file 2026-08-12 11:33:08 -07:00
Andrew Ho d169312edd Merge branch 'curl_interp_pa' into gpu_em 2026-08-12 10:15:28 -07:00
Andrew Ho 8812081cfc review comments 2026-08-12 10:14:16 -07:00
Andrew Ho b20051c06b Merge branch 'curl_interp_pa' into gpu_em 2026-08-12 08:45:25 -07:00
Andrew Ho d66068b754 fixed comment 2026-08-12 08:45:13 -07:00
Andrew Ho 362ca5b66d Merge branch 'curl_interp_pa' into gpu_em 2026-08-12 08:43:31 -07:00
Andrew Ho f9282b38f6 Make lor_ams produce a consistent gradient sign for RT as Curl
The sign shouldn't matter, but just for consistency
2026-08-12 08:41:42 -07:00
Tzanio Kolev aab2e1ebf8 Merge pull request #5337 from mfem/densetensor-move-fix
Add explicit move and copy operators to DenseTensor
2026-08-12 08:08:27 -07:00
Tzanio Kolev 8c2a8580b6 Merge pull request #5445 from mfem/macos-make-fix
Add a workaround for an issue with MacOS's default `make`
2026-08-12 08:08:04 -07:00
Veselin Dobrev 9141e85e15 Renamed an internal variable and an internal function. 2026-08-12 01:00:01 -07:00
Andrew Ho e57b63c660 Merge branch 'curl_interp_pa' into gpu_em 2026-08-11 21:30:06 -07:00
Andrew Ho 5d1958cfdf Remove rotated gradient 2026-08-11 21:23:53 -07:00
Andrew Ho 614a355c04 Merge branch 'curl_interp_pa' into gpu_em 2026-08-11 15:46:20 -07:00
Andrew Ho 79a88dfef5 undid change of removing ProjectGrad from 2D RT space quad and triangle elements
This is used by HypreAMS, unclear if it's ok to change HypreAMS to use
the CurlInterpolator instead of GradInterpolator for all possible edge
spaces.
2026-08-11 15:44:59 -07:00
Andrew Ho bd13f53db1 Merge branch 'curl_interp_pa' into gpu_em 2026-08-11 15:08:55 -07:00
Andrew Ho eb738baebe Also test that curl interpolator produces the right rotation 2026-08-11 14:44:41 -07:00
Will Pazner 46c5aed37b Use only explicit capture in DifferentiableOperator lambda 2026-08-11 14:37:29 -07:00
Will Pazner 6b8f53308f Make PEDANTIC_FLAG logic more robust 2026-08-11 14:32:12 -07:00
Will Pazner 1369d61457 Rename captured variable 2026-08-11 14:32:01 -07:00
Will Pazner e7d6b370dc Silence -Wshadow false positives on clang version < 17 2026-08-11 12:48:21 -07:00
Will Pazner ba07e91128 Enable -pedantic only for gcc and clang 2026-08-11 12:48:02 -07:00
Will Pazner ebdf68a1c3 Whitespace in defaults.mk 2026-08-11 12:47:44 -07:00
Andrew Ho 5f31928c2b Merge branch 'curl_interp_pa' into gpu_em 2026-08-11 12:33:50 -07:00
Andrew Ho 0cf5aca53e updated comment 2026-08-11 12:32:16 -07:00
Andrew Ho 2357771384 Merge branch 'curl_interp_pa' into gpu_em 2026-08-11 12:27:23 -07:00
Andrew Ho 7efeb617b1 changelog 2026-08-11 12:25:31 -07:00
Andrew Ho fe025de316 Fixed bug in FA ProjectCurl for 2D RT->H1
Added unit tests for 2D CurlInterpolator
2026-08-11 12:14:43 -07:00
Will Pazner 76b5f341cc Merge pull request #5443 from mfem/raja-cpp
Bump RAJA required C++ version in CMake
2026-08-11 11:06:25 -07:00
Ce Qin ea8468ea95 Merge remote-tracking branch 'origin/master' into par-sesquilinear-device-diagonal-dev
# Conflicts:
#	fem/complex_fem.cpp
2026-08-11 22:23:22 +08:00
Tzanio Kolev 2ea59935d8 Merge branch 'master' into fix-pncmesh-rebalance-attributes 2026-08-10 11:02:57 -07:00
Andrew Ho 4e5ebe6451 Merge branch 'master' into gpu_em 2026-08-10 09:58:41 -07:00
Andrew Ho 04f23f353c Merge branch 'master' into curl_interp_pa 2026-08-10 09:57:49 -07:00
Tzanio Kolev 0a76b8bfb2 Merge pull request #5434 from adamqc/fix-integrated-gll-bdr-projection
Fix IntegratedGLL projection for Nedelec segment elements
2026-08-10 09:11:24 -07:00
Tzanio Kolev 6116b49933 Merge branch 'master' into fix-integrated-gll-bdr-projection
Conflicts:
	tests/unit/fem/test_project_bdr.cpp
2026-08-10 09:09:05 -07:00
Tzanio Kolev 09f6023468 Merge pull request #5426 from mfem/najlkin/fix-getedgetrans
[BUG] Fixed projection on periodic NC meshes
2026-08-10 09:03:48 -07:00
Veselin Dobrev 610a8f9c0b Merge branch 'master' into gpu_em 2026-08-09 23:06:44 -07:00
Veselin Dobrev 8f01292a45 Merge branch 'master' into curl_interp_pa 2026-08-09 23:00:52 -07:00
Veselin Dobrev f7056be951 Merge pull request #5334 from mfem/hcurl_mass_pa
VectorFEMassIntegrator ApplyPA improvements
2026-08-09 22:54:07 -07:00
Veselin Dobrev 790848019e Add a workaround for an issue with MacOS's default 'make': when
running 'make all -j 12' two times in a row, the second run hangs.
2026-08-08 22:22:26 -07:00
Tzanio Kolev 01b146ab01 Merge pull request #5385 from mfem/specialization-tests
Extra Specialization Tests
2026-08-08 11:01:03 -07:00
Tzanio Kolev ae87b89f16 Merge branch 'master' into specialization-tests 2026-08-07 11:54:45 -07:00
Will Pazner 28e0f3569a Merge pull request #5351 from mfem/batched-lu-fix-5342
Batched LU failure handling consistency
2026-08-07 09:32:12 -07:00
Will Pazner 3c9ee8ff42 Change StaticAssertCudaOrHipLanguage to RequireCudaOrHipLanguage
Add constexpr default template parameter to simplify usage.
2026-08-07 09:01:00 -07:00
Andrew Ho f49b9a58e8 missed the 2d case 2026-08-07 08:41:28 -07:00
Ce Qin 8bfac662f4 Fix code-style 2026-08-07 19:58:50 +08:00
Andrew HoandJohn Camier 905696021a Update tests/unit/fem/specializations/test_qinterp_det.cpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-08-06 10:01:56 -07:00
Andrew HoandJohn Camier dc6e1ff4ea Update tests/unit/fem/specializations/test_qinterp_grad.cpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-08-06 10:01:47 -07:00
Ce Qin ed563f3090 Expand ParNCMesh rebalance attribute coverage 2026-08-06 21:47:11 +08:00
Ce Qin 51f2f5dd78 Simplify integrated ND shape evaluation and tests 2026-08-06 21:33:14 +08:00
Tzanio Kolev 4e828b9240 Merge pull request #5437 from mfem/grad-kernel-sm0-size
Grad kernel bug
2026-08-06 03:46:47 -07:00
Veselin Dobrev d627b19f06 Merge pull request #5297 from mfem/pncmesh-spacing
General NC mesh spacing in parallel
2026-08-05 14:35:52 -07:00
Jan Nikl 9b35464986 Added a tet unit test. 2026-08-05 13:27:48 -07:00
Andrew Ho f14a9bb53f Bump RAJA required C++ version in CMake 2026-08-05 13:09:02 -07:00
Jan Nikl 4eafaaa628 Added doxygen to GetTraceCollection(). 2026-08-05 11:35:44 -07:00
Jan Nikl bcab63b41c Added an assert for local edge index. 2026-08-05 11:25:48 -07:00
Jan Nikl 28c1f905b6 Removed non-L2 case from GetEdgeTransformation(). 2026-08-05 11:20:30 -07:00
Gabriele Bozzola 66dbe60cb1 Makefile: defer CUDA architecture flag selection 2026-08-05 07:05:36 -07:00
Andrew Ho f898d0bcde Merge branch 'hcurl_mass_pa' into curl_interp_pa 2026-08-05 07:01:19 -07:00
Gabriele Bozzola b8fcd640e5 Makefile: support multiple CUDA architectures
This PR changes the Makefile so that CUDA_ARCH can accept a
comma-separated list of compute capabilities (e.g.
CUDA_ARCH=sm_70,sm_80), mirroring the multi-architecture support the
CMake build already provides.
2026-08-05 01:26:56 -07:00
Jan Nikl c566a165b2 Added a check for 3D. 2026-08-04 14:44:57 -07:00
Andrew Ho 733d0bd177 Merge branch 'master' into hcurl_mass_pa 2026-08-04 14:05:34 -07:00
Andrew Ho 6e2bd88274 fallback version appears to still be faster for CPU for ND to ND 2026-08-04 14:00:38 -07:00
Andrew Ho 7a0a7bd1da style 2026-08-04 13:58:10 -07:00
Andrew Ho e59487bf14 move weak curl PA test to test_pa_coeff 2026-08-04 12:42:15 -07:00
Andrew Ho 647750ffa9 Merge remote-tracking branch 'origin/gpu_em' into gpu_em 2026-08-04 12:30:50 -07:00
Jan Nikl db42eb3255 Added edge to face table and used it for GetEdgeTransformation(). 2026-08-04 12:20:38 -07:00
Veselin Dobrev 1c19aba72a Merge pull request #5405 from mfem/jdongg/fix-gauss-jacobi-mpfr
Remove Gauss-Jacobi warning for missing MPFR implementation
2026-08-04 12:11:34 -07:00
Veselin Dobrev 1631ec67fa Merge pull request #5251 from Heinrich-BR/mixed-sesquilinear-dev
Mixed Sesquilinear Forms
2026-08-04 12:09:49 -07:00
Andrew Ho eceb502df3 Merge branch 'curl_interp_pa' into gpu_em 2026-08-04 11:54:55 -07:00
Andrew Ho bfdaf07a19 Merge branch 'hcurl_mass_pa' into curl_interp_pa 2026-08-04 11:50:49 -07:00
Andrew Ho d4b59fe357 appears to always be better to call the smem version even for CPU paths 2026-08-04 11:47:27 -07:00
Andrew Ho cdf077b560 Merge branch 'master' into hcurl_mass_pa 2026-08-04 11:06:42 -07:00
Andrew Ho 50ce940dee changelog 2026-08-04 11:04:54 -07:00
Jan Nikl d3307a6957 Added constexpr in the unit test. 2026-08-04 10:58:02 -07:00
Jan Nikl bde4cbbccd Changed parameters of the unit test. 2026-08-04 10:55:51 -07:00
Julian Andrej c8b64fef23 add tests and remove possible copy 2026-08-04 10:36:07 -07:00
Dylan Copeland 4c16395398 CHANGELOG 2026-08-04 09:05:44 -07:00
Jan Nikl cfb05a4a60 Added a unit test for boundary projection on nodal NC mesh. 2026-08-03 17:50:12 -07:00
Vlado Tomov LOFT e9cce62beb bug then d1d > q1d (happens in 3d for q1 remhos test) 2026-08-03 12:03:34 -07:00
Andrew Ho 7bb2d30100 Merge branch 'master' into batched-lu-fix-5342 2026-08-03 08:00:30 -07:00
Ce Qin 51a0058f65 Preserve element attributes during ParNCMesh rebalance 2026-08-01 23:09:30 +08:00
Ce Qin fd223f68b5 Fix integrated ND segment projection 2026-08-01 12:27:12 +08:00
Ce Qin eac57686c5 Rename the Hypre diagonal kernel 2026-07-31 09:29:34 +08:00
Andrew Ho ad962de425 fix merge 2026-07-30 10:18:23 -07:00
Andrew Ho c44c2f0cdf Merge branch 'master' into specialization-tests 2026-07-30 10:09:56 -07:00
John Camier 25a1c8f4a4 Merge branch 'master' into tuple-refactor 2026-07-30 10:02:33 -04:00
Ce Qin a60ba38833 Share complex operator construction 2026-07-30 14:13:03 +08:00
Ce Qin 2fa81463ae Share imaginary essential diagonal handling 2026-07-29 13:41:09 +08:00
Andrew Ho 0909dc634a Merge branch 'master' into specialization-tests 2026-07-28 11:23:20 -07:00
Tzanio Kolev dc995c4aa0 Merge branch 'master' into cuda-or-hip-in-c++-mode 2026-07-28 10:35:26 -07:00
Jan Nikl b9c960cc0d Removed separate edge transformation basis type. 2026-07-24 10:29:48 -07:00
Jan Nikl 0aa392a4ea Implemented GetEdgeTransformation for L2 elements. 2026-07-23 23:42:02 -07:00
John Camier ab6d0d9777 Merge branch 'master' into tuple-refactor 2026-07-23 13:28:51 -04:00
John Camier 3ce8b9e250 Merge branch 'master' into hcurl_mass_pa 2026-07-23 13:28:17 -04:00
Ce Qin ffa3d0789b Make complex system assembly device-safe 2026-07-23 22:57:05 +08:00
jdongg eaf91c9c08 Rebase onto master 2026-07-22 16:13:24 -07:00
jdongg 8330565463 Fix horizontal overflow of warning message 2026-07-22 16:00:15 -07:00
Justin Dong 6a6e9b5d6b Merge branch 'master' into jdongg/fix-gauss-jacobi-mpfr 2026-07-22 13:11:36 -07:00
Andrew Ho 609a9c0e3b Merge branch 'hcurl_mass_pa' into gpu_em 2026-07-21 10:23:34 -07:00
Andrew Ho 340fe85001 review comments 2026-07-20 20:46:47 -07:00
John Camier 4c6291f018 Merge branch 'master' into hcurl_mass_pa 2026-07-17 13:18:09 -07:00
“Henrique c6378788af Add guard fix to (Par)SesquilinearForm 2026-07-17 18:22:08 +01:00
Julian Andrej d6fffff08c remove unreachable macro 2026-07-17 08:24:10 -07:00
“Henrique d65409fdc0 Fix FormRectangularLinearSystem guard 2026-07-17 15:16:40 +01:00
“Henrique 8b14357249 Added MixedSesquilinearForm changes to CHANGELOG 2026-07-16 15:42:55 +01:00
“Henrique cc1c6daed3 Fix ParMixedSesquilinearForm::FormRectangularLinearSystem bug 2026-07-16 15:39:06 +01:00
Dylan Copeland 28b6c85b44 Using mt19937. 2026-07-15 16:01:46 -07:00
“Henrique 0a43f3ca1f Formatting fix 2026-07-15 21:07:46 +01:00
“Henrique 212edacfd1 Linting 2026-07-15 21:07:46 +01:00
“Henrique 45cd0db146 More review suggestions 2026-07-15 21:07:46 +01:00
Henrique BRandJan Nikl f0505ec6eb Apply suggestions from code review
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-07-15 21:07:46 +01:00
“Henrique d3fda1ed30 Comment fixes 2026-07-15 21:07:46 +01:00
“Henrique f63e95a7a1 Expand tests to cover more cases 2026-07-15 21:07:46 +01:00
“Henrique 24e63e6802 Fix serial indexing bug 2026-07-15 21:07:46 +01:00
“Henrique e8961b32ff Fix indexing issue 2026-07-15 21:07:46 +01:00
“Henrique 9c4fa75530 Remove redundant code 2026-07-15 21:07:46 +01:00
“Henrique b904dd0131 More review fixes 2026-07-15 21:07:45 +01:00
“Henrique 24652e2a36 Linting 2026-07-15 21:07:45 +01:00
“Henrique 529209bcf2 Review suggestions 2026-07-15 21:07:45 +01:00
“Henrique 501e37d105 Small fixes 2026-07-15 21:07:45 +01:00
“Henrique d48384f9f4 Linting 2026-07-15 21:07:45 +01:00
“Henrique 94a815d9c9 Added unit test 2026-07-15 21:07:45 +01:00
“Henrique 1a03792398 Add Update method 2026-07-15 21:07:45 +01:00
“Henrique 118e97772c Change Hypre_ParCSR to MFEM_SPARSEMAT 2026-07-15 21:07:45 +01:00
Henrique BRandSocratis Petrides f1138eae7a Apply suggestions from code review
Co-authored-by: Socratis Petrides <petrides1@llnl.gov>
2026-07-15 21:07:45 +01:00
“Henrique 449199525b Added (Par)MixedSesquilinearForms 2026-07-15 21:07:45 +01:00
John Camier 17ecabf915 Merge branch 'master' into tuple-refactor 2026-07-15 09:01:56 -07:00
John Camier de1a876e39 Merge branch 'master' into jdongg/fix-gauss-jacobi-mpfr 2026-07-15 07:38:48 -07:00
Dylan Copeland 538711c13f Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-07-14 15:36:04 -07:00
Dylan Copeland 412cc42685 Added new miniapps to doxygen html documentation. 2026-07-14 15:35:42 -07:00
Veselin Dobrev c8b1dcad70 Fix the non-GPU build 2026-07-14 05:35:20 -07:00
Veselin Dobrev fa006da71e Modifications allowing the use of 'mfem.hpp' in pure c++ source files when
the library is built with CUDA or HIP support.
2026-07-14 04:38:48 -07:00
Andrew Ho 1e5f9e4d6b Merge branch 'hcurl_mass_pa' into gpu_em 2026-07-13 18:50:46 -07:00
Andrew Ho b0cc0a9b8c fix specializations
this works for the default for linear, quadratic, and cubic meshes
2026-07-13 18:42:33 -07:00
Andrew Ho e839a5e8ab Merge remote-tracking branch 'origin/hcurl_mass_pa' into hcurl_mass_pa 2026-07-10 10:29:00 -07:00
Andrew Ho 9d40c8b40c Merge branch 'master' into hcurl_mass_pa 2026-07-10 10:28:35 -07:00
Andrew Ho 069c618def review comments 2026-07-10 10:27:05 -07:00
jdongg e6d5e98a06 Remove error message for missing Gauss-Jacobi MPFR implementation and use double precision. Add warning. 2026-07-09 13:42:07 -07:00
Dylan Copeland cfa3440178 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-07-09 13:19:50 -07:00
Andrew Ho ed9a29130f changelog 2026-07-09 11:31:59 -07:00
Andrew Ho 9a80c8cd14 Merge branch 'curl_interp_pa' into gpu_em 2026-07-09 11:26:33 -07:00
Andrew Ho 143d7bf31b changelog 2026-07-09 11:26:19 -07:00
Andrew Ho 8358ee93fa Extracted changes from gpu_em to for lower dimension CurlInterpolator 2026-07-09 11:24:11 -07:00
Andrew Ho eb38d6ecd8 Merge branch 'master' into curl_interp_pa 2026-07-09 11:16:22 -07:00
Andrew Ho 5f80fb1eb7 change to use override 2026-07-09 11:12:55 -07:00
Andrew Ho 52efc31130 style 2026-07-09 10:44:29 -07:00
Andrew Ho b7dc53af15 added patches from Kris to support out of plane 2D EM 2026-07-09 10:40:24 -07:00
Andrew Ho 2b5c0c6fe4 formatting 2026-07-08 19:02:49 -07:00
Andrew Ho 7b8af2b05f Merge branch 'master' into gpu_em 2026-07-08 16:20:32 -07:00
Andrew Ho 1433d4aec4 added checks for map type 2026-07-08 16:07:52 -07:00
Andrew Ho 74d1579371 changelog 2026-07-08 15:14:23 -07:00
Andrew Ho ea83267885 Added support for L2 Integral spaces to MixedScalarCurlIntegrator 2026-07-08 15:10:35 -07:00
Will Pazner 49201d41c3 Support batched LOR assembly on highly connected meshes
The same change was made for full assembly in PR #4646.
2026-07-08 12:24:53 -07:00
Tzanio Kolev a53c446dd7 Merge branch 'master' into hcurl_mass_pa 2026-07-07 13:31:12 -07:00
Andrew Ho 3c8c8c21a9 Merge branch 'master' into hcurl_mass_pa 2026-07-07 11:19:23 -07:00
John Camier 50d58159bd Merge branch 'master' into tuple-refactor 2026-07-03 19:04:42 +02:00
Andrew Ho bab4314cf3 Merge branch 'gpu-qinterp-integ' into gpu_em 2026-07-02 18:10:10 -07:00
Andrew Ho e59d1835c3 compiler warnings 2026-07-02 08:41:58 -07:00
Dylan Copeland fbb0e44dce Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-06-30 17:43:43 -07:00
Andrew Ho 3cdaebdcaa formatting 2026-06-30 14:55:23 -07:00
Andrew Ho 9e8a7c456f Added Kris's mixed dot product integrator PA 2026-06-30 14:41:32 -07:00
Andrew Ho b39719984a Merge branch 'curl_interp_pa' into gpu_em 2026-06-30 14:21:26 -07:00
Andrew Ho a95278fe72 Merge branch 'bugfix-project' into gpu_em 2026-06-30 14:20:49 -07:00
Andrew Ho f2f366efa2 Merge branch 'gpu-qinterp-integ' into gpu_em 2026-06-30 14:20:34 -07:00
LwhJesse cc585df285 Apply batched LU code style 2026-07-01 03:23:06 +08:00
LwhJesse c7f2950458 Address batched LU review feedback 2026-07-01 03:02:31 +08:00
Jesse Li 068b61eb3f Merge branch 'master' into batched-lu-fix-5342 2026-06-30 17:41:40 +08:00
LwhJesse 3419a50655 Make batched LU failure test robust on GPU 2026-06-30 17:24:25 +08:00
Andrew Ho f4ad8b8f92 formatting 2026-06-29 14:50:31 -07:00
Andrew Ho e04c90b678 thread assignment error 2026-06-29 14:46:08 -07:00
Andrew Ho abbfe7cf71 Merge branch 'hcurl_mass_pa' into curl_interp_pa 2026-06-29 14:09:12 -07:00
Andrew Ho 7d91917d7a missing paren wrapper 2026-06-29 14:08:47 -07:00
Andrew Ho 6c2a78d5bd extract curl interpolator and a few other misc fixes 2026-06-29 11:49:07 -07:00
Andrew Ho 82f03e136d Merge branch 'master' into hcurl_mass_pa 2026-06-29 11:08:31 -07:00
Andrew Ho 46a84f6417 Merge branch 'hcurl_domain_lf' into hcurl_mass_pa 2026-06-29 11:06:31 -07:00
Andrew Ho f6b333681f duplicate test names 2026-06-26 14:00:23 -07:00
Andrew Ho 644b4ef141 fix formatting 2026-06-26 13:22:42 -07:00
Andrew Ho 08d6dd777a add tests which mimic users adding their own specializations 2026-06-26 13:16:06 -07:00
Andrew Ho 879413e774 Merge branch 'gpu-qinterp-integ' into specialization-tests 2026-06-26 11:57:24 -07:00
John Camier 63627acf30 Merge branch 'master' into tuple-refactor 2026-06-25 07:46:45 +02:00
Jesse Li ce80de49d0 Merge branch 'master' into batched-lu-fix-5342 2026-06-18 20:09:31 +08:00
Andrew Ho 45a62e8bcd include a mesh with curvature, use MFEM_Approx instead of tol 2026-06-17 22:00:33 -07:00
Andrew Ho 4ee2e40d34 Add test comparing partial assembly vs. full assembly results 2026-06-17 13:12:58 -07:00
John Camier 86af0f883c Merge branch 'master' into tuple-refactor 2026-06-17 08:59:28 -07:00
LwhJesse c529d34eea Fix GPU BLAS helper build guard 2026-06-17 20:09:21 +08:00
LwhJesse 742d043ead Improve batched LU failure checks 2026-06-17 20:06:14 +08:00
Dylan Copeland a67c93d0b8 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-06-15 13:50:23 -07:00
Andrew Ho 43532923f7 missing parenthesis protection wrappers 2026-06-14 18:44:02 -07:00
Andrew Ho 9980f767f8 bugfixes for wrappers 2026-06-12 12:53:14 -07:00
Andrew Ho 3b89be0ec6 Added wrappers for simplifying mod/div usage in flattened 3D thread blocks 2026-06-12 09:09:46 -07:00
Andrew Ho 4c12e3815b Merge remote-tracking branch 'base/hcurl_mass_pa' into hcurl_mass_pa 2026-06-10 15:32:57 -07:00
Andrew Ho 44d2d0c75b Merge remote-tracking branch 'base/hcurl_domain_lf' into hcurl_mass_pa 2026-06-10 15:28:57 -07:00
John Camier dbb5fe2f0e Merge branch 'master' into tuple-refactor 2026-06-09 06:54:17 -07:00
Andrew Ho 620e49aea6 Merge branch 'master' into hcurl_mass_pa 2026-06-08 12:42:10 -07:00
Tzanio Kolev 94da954917 Merge branch 'master' into tuple-refactor 2026-06-05 16:01:05 -07:00
LwhJesse 006855bec2 Handle batched LU failures in GPU backends 2026-06-03 13:09:17 +08:00
Dylan Copeland 96f9456a7d Reformatting. 2026-05-31 17:38:46 -07:00
Tzanio Kolev 6ce18b2005 Merge branch 'master' into tuple-refactor 2026-05-27 09:24:12 -07:00
Dylan Copeland c26f1937a9 Minor fixes suggested by copilot. 2026-05-26 22:06:49 -07:00
Dylan Copeland 8431604228 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-05-26 21:28:49 -07:00
Julian Andrej c09b6d8a1d make style 2026-05-26 19:49:07 -07:00
Julian AndrejandCopilot Autofix powered by AI 19d9175833 replace tuple implementation with generic sized
Apply suggestions from code review

Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>

Add tuple include

Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>

use move instead of copy

properly do forwards
2026-05-26 17:41:13 -07:00
Will Pazner e01d5afadb Add move and copy operators to DenseTensor
The default-provided move and copy could cause a crash because the
internal Mk DenseMatrix may be dangling, and so it cannot be moved
or copied into.
2026-05-19 14:04:11 -07:00
Andrew Ho 24bc9d48a1 extracted vector fe mass integrator changes from gpu-maxwell 2026-05-18 11:23:00 -07:00
Dylan Copeland 27deb9cdd2 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-05-06 12:02:19 -07:00
Dylan Copeland a7b30bed56 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-04-21 19:20:03 -07:00
Dylan Copeland fd63847904 Merge branch 'master' of github.com:mfem/mfem into pncmesh-spacing 2026-04-13 20:09:16 -07:00
Dylan Copeland 5269fc2bf2 gitignore 2026-04-09 16:08:14 -07:00
Dylan Copeland 4036a7d0c2 Remove unused variable. 2026-04-09 15:22:56 -07:00
Dylan Copeland 8099ca947e General spacing for refinement of parallel NC meshes. Added a parallel miniapp, demonstrating 3:1 refinement. 2026-04-09 14:44:53 -07:00
98 changed files with 9543 additions and 2127 deletions
+1
View File
@@ -260,6 +260,7 @@ miniapps/meshing/polar-nc
miniapps/meshing/mesh-quality
miniapps/meshing/hpref
miniapps/meshing/phpref
miniapps/meshing/pref321
miniapps/meshing/mobius-strip.mesh
miniapps/meshing/klein-bottle.mesh
miniapps/meshing/toroid-*.mesh
+42
View File
@@ -46,8 +46,19 @@ Discretization improvements
- Extend FindPointsGSLIB to support surface meshes.
- Added support for complex-valued mixed bilinear forms via the new classes
MixedSesquilinearForm and ParMixedSesquilinearForm, mirroring the existing
SesquilinearForm classes. Rectangular complex operators are now also
handled correctly by ComplexSparseMatrix::GetSystemMatrix and
ComplexHypreParMatrix::GetSystemMatrix, which previously assumed equal
trial and test spaces.
Meshing improvements
--------------------
- Added support for nonuniform anisotropic mesh refinement on parallel quad/hex
meshes with arbitrary spacing in each direction. This enables in particular
3:1 refinement in parallel, as demonstrated in the new meshing miniapp pref321.
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
bounds on the determinant of the mesh transformation Jacobian.
@@ -68,6 +79,11 @@ Linear and nonlinear solvers
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
DPG miniapps).
- Added new class MultiVector: an array of Vectors of different sizes where each
Vector can be allocated independently. Also, added associated methods in class
Operator: MultMV, MultTransposeMV, and GetGradientMV, that use MultiVector
objects for input and/or output parameters. [PR #5249]
GPU computing
-------------
- Improved partial assembly for VectorDivergenceIntegrator with shared-memory
@@ -81,15 +97,35 @@ GPU computing
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
- Added partial assembly support for MixedScalarWeakGradientIntegrator.
- Added partial assembly support for MixedDotProductIntegrator.
- Added partial assembly support for MixedScalarCrossProductIntegrator.
- Added partial assembly support for MixedScalarWeakCrossProductIntegrator.
- Added support for device partial assembly CurlInterpolator.
This supports 2D and 3D variants:
2D H1 (out-of-plane) to RT (in-plane)
2D ND (in-plane) to Integral L2 (out-of-plane)
3D ND to RT
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
- Changed VectorFEMassIntegrator to use kernel specialization dispatch for
partial assembly.
- Added support for FiniteElement::MapType::INTEGRAL spaces to
QuadratureInterpolator.
- Added support for FiniteElement::MapType::INTEGRAL spaces to
MixedScalarCurlIntegrator.
New and updated examples and miniapps
-------------------------------------
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
@@ -104,6 +140,12 @@ Miscellaneous
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
method will return immediately if no sign flips are needed.
API changes
-----------
- Removed ProjectGrad from 2D RT elements. Users should use ProjectCurl instead.
This also fixes a bug where ProjectCurl was returning the negative curl,
identical to ProjectGrad.
Version 4.9, released on Dec 11, 2025
=====================================
+3 -12
View File
@@ -88,18 +88,9 @@ if (MFEM_USE_STRUMPACK OR MFEM_USE_MUMPS)
# Just needed to find the MPI_Fortran libraries to link with
set(XSDK_ENABLE_Fortran ON)
endif()
# Ginkgo requires C++17:
if ((MFEM_USE_GINKGO) AND ("${CMAKE_CXX_STANDARD}" LESS "17"))
set(CMAKE_CXX_STANDARD 17 CACHE STRING "C++ standard to use." FORCE)
# Google Benchmark, SUNDIALS, STRUMPACK, Tribol, RAJA and Umpire require C++14:
elseif ((MFEM_USE_BENCHMARK OR
MFEM_USE_SUNDIALS OR
MFEM_USE_STRUMPACK OR
MFEM_USE_TRIBOL OR
MFEM_USE_RAJA OR
MFEM_USE_UMPIRE) AND
("${CMAKE_CXX_STANDARD}" LESS "14"))
set(CMAKE_CXX_STANDARD 14 CACHE STRING "C++ standard to use." FORCE)
# RAJA requires C++20:
if ((MFEM_USE_UMPIRE OR MFEM_USE_RAJA) AND ("${CMAKE_CXX_STANDARD}" LESS "20"))
set(CMAKE_CXX_STANDARD 20 CACHE STRING "C++ standard to use." FORCE)
endif()
# Include xSDK default CMake file.
+33 -8
View File
@@ -28,11 +28,8 @@ MPICXX = mpicxx
BASE_FLAGS = -std=c++17
OPTIM_FLAGS = -O3 $(BASE_FLAGS)
# Shadow warnings for clang only; GCC's -Wshadow flags more.
SHADOW_WARNING_FLAG = $(if $(findstring clang,\
$(shell $(MFEM_HOST_CXX) --version 2>/dev/null)),-Wshadow,)
WARNING_FLAGS = -pedantic -Wall $(SHADOW_WARNING_FLAG)
# The variable WARNING_FLAGS depends on which compiler is used, and is defined
# later in this file.
DEBUG_FLAGS = $(strip -g $(addprefix $(XCOMPILER),$(WARNING_FLAGS)) $(BASE_FLAGS))
# Prefixes for passing flags to the compiler and linker when using CXX or MPICXX
@@ -52,6 +49,10 @@ SHARED = NO
#
# If you set MFEM_USE_ENZYME=YES, must use CUDA_CXX=clang++
CUDA_CXX = nvcc
# CUDA compute capability used during compilation, e.g. sm_60. Multiple
# architectures can be requested as a comma-separated list, e.g. sm_70,sm_80.
# A single value may also be one of the nvcc special values "all",
# "all-major", or "native".
CUDA_ARCH = sm_60
# Base CUDA install directory, only needed if building with clang+cuda:
# The default setting is:
@@ -60,11 +61,23 @@ CUDA_ARCH = sm_60
# 3. Use /usr/local/cuda
CUDA_DIR = $(or $(CUDA_HOME),$(patsubst %/,%,$(dir \
$(patsubst %/,%,$(dir $(shell command -v nvcc))))),/usr/local/cuda)
# Derive nvcc/clang architecture flags from CUDA_ARCH. A comma-separated list
# expands into one -gencode / --cuda-gpu-arch flag per architecture; otherwise
# use the -arch / --cuda-gpu-arch shorthand.
MFEM_COMMA := ,
CUDA_ARCH_NUMS = $(patsubst sm_%,%,$(subst $(MFEM_COMMA), ,$(CUDA_ARCH)))
NVCC_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
$(foreach arch,$(CUDA_ARCH_NUMS),\
-gencode arch=compute_$(arch)$(MFEM_COMMA)code=sm_$(arch)),\
-arch=$(CUDA_ARCH)))
CLANG_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
$(foreach arch,$(CUDA_ARCH_NUMS),--cuda-gpu-arch=sm_$(arch)),\
--cuda-gpu-arch=$(CUDA_ARCH)))
# flags for clang+cuda
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) --cuda-gpu-arch=$(CUDA_ARCH)
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) $(CLANG_ARCH_FLAGS)
# flags for nvcc
NVCC_FLAGS = -x=cu --expt-extended-lambda --expt-relaxed-constexpr \
-arch=$(CUDA_ARCH) -isystem "$(CUDA_DIR)/include"
$(NVCC_ARCH_FLAGS) -isystem "$(CUDA_DIR)/include"
# Prefixes for passing flags to the host compiler and linker when using
# CUDA_CXX=nvcc
CUDA_XCOMPILER = -Xcompiler=
@@ -382,7 +395,7 @@ CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
CUDSS_LIB = \
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
# The cuDSS communication and threading libraries.
# The cuDSS communication and threading libraries.
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
@@ -665,3 +678,15 @@ VERBOSE = NO
# Optional build tag
MFEM_BUILD_TAG = $(shell uname -snm)
# Enable -pedantic flag only for gcc or clang. nvcc complains with -pedantic
# because of line directives.
PEDANTIC_FLAG = $(if \
$(findstring NVIDIA,$(shell $(MFEM_CXX) --version 2>&1)),, \
$(if $(or \
$(findstring gcc version,$(shell $(MFEM_CXX) -v 2>&1)), \
$(findstring clang version,$(shell $(MFEM_CXX) -v 2>&1))),-pedantic,))
# Enable shadow warnings for clang only; GCC's -Wshadow flags more.
SHADOW_WARNING_FLAG = $(if $(findstring clang,\
$(shell $(MFEM_HOST_CXX) --version 2>/dev/null)),-Wshadow,)
WARNING_FLAGS = $(PEDANTIC_FLAG) -Wall $(SHADOW_WARNING_FLAG)
+2 -1
View File
@@ -1083,7 +1083,8 @@ EXCLUDE_PATTERNS =
# ANamespace::AClass, ANamespace::*Test
EXCLUDE_SYMBOLS = mfem::internal \
mfem::kernels::internal
mfem::kernels::internal \
mfem::future::detail
# The EXAMPLE_PATH tag can be used to specify one or more files or directories
# that contain example code fragments that are included (see the \include
+4
View File
@@ -201,6 +201,7 @@ namespace mfem {
* - <a class="el" href="nurbs__naca__cmesh_8cpp_source.html">NURBS NACA Mesher</a>: generate NURBS based mesh around a NACA foil
* - <a class="el" href="nurbs__printfunc_8cpp_source.html">NURBS Printer</a>: print the NURBS-basis
* - <a class="el" href="nurbs__mesh_info_8cpp_source.html">NURBS Mesh info</a>: print the info of a NURBS mesh
* - <a class="el" href="nurbs__surface_8cpp_source.html">NURBS Surface</a>: interpolate a 3D Surface in a NURBS Patch
*
* <H3>Miniapps</H3>
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
@@ -245,6 +246,9 @@ namespace mfem {
* - <a class="el" href="pdiffusion_8cpp_source.html">DPG Diffusion example</a>: DPG formulation for the diffusion problem
* - <a class="el" href="pmaxwell_8cpp_source.html">DPG Maxwell example</a>: DPG formulation for the indefinite Maxwell problem
* - <a class="el" href="lor__elast_8cpp_source.html">LOR Elasticity</a>: solve linear elasticity with LOR preconditioning on GPUs
* - <a class="el" href="reflector_8cpp_source.html">Reflector Miniapp</a>: reflect a mesh about a plane
* - <a class="el" href="ref321_8cpp_source.html">3:1 Refinement Miniapp</a>: perform 3:1 anisotropic mesh refinements
* - <a class="el" href="pref321_8cpp_source.html">3:1 Refinement Miniapp</a>: parallel 3:1 anisotropic mesh refinements
*
* See also the <a class="el" href="https://mfem.org/examples/">examples documentation</a> online.
*/
+25
View File
@@ -1255,6 +1255,31 @@ void BilinearForm::Mult(const Vector &x, Vector &y) const
}
}
void BilinearForm::AddMult(const Vector &x, Vector &y, const real_t a) const
{
if (ext)
{
ext->AddMult(x, y, a);
}
else
{
mat->AddMult(x, y, a);
}
}
void BilinearForm::AddMultTranspose(const Vector &x, Vector &y,
const real_t a) const
{
if (ext)
{
ext->AddMultTranspose(x, y, a);
}
else
{
mat->AddMultTranspose(x, y, a);
}
}
void BilinearForm::MultTranspose(const Vector & x, Vector & y) const
{
if (ext)
+3 -4
View File
@@ -307,8 +307,8 @@ public:
{ mat->Mult(x, y); mat_e->AddMult(x, y); }
/// Add the matrix vector multiple to a vector: $ y += a M x $
void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const override
{ mat -> AddMult (x, y, a); }
void AddMult(const Vector &x, Vector &y,
const real_t a = 1.0) const override;
/** @brief Add the original uneliminated matrix vector multiple to a vector.
The original matrix is $ M + Me $ so we have:
@@ -318,8 +318,7 @@ public:
/// Add the matrix transpose vector multiplication: $ y += a M^T x $
void AddMultTranspose(const Vector & x, Vector & y,
const real_t a = 1.0) const override
{ mat->AddMultTranspose(x, y, a); }
const real_t a = 1.0) const override;
/** @brief Add the original uneliminated matrix transpose vector
multiple to a vector. The original matrix is $ M + M_e $
+12 -2
View File
@@ -1997,7 +1997,11 @@ void PADiscreteLinearOperatorExtension::Assemble()
}
else
{
mfem_error("A real ElementRestriction is required in this setting!");
const L2ElementRestriction* l2_elem_restrict =
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
MFEM_VERIFY(l2_elem_restrict,
"A real ElementRestriction is required in this setting!");
test_multiplicity = 1.0;
}
auto tm = test_multiplicity.ReadWrite();
@@ -2036,7 +2040,13 @@ void PADiscreteLinearOperatorExtension::AddMult(
}
else
{
mfem_error("In this setting you need a real ElementRestriction!");
const L2ElementRestriction* l2_elem_restrict =
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
MFEM_VERIFY(l2_elem_restrict,
"In this setting you need a real ElementRestriction!");
tempY.SetSize(y.Size());
l2_elem_restrict->MultTranspose(localTest, tempY);
y += tempY;
}
}
+464 -331
View File
File diff suppressed because it is too large Load Diff
+961 -138
View File
File diff suppressed because it is too large Load Diff
+352
View File
@@ -392,6 +392,9 @@ private:
bool RealInteg();
bool ImagInteg();
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
OperatorHandle &A) const;
public:
SesquilinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
@@ -505,6 +508,186 @@ public:
virtual ~SesquilinearForm();
};
/** Class for a mixed sesquilinear form
A mixed sesquilinear form is a generalization of a mixed bilinear form to
complex-valued fields. Mixed sesquilinear forms are linear in the second
argument but the first argument involves a complex conjugate in the sense
that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the MixedSesquilinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class MixedSesquilinearForm
{
private:
ComplexOperator::Convention conv;
MixedBilinearForm * mblfr;
MixedBilinearForm * mblfi;
/* These methods check if the real/imag parts of the sesqulinear form are not
empty */
bool RealInteg();
bool ImagInteg();
public:
MixedSesquilinearForm(
FiniteElementSpace * trial_fes,
FiniteElementSpace * test_fes,
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
/** @brief Create a MixedSesquilinearForm on the given trial and test
FiniteElementSpaces, using the same integrators as the
MixedBilinearForms @a bfr and @a bfi.
The FiniteElementSpace pointers are not owned by the newly constructed
object.
The integrators are copied as pointers and they are not owned by the
newly constructed MixedSesquilinearForm. */
MixedSesquilinearForm(
FiniteElementSpace * trial_fes,
FiniteElementSpace * test_fes,
MixedBilinearForm * bfr,
MixedBilinearForm * bfi,
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::LEGACY (default)
- AssemblyLevel::FULL
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
mblfr->SetAssemblyLevel(assembly_level);
mblfi->SetAssemblyLevel(assembly_level);
}
MixedBilinearForm & real() { return *mblfr; }
MixedBilinearForm & imag() { return *mblfi; }
const MixedBilinearForm & real() const { return *mblfr; }
const MixedBilinearForm & imag() const { return *mblfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds new Domain Integrator, restricted to specific attributes.
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & elem_marker);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/** @brief Adds new boundary Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The mfem::array @a bdr_marker is stored internally as a pointer to the given
mfem::Array<int> object. */
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The mfem::array @a bdr_marker is stored internally as a pointer to the given
mfem::Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & bdr_marker);
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
This type of integrator assembles terms over all faces of the mesh using
the face FE from the trial space and the two adjacent volume FEs from
the test space. */
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Updates the internal mixed forms with the new finite element space.
virtual void Update();
/** @brief Return a ComplexSparseMatrix wrapping the local (L-dof) real
and imaginary matrices of the form.
The returned wrapper has to be deleted by the caller, but it does not
own the wrapped real and imaginary matrices, which remain owned by
this form. */
ComplexSparseMatrix *AssembleComplexSparseMatrix();
/// Return the trial FE space associated with the MixedSesquilinearForm.
FiniteElementSpace *TrialFESpace() { return mblfr->TrialFESpace(); }
/// Read-only access to the associated trial FiniteElementSpace.
const FiniteElementSpace *TrialFESpace() const { return mblfr->TrialFESpace(); }
/// Return the test FE space associated with the MixedSesquilinearForm.
FiniteElementSpace *TestFESpace() { return mblfr->TestFESpace(); }
/// Read-only access to the associated test FiniteElementSpace.
const FiniteElementSpace *TestFESpace() const { return mblfr->TestFESpace(); }
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
const Array<int> & ess_test_tdof_list,
Vector & x,
Vector & b,
OperatorHandle & A,
Vector & X,
Vector & B);
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
const Array<int> & ess_test_tdof_list,
OperatorHandle & A);
virtual ~MixedSesquilinearForm();
};
#ifdef MFEM_USE_MPI
/// Class for parallel complex-valued grid function - real + imaginary part
@@ -806,6 +989,12 @@ private:
bool RealInteg();
bool ImagInteg();
void SetImaginaryEssentialDiagonalToZero(
const Array<int> &ess_tdof_list, OperatorHandle &A);
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
OperatorHandle &A) const;
public:
ParSesquilinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
@@ -921,6 +1110,169 @@ public:
virtual ~ParSesquilinearForm();
};
/** Class for a parallel mixed sesquilinear form
A mixed sesquilinear form is a generalization of a mixed bilinear form to
complex-valued fields. Mixed sesquilinear forms are linear in the second
argument but the first argument involves a complex conjugate in the sense
that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ParMixedSesquilinearForm either the real
or imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class ParMixedSesquilinearForm
{
private:
ComplexOperator::Convention conv;
ParMixedBilinearForm * pmblfr;
ParMixedBilinearForm * pmblfi;
/* These methods check if the real/imag parts of the sesqulinear form are
not empty */
bool RealInteg();
bool ImagInteg();
public:
ParMixedSesquilinearForm(
ParFiniteElementSpace * trial_fes,
ParFiniteElementSpace * test_fes,
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
/** @brief Create a ParMixedSesquilinearForm on the given trial and test
ParFiniteElementSpaces, using the same integrators as the
ParMixedBilinearForms @a pbfr and @a pbfi.
The ParFiniteElementSpace pointers are not owned by the newly
constructed object.
The integrators are copied as pointers and they are not owned by the
newly constructed ParMixedSesquilinearForm. */
ParMixedSesquilinearForm(
ParFiniteElementSpace * trial_fes,
ParFiniteElementSpace * test_fes,
ParMixedBilinearForm * pbfr,
ParMixedBilinearForm * pbfi,
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::LEGACY (default)
- AssemblyLevel::FULL
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
pmblfr->SetAssemblyLevel(assembly_level);
pmblfi->SetAssemblyLevel(assembly_level);
}
ParMixedBilinearForm & real() { return *pmblfr; }
ParMixedBilinearForm & imag() { return *pmblfi; }
const ParMixedBilinearForm & real() const { return *pmblfr; }
const ParMixedBilinearForm & imag() const { return *pmblfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds new Domain Integrator, restricted to specific attributes.
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & elem_marker);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/** @brief Adds new boundary Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The mfem::array @a bdr_marker is stored internally as a pointer to the given
mfem::Array<int> object. */
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The mfem::array @a bdr_marker is stored internally as a pointer to the given
mfem::Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> & bdr_marker);
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
This type of integrator assembles terms over all faces of the mesh using
the face FE from the trial space and the two adjacent volume FEs from
the test space. */
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag);
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
BilinearFormIntegrator * bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Updates the internal mixed forms with the new finite element space.
virtual void Update();
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
/** The returned matrix has to be deleted by the caller. */
ComplexHypreParMatrix * ParallelAssemble();
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
const Array<int> & ess_test_tdof_list,
Vector & x,
Vector & b,
OperatorHandle & A,
Vector & X,
Vector & B);
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
const Array<int> & ess_test_tdof_list,
OperatorHandle & A);
virtual ~ParMixedSesquilinearForm();
};
#endif // MFEM_USE_MPI
}
+48
View File
@@ -51,4 +51,52 @@ DifferentiableOperator::DifferentiableOperator(
}
}
void FDJacobian::Mult(const Vector &v, Vector &y) const
{
// See [1] for choice of eps.
//
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
// finite difference matrix-vector products in Newton-Krylov solvers for
// implicit climate dynamics with spectral elements. Procedia Computer
// Science, 51, pp.2036-2045.
real_t eps;
if (fixed_eps > 0.0)
{
eps = fixed_eps;
}
else
{
const real_t vnorm_local = v.Norml2();
real_t vnorm;
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
MPI_COMM_WORLD);
eps = lambda * (lambda + xnorm / vnorm);
}
// x + eps * v
{
const auto d_v = v.Read();
const auto d_x = x.Read();
auto d_xpev = xpev.Write();
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_xpev[i] = d_x[i] + eps * d_v[i];
});
}
// y = f(x + eps * v)
op.Mult(xpev, y);
// y = (f(x + eps * v) - f(x)) / eps
{
const auto d_f = f.Read();
auto d_y = y.ReadWrite();
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_y[i] = (d_y[i] - d_f[i]) / eps;
});
}
}
#endif // MFEM_USE_MPI
+23 -22
View File
@@ -697,17 +697,18 @@ void DifferentiableOperator::AddIntegrator(
// The explicit captures are necessary to avoid dependency on
// the specific instance of this class (this pointer).
restriction_callback =
[=, solutions = this->solutions, parameters = this->parameters]
(std::vector<Vector> &sol,
const std::vector<Vector> &par,
std::vector<Vector> &f)
restriction_callback = [element_dof_ordering,
solutions_ = this->solutions,
parameters_ = this->parameters]
(std::vector<Vector> &sol,
const std::vector<Vector> &par,
std::vector<Vector> &f)
{
restriction<entity_t>(solutions, sol, f,
restriction<entity_t>(solutions_, sol, f,
element_dof_ordering);
restriction<entity_t>(parameters, par, f,
restriction<entity_t>(parameters_, par, f,
element_dof_ordering,
solutions.size());
solutions_.size());
};
prolongation_transpose = get_prolongation_transpose(
@@ -835,19 +836,19 @@ void DifferentiableOperator::AddIntegrator(
// capture by ref:
&restriction_cb = this->restriction_callback,
&fields_e = this->fields_e,
&residual_e = this->residual_e,
&output_restriction_transpose = this->output_restriction_transpose
&fields_e_ = this->fields_e,
&residual_e_ = this->residual_e,
&output_restriction_transpose_ = this->output_restriction_transpose
]
(std::vector<Vector> &sol, const std::vector<Vector> &par, Vector &res)
mutable // mutable: needed to modify 'shmem_cache'
{
restriction_cb(sol, par, fields_e);
restriction_cb(sol, par, fields_e_);
residual_e = 0.0;
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof, num_entities);
residual_e_ = 0.0;
auto ye = Reshape(residual_e_.ReadWrite(), test_vdim, num_test_dof, num_entities);
auto wrapped_fields_e = wrap_fields(fields_e,
auto wrapped_fields_e = wrap_fields(fields_e_,
action_shmem_info.field_sizes,
num_entities);
@@ -878,7 +879,7 @@ void DifferentiableOperator::AddIntegrator(
y, fhat, output_fop, output_dtq_shmem[0],
scratch_shmem, dimension, use_sum_factorization);
}, num_entities, thread_blocks, action_shmem_info.total_size, shmem_cache.ReadWrite());
output_restriction_transpose(residual_e, res);
output_restriction_transpose_(residual_e_, res);
});
// Without this compile-time check, some valid instantiations of this method
@@ -1193,7 +1194,7 @@ void DifferentiableOperator::AddIntegrator(
// capture by ref:
&qpdc_mem = derivative_qp_caches_ref,
&fields = fields_ref
&fields_ = fields_ref
](std::vector<Vector> &f_e, SparseMatrix *&A) mutable
{
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
@@ -1241,14 +1242,14 @@ void DifferentiableOperator::AddIntegrator(
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
trial_field = &fields_[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
(&fields_[output_to_field[0]].data);
A = new SparseMatrix(test_fes->GetVSize(), trial_fes->GetVSize());
@@ -1334,7 +1335,7 @@ void DifferentiableOperator::AddIntegrator(
input_to_field,
output_to_field,
&spmatcb = assemble_derivative_sparsematrix_callbacks_ref,
&fields = fields_ref
&fields_ = fields_ref
](std::vector<Vector> &f_e, HypreParMatrix *&A) mutable
{
SparseMatrix *spmat = nullptr;
@@ -1366,14 +1367,14 @@ void DifferentiableOperator::AddIntegrator(
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
trial_field = &fields_[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
(&fields_[output_to_field[0]].data);
if (same_test_and_trial)
{
+742 -768
View File
File diff suppressed because it is too large Load Diff
+9 -52
View File
@@ -597,7 +597,7 @@ struct ThreadBlocks
int z = 1;
};
#if defined(MFEM_USE_CUDA_OR_HIP)
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
template <typename func_t>
__global__ void forall_kernel_shmem(func_t f, int n)
{
@@ -617,10 +617,11 @@ void forall(func_t f,
int num_shmem = 0,
real_t *shmem = nullptr)
{
if (Device::Allows(Backend::CUDA_MASK) ||
Device::Allows(Backend::HIP_MASK))
internal::RequireKernelCompilation();
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK))
{
#if defined(MFEM_USE_CUDA_OR_HIP)
// int gridsize = (N + Z - 1) / Z;
int num_bytes = num_shmem * sizeof(decltype(shmem));
dim3 block_size(blocks.x, blocks.y, blocks.z);
@@ -631,9 +632,10 @@ void forall(func_t f,
MFEM_GPU_CHECK(hipGetLastError());
#endif
MFEM_DEVICE_SYNC;
#endif
return;
}
else if (Device::Allows(Backend::CPU_MASK))
#endif
if (Device::Allows(Backend::CPU_MASK))
{
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
"Backend::CPU needs a pre-allocated shared memory block");
@@ -671,52 +673,7 @@ public:
MPI_COMM_WORLD);
}
void Mult(const Vector &v, Vector &y) const override
{
// See [1] for choice of eps.
//
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
// finite difference matrix-vector products in Newton-Krylov solvers for
// implicit climate dynamics with spectral elements. Procedia Computer
// Science, 51, pp.2036-2045.
real_t eps;
if (fixed_eps > 0.0)
{
eps = fixed_eps;
}
else
{
const real_t vnorm_local = v.Norml2();
real_t vnorm;
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
MPI_COMM_WORLD);
eps = lambda * (lambda + xnorm / vnorm);
}
// x + eps * v
{
const auto d_v = v.Read();
const auto d_x = x.Read();
auto d_xpev = xpev.Write();
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_xpev[i] = d_x[i] + eps * d_v[i];
});
}
// y = f(x + eps * v)
op.Mult(xpev, y);
// y = (f(x + eps * v) - f(x)) / eps
{
const auto d_f = f.Read();
auto d_y = y.ReadWrite();
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_y[i] = (d_y[i] - d_f[i]) / eps;
});
}
}
void Mult(const Vector &v, Vector &y) const override;
virtual MemoryClass GetMemoryClass() const override
{
+6 -5
View File
@@ -1316,13 +1316,14 @@ void VectorFiniteElement::Project_RT(
}
}
void VectorFiniteElement::ProjectGrad_RT(
void VectorFiniteElement::ProjectCurl2D_RT(
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
ElementTransformation &Trans, DenseMatrix &grad) const
{
// 2D "ProjectCurl_RT"
if (dim != 2)
{
mfem_error("VectorFiniteElement::ProjectGrad_RT works only in 2D!");
mfem_error("VectorFiniteElement::ProjectCurl2D_RT works only in 2D!");
}
DenseMatrix dshape(fe.GetDof(), fe.GetDim());
@@ -1333,8 +1334,8 @@ void VectorFiniteElement::ProjectGrad_RT(
for (int k = 0; k < dof; k++)
{
fe.CalcDShape(Nodes.IntPoint(k), dshape);
tk[0] = nk[d2n[k]*dim+1];
tk[1] = -nk[d2n[k]*dim];
tk[0] = -nk[d2n[k]*dim+1];
tk[1] = nk[d2n[k]*dim];
dshape.Mult(tk, grad_k);
for (int j = 0; j < grad_k.Size(); j++)
{
@@ -1381,7 +1382,7 @@ void VectorFiniteElement::ProjectCurl_ND(
}
}
void VectorFiniteElement::ProjectCurl_RT(
void VectorFiniteElement::ProjectCurl3D_RT(
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
ElementTransformation &Trans, DenseMatrix &curl) const
{
+10 -7
View File
@@ -957,10 +957,11 @@ protected:
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &I) const;
// rotated gradient in 2D
void ProjectGrad_RT(const real_t *nk, const Array<int> &d2n,
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &grad) const;
// Input is a scalar representing the Z (out of plane) component, Output is
// the X-Y (in-plane) RT curl
void ProjectCurl2D_RT(const real_t *nk, const Array<int> &d2n,
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &grad) const;
// Compute the curl as a discrete operator from ND FE (fe) to ND FE (this).
// The natural FE for the range is RT, so this is an approximation.
@@ -968,9 +969,9 @@ protected:
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &curl) const;
void ProjectCurl_RT(const real_t *nk, const Array<int> &d2n,
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &curl) const;
void ProjectCurl3D_RT(const real_t *nk, const Array<int> &d2n,
const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &curl) const;
/** @brief Project a vector coefficient onto the ND basis functions
@param tk Edge tangent vectors for this element type
@@ -1446,6 +1447,8 @@ public:
dof2quad_array_open);
}
const Poly_1D::Basis &GetOpenBasis1D() const { return obasis1d; }
virtual ~VectorTensorFiniteElement();
};
+38 -1
View File
@@ -1282,12 +1282,49 @@ ND_SegmentElement::ND_SegmentElement(const int p, const int ob_type)
}
}
void ND_SegmentElement::CalcShape(const IntegrationPoint &ip,
Vector &shape) const
{
if (obasis1d.IsIntegratedType()) { obasis1d.ScaleIntegrated(false); }
obasis1d.Eval(ip.x, shape);
}
void ND_SegmentElement::CalcVShape(const IntegrationPoint &ip,
DenseMatrix &shape) const
{
Vector vshape(shape.Data(), dof);
obasis1d.Eval(ip.x, vshape);
CalcShape(ip, vshape);
}
void ND_SegmentElement::ProjectIntegrated(VectorCoefficient &vc,
ElementTransformation &Trans,
Vector &dofs) const
{
MFEM_ASSERT(obasis1d.IsIntegratedType(), "Not integrated type");
real_t vk[Geometry::MaxDim];
Vector xk(vk, vc.GetVDim());
const real_t *cp = poly1d.ClosedPoints(dof, BasisType::GaussLobatto);
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, dof);
IntegrationPoint ip;
for (int i = 0; i < dof; i++)
{
const real_t h = cp[i+1] - cp[i];
real_t val = 0.0;
for (int q = 0; q < ir.GetNPoints(); q++)
{
const IntegrationPoint &ip1d = ir.IntPoint(q);
ip.x = cp[i] + h*ip1d.x;
Trans.SetIntPoint(&ip);
vc.Eval(xk, Trans, ip);
val += ip1d.weight*Trans.Jacobian().InnerProduct(tk, vk);
}
dofs(i) = val*h;
}
}
const real_t ND_WedgeElement::tk[15] =
+10 -3
View File
@@ -303,8 +303,7 @@ public:
/** @brief Construct the ND_SegmentElement of order @a p and open
BasisType @a ob_type */
ND_SegmentElement(const int p, const int ob_type = BasisType::GaussLegendre);
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override
{ obasis1d.Eval(ip.x, shape); }
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override;
void CalcVShape(const IntegrationPoint &ip,
DenseMatrix &shape) const override;
void CalcVShape(ElementTransformation &Trans,
@@ -325,7 +324,10 @@ public:
using FiniteElement::Project;
void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const override
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
{
if (obasis1d.IsIntegratedType()) { ProjectIntegrated(vc, Trans, dofs); }
else { Project_ND(tk, dof2tk, vc, Trans, dofs); }
}
void ProjectMatrixCoefficient(MatrixCoefficient &mc,
ElementTransformation &T,
Vector &dofs) const override
@@ -338,6 +340,11 @@ public:
ElementTransformation &Trans,
DenseMatrix &grad) const override
{ ProjectGrad_ND(tk, dof2tk, fe, Trans, grad); }
protected:
void ProjectIntegrated(VectorCoefficient &vc,
ElementTransformation &Trans,
Vector &dofs) const;
};
class ND_WedgeElement : public VectorFiniteElement
+6 -16
View File
@@ -73,16 +73,11 @@ public:
void Project(const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &I) const override
{ Project_RT(nk, dof2nk, fe, Trans, I); }
// Gradient + rotation = Curl: H1 -> H(div)
void ProjectGrad(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &grad) const override
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
// Curl = Gradient + rotation: H1 -> H(div)
void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const override
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
void GetFaceMap(const int face_id, Array<int> &face_map) const override;
@@ -148,7 +143,7 @@ public:
void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const override
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
/// @brief Return the mapping from lexicographically ordered face DOFs to
/// lexicographically ordered element DOFs corresponding to local face
@@ -210,16 +205,11 @@ public:
void Project(const FiniteElement &fe, ElementTransformation &Trans,
DenseMatrix &I) const override
{ Project_RT(nk, dof2nk, fe, Trans, I); }
// Gradient + rotation = Curl: H1 -> H(div)
void ProjectGrad(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &grad) const override
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
// Curl = Gradient + rotation: H1 -> H(div)
void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const override
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
};
@@ -274,7 +264,7 @@ public:
void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const override
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
};
class RT_WedgeElement : public VectorFiniteElement
@@ -332,7 +322,7 @@ public:
void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const override
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
};
/** Arbitrary order H(Div) basis functions defined on pyramid-shaped elements
@@ -428,7 +418,7 @@ public:
virtual void ProjectCurl(const FiniteElement &fe,
ElementTransformation &Trans,
DenseMatrix &curl) const
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
void CalcRawVShape(const IntegrationPoint &ip,
DenseMatrix &shape) const;
+4
View File
@@ -100,6 +100,10 @@ public:
return FiniteElementForGeometry(GeomType);
}
/** @brief Returns a collection of the trace elements.
@note The collection is owned by the caller and is NOT deleted in the
destructor. */
virtual FiniteElementCollection *GetTraceCollection() const;
virtual ~FiniteElementCollection();
+4 -4
View File
@@ -556,7 +556,7 @@ void obboxsurf_calc_3(Vector &bb,
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, n);
lag(I0, work, n, 1, 0);
for (int ie = 0; ie < nel; ie++,x+=n2,y+=n2,z+=n2)
for (int ie = 0; (unsigned)ie < nel; ie++,x+=n2,y+=n2,z+=n2)
{
struct gslib::dbl_range ab[3];
struct gslib::dbl_range tb[3];
@@ -780,7 +780,7 @@ void obboxedge_calc_2(Vector &bb,
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, nr);
lag(I0r, work, nr,1, 0);
for (int ie = 0; ie < nel; ie++,x+=nr,y+=nr)
for (int ie = 0; (unsigned)ie < nel; ie++,x+=nr,y+=nr)
{
double x0[2], A[4];
struct gslib::dbl_range ab[2], tb[2];
@@ -892,7 +892,7 @@ void obboxedge_calc_3(Vector &bb,
gslib::lagrange_fun *const lag = gslib::gll_lag_setup(work, nr);
lag(I0r, work, nr, 1, 0);
for (int ie = 0; ie < nel; ie++,x+=nr,y+=nr,z+=nr)
for (int ie = 0; (unsigned)ie < nel; ie++,x+=nr,y+=nr,z+=nr)
{
double x0[3], A[9], Ai[9];
struct gslib::dbl_range ab[3], tb[3];
@@ -4518,7 +4518,7 @@ Mesh* FindPointsGSLIB::GetBoundingBoxMesh(int type)
int eidx = 0;
if (myid == save_rank)
{
for (int p = 0; p < gsl_comm->np; p++)
for (int p = 0; (unsigned)p < gsl_comm->np; p++)
{
if (static_cast<unsigned int>(p) != save_rank)
{
+9
View File
@@ -368,6 +368,8 @@ void HybridizationExtension::ConstructH()
CAhatInvCt = 0.0;
// Fill the face-to-face adjacency array. Two faces are adjacent if they are
// incident to a common element.
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int fi)
{
const int begin_f = d_face_face_offsets[fi];
@@ -403,6 +405,12 @@ void HybridizationExtension::ConstructH()
}
}
}
// Fill unused entries with -1 to indicate invalid
const int end_f = d_face_face_offsets[fi + 1];
for (int i = begin_f + idx; i < end_f; ++i)
{
d_face_to_face[i] = -1;
}
});
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int fi)
@@ -412,6 +420,7 @@ void HybridizationExtension::ConstructH()
for (int idx_j = begin; idx_j < end; ++idx_j)
{
const int fj = d_face_to_face[idx_j];
if (fj < 0) { break; }
for (int ei = 0; ei < 2; ++ei)
{
const int e = d_face_to_el(0, ei, fi);
+2
View File
@@ -178,6 +178,8 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
MFEM_VERIFY(el.GetMapType() == FiniteElement::VALUE,
"Only value map type currently supported");
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Trans);
if (DeviceCanUseCeed())
+35 -22
View File
@@ -147,18 +147,16 @@ void PAHcurlMassAssembleDiagonal3D(const int D1D,
}); // end of element loop
}
void PAHcurlMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &bo,
const Array<real_t> &bc,
const Array<real_t> &bot,
const Array<real_t> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y)
void PAHcurlMassApply2D(const int NE, const bool symmetric,
[[maybe_unused]] const bool scalar_coeff,
const Array<real_t> &bo, const Array<real_t> &bc,
const Array<real_t> &bot, const Array<real_t> &bct,
const Vector &pa_data, const Vector &x, Vector &y,
const int D1D, [[maybe_unused]] const int TestD1D,
const int Q1D)
{
MFEM_ASSERT(D1D == TestD1D,
"Trial and Test space must have the same number of dofs");
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(bc.Read(), Q1D, D1D);
auto Bot = Reshape(bot.Read(), D1D-1, Q1D);
@@ -277,18 +275,16 @@ void PAHcurlMassApply2D(const int D1D,
}); // end of element loop
}
void PAHcurlMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &bo,
const Array<real_t> &bc,
const Array<real_t> &bot,
const Array<real_t> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y)
void PAHcurlMassApply3D(const int NE, const bool symmetric,
[[maybe_unused]] const bool scalar_coeff,
const Array<real_t> &bo, const Array<real_t> &bc,
const Array<real_t> &bot, const Array<real_t> &bct,
const Vector &pa_data, const Vector &x, Vector &y,
const int D1D, [[maybe_unused]] const int TestD1D,
const int Q1D)
{
MFEM_VERIFY(D1D == TestD1D,
"Trial and test spaces must have same number of dofs");
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Error: D1D > MAX_D1D");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
@@ -789,6 +785,23 @@ void PAHcurlL2Setup2D(const int Q1D,
});
}
void PAHcurlL2IntSetup2D(const int Q1D, const int NE, const Array<real_t> &w,
Vector &coeff, const Vector &detJ, Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto C = Reshape(coeff.Read(), NQ, NE);
auto J = Reshape(detJ.Read(), NQ, NE);
auto y = Reshape(op.Write(), NQ, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < NQ; ++q)
{
y(q,e) = W[q] * C(q,e) / J(q,e);
}
});
}
void PAHcurlL2Setup3D(const int NQ,
const int coeffDim,
const int NE,
+278 -190
View File
@@ -181,228 +181,312 @@ inline void SmemPAHcurlMassAssembleDiagonal3D(const int d1d,
}
// PA H(curl) Mass Apply 2D kernel
void PAHcurlMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &bo,
const Array<real_t> &bc,
const Array<real_t> &bot,
const Array<real_t> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
void PAHcurlMassApply2D(const int NE, const bool symmetric,
const bool scalar_coeff, const Array<real_t> &bo,
const Array<real_t> &bc, const Array<real_t> &bot,
const Array<real_t> &bct, const Vector &pa_data,
const Vector &x, Vector &y, const int TrialD1D,
const int TestD1D, const int Q1D);
// PA H(curl) Mass Apply 3D kernel
void PAHcurlMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &bo,
const Array<real_t> &bc,
const Array<real_t> &bot,
const Array<real_t> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
void PAHcurlMassApply3D(const int NE, const bool symmetric,
[[maybe_unused]] const bool scalar_coeff,
const Array<real_t> &bo, const Array<real_t> &bc,
const Array<real_t> &bot, const Array<real_t> &bct,
const Vector &pa_data, const Vector &x, Vector &y,
const int TrialD1D, [[maybe_unused]] const int TestD1D,
const int Q1D);
// Shared memory PA H(curl) Mass Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAHcurlMassApply3D(const int d1d,
const int q1d,
const int NE,
const bool symmetric,
const Array<real_t> &bo,
const Array<real_t> &bc,
const Array<real_t> &bot,
const Array<real_t> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y)
template <int T_D1D = 0, int T_Q1D = 0, int TBATCH = 0, bool ACCUMULATE = true>
inline void SmemPAHcurlMassApply3D(
const int NE, const bool symmetric, [[maybe_unused]] const bool scalar_coeff,
const Array<real_t> &bo, const Array<real_t> &bc,
[[maybe_unused]] const Array<real_t> &bot,
[[maybe_unused]] const Array<real_t> &bct, const Vector &pa_data,
const Vector &x, Vector &y, const int d1d = 0,
[[maybe_unused]] const int test_d1d = 0, const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(T_D1D || d1d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Error: d1d > HCURL_MAX_D1D");
MFEM_VERIFY(T_Q1D || q1d <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
"Error: q1d > HCURL_MAX_Q1D");
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_ASSERT(Q1D >= D1D, "Expected Q1D >= D1D");
const int dataSize = symmetric ? 6 : 9;
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(bc.Read(), Q1D, D1D);
auto op = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, dataSize, NE);
auto X = Reshape(x.Read(), 3*(D1D-1)*D1D*D1D, NE);
auto Y = Reshape(y.ReadWrite(), 3*(D1D-1)*D1D*D1D, NE);
// assume trial space == test space
auto Bo = bo.Read();
auto Bc = bc.Read();
auto op =
Reshape(pa_data.Read(), Q1D, Q1D, Q1D, dataSize, NE);
auto X_ = Reshape(x.Read(), 3 * (D1D - 1) * D1D * D1D, NE);
auto y_ = y.ReadWrite();
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
constexpr int MD_ = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
constexpr int MQ_ = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
constexpr int MDQ_ = std::max(MD_, MQ_);
constexpr int MB_ = TBATCH ? TBATCH : 1;
mfem::forall_2D_batch<MDQ_ * MDQ_ * MDQ_ * MB_>(
NE, MDQ_ * MDQ_ * MDQ_, 1, MB_, [=] MFEM_HOST_DEVICE(int e)
{
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
constexpr int nbz = TBATCH ? TBATCH : 1;
int tidz = MFEM_THREAD_ID(z);
#else
constexpr int nbz = 1;
constexpr int tidz = 0;
#endif
constexpr int VDIM = 3;
constexpr int MD1D = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
constexpr int MQ1D = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MDQ = std::max(MD1D, MQ1D);
MFEM_SHARED real_t sBo[MQ1D][MD1D];
MFEM_SHARED real_t sBc[MQ1D][MD1D];
// nvcc limit work-around: can't have Y_ be captured first in
// if constexpr, so capture y_ and construct Y_ locally
// only works on GPU
auto Y = Reshape(y_, VDIM * (D1D - 1) * D1D * D1D, NE);
real_t op9[9];
MFEM_SHARED real_t sop[9*MQ1D*MQ1D];
MFEM_SHARED real_t mass[MQ1D][MQ1D][3];
MFEM_SHARED real_t sBo[MDQ * (MD1D - 1)];
MFEM_SHARED real_t sBc[MDQ * MD1D];
auto BO = Reshape(sBo, Q1D, D1D - 1);
auto BC = Reshape(sBc, Q1D, D1D);
MFEM_SHARED real_t sX[MD1D][MD1D][MD1D];
MFEM_SHARED real_t sX[nbz * VDIM * (MD1D - 1) * MD1D * MD1D];
MFEM_SHARED real_t sm0[nbz * VDIM * MDQ * MDQ * MDQ];
MFEM_SHARED real_t sm1[nbz * VDIM * MDQ * MDQ * MDQ];
MFEM_FOREACH_THREAD(qx,x,Q1D)
real_t(*X)[nbz][(MD1D - 1) * MD1D * MD1D] =
(real_t(*)[nbz][(MD1D - 1) * MD1D * MD1D])(sX);
// shapes of buffers always use MQ1D to mitigate shared memory bank
// conflicts
real_t(*DDQ)[nbz][MQ1D][MQ1D][MQ1D] =
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
real_t(*DQQ)[nbz][MQ1D][MQ1D][MQ1D] =
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm1);
real_t(*QQQ)[nbz][MQ1D][MQ1D][MQ1D] =
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
real_t(*QQD)[nbz][MQ1D][MQ1D][MQ1D] =
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm1);
real_t(*QDD)[nbz][MQ1D][MQ1D][MQ1D] =
(real_t(*)[nbz][MQ1D][MQ1D][MQ1D])(sm0);
// load dofs into smem
const int offset = (D1D - 1) * D1D * D1D;
MFEM_FOREACH_THREAD_DIRECT(ix, x, offset)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
for (int dim = 0; dim < VDIM; ++dim)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
for (int i=0; i<dataSize; ++i)
{
op9[i] = op(qx,qy,qz,i,e);
}
}
X[dim][tidz][ix] = X_(ix + dim * offset, e);
}
}
const int tidx = MFEM_THREAD_ID(x);
const int tidy = MFEM_THREAD_ID(y);
const int tidz = MFEM_THREAD_ID(z);
// load basis functions data
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(ix, x, D1D * Q1D) { sBc[ix] = Bc[ix]; }
MFEM_FOREACH_THREAD_DIRECT(ix, x, (D1D - 1) * Q1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
sBo[ix] = Bo[ix];
}
}
for (int dim0 = 0; dim0 < VDIM; ++dim0)
{
MFEM_SYNC_THREAD;
// sum factor to QQQ = Q_{dim0,dim1} B X_{dim1}
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, Q1D, D1Dy, D1Dz,
Q1D, Q1D, Q1D)
{
sBc[q][d] = Bc(q,d);
if (d < D1D-1)
real_t u = 0;
for (int dx = 0; dx < D1Dx; ++dx)
{
sBo[q][d] = Bo(q,d);
real_t b;
if (dim1 == 0)
{
b = BO(qx, dx);
}
else
{
b = BC(qx, dx);
}
u += X[dim1][tidz][dx + (dy + dz * D1Dy) * D1Dx] * b;
}
DDQ[dim1][tidz][dz][dy][qx] = u;
}
}
MFEM_SYNC_THREAD;
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
// const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, Q1D, Q1D, D1Dz,
Q1D, Q1D, Q1D)
{
real_t u = 0;
for (int dy = 0; dy < D1Dy; ++dy)
{
real_t b;
if (dim1 == 1)
{
b = BO(qy, dy);
}
else
{
b = BC(qy, dy);
}
u += DDQ[dim1][tidz][dz][dy][qx] * b;
}
DQQ[dim1][tidz][dz][qy][qx] = u;
}
}
MFEM_SYNC_THREAD;
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
const int D1Dz = (dim1 == 2) ? D1D - 1 : D1D;
// const int D1Dy = (dim1 == 1) ? D1D - 1 : D1D;
// const int D1Dx = (dim1 == 0) ? D1D - 1 : D1D;
MFEM_FOREACH_THREAD_DIRECT_3D(qx, qy, qz, x, Q1D, Q1D, Q1D)
{
real_t u = 0;
for (int dz = 0; dz < D1Dz; ++dz)
{
real_t b;
if (dim1 == 2)
{
b = BO(qz, dz);
}
else
{
b = BC(qz, dz);
}
u += DQQ[dim1][tidz][dz][qy][qx] * b;
}
// pa_data is row major
int idx;
if (symmetric)
{
int row;
int col;
if (dim0 > dim1)
{
row = dim1;
col = dim0;
}
else
{
row = dim0;
col = dim1;
}
idx = col + VDIM * row - row * (row + 1) / 2;
}
else
{
idx = dim0 * VDIM + dim1;
}
QQQ[dim1][tidz][qz][qy][qx] = op(qx, qy, qz, idx, e) * u;
}
}
MFEM_SYNC_THREAD;
// sum factor back to Y
// Assume bot and bct == bo^t and bc^t respectively (i.e. test ==
// trial functions), skip loading them again.
{
const int D1Dz = (dim0 == 2) ? D1D - 1 : D1D;
const int D1Dy = (dim0 == 1) ? D1D - 1 : D1D;
const int D1Dx = (dim0 == 0) ? D1D - 1 : D1D;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, D1Dz, Q1D, Q1D,
Q1D, Q1D, Q1D)
{
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
real_t u = 0;
for (int qz = 0; qz < Q1D; ++qz)
{
real_t b = 0;
if (dim0 == 2)
{
b = BO(qz, dz);
}
else
{
b = BC(qz, dz);
}
u += QQQ[dim1][tidz][qz][qy][qx] * b;
}
QQD[dim1][tidz][qy][qx][dz] = u;
}
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, D1Dy, D1Dz, Q1D,
Q1D, Q1D, Q1D)
{
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
real_t u = 0;
for (int qy = 0; qy < Q1D; ++qy)
{
real_t b;
if (dim0 == 1)
{
b = BO(qy, dy);
}
else
{
b = BC(qy, dy);
}
u += QQD[dim1][tidz][qy][qx][dz] * b;
}
QDD[dim1][tidz][qx][dz][dy] = u;
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT_3D(dx, dy, dz, x, D1Dx, D1Dy, D1Dz)
{
int ix = dx + D1Dx * (dy + D1Dy * dz);
real_t u = 0;
for (int qx = 0; qx < Q1D; ++qx)
{
real_t b;
if (dim0 == 0)
{
b = BO(qx, dx);
}
else
{
b = BC(qx, dx);
}
for (int dim1 = 0; dim1 < VDIM; ++dim1)
{
u += QDD[dim1][tidz][qx][dz][dy] * b;
}
}
if constexpr (ACCUMULATE)
{
Y(ix + dim0 * offset, e) += u;
}
else
{
Y(ix + dim0 * offset, e) = u;
}
}
}
}
MFEM_SYNC_THREAD;
for (int qz=0; qz < Q1D; ++qz)
{
int osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y, z components
{
const int D1Dz = (c == 2) ? D1D - 1 : D1D;
const int D1Dy = (c == 1) ? D1D - 1 : D1D;
const int D1Dx = (c == 0) ? D1D - 1 : D1D;
MFEM_FOREACH_THREAD(dz,z,D1Dz)
{
MFEM_FOREACH_THREAD(dy,y,D1Dy)
{
MFEM_FOREACH_THREAD(dx,x,D1Dx)
{
sX[dz][dy][dx] = X(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e);
}
}
}
MFEM_SYNC_THREAD;
if (tidz == qz)
{
for (int i=0; i<dataSize; ++i)
{
sop[i + (dataSize*tidx) + (dataSize*Q1D*tidy)] = op9[i];
}
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u = 0.0;
for (int dz = 0; dz < D1Dz; ++dz)
{
const real_t wz = (c == 2) ? sBo[qz][dz] : sBc[qz][dz];
for (int dy = 0; dy < D1Dy; ++dy)
{
const real_t wy = (c == 1) ? sBo[qy][dy] : sBc[qy][dy];
for (int dx = 0; dx < D1Dx; ++dx)
{
const real_t t = sX[dz][dy][dx];
const real_t wx = (c == 0) ? sBo[qx][dx] : sBc[qx][dx];
u += t * wx * wy * wz;
}
}
}
mass[qy][qx][c] = u;
} // qx
} // qy
} // tidz == qz
osc += D1Dx * D1Dy * D1Dz;
MFEM_SYNC_THREAD;
} // c
MFEM_SYNC_THREAD; // Sync mass[qy][qx][d] and sop
osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y, z components
{
const int D1Dz = (c == 2) ? D1D - 1 : D1D;
const int D1Dy = (c == 1) ? D1D - 1 : D1D;
const int D1Dx = (c == 0) ? D1D - 1 : D1D;
real_t dxyz = 0.0;
MFEM_FOREACH_THREAD(dz,z,D1Dz)
{
const real_t wz = (c == 2) ? sBo[qz][dz] : sBc[qz][dz];
MFEM_FOREACH_THREAD(dy,y,D1Dy)
{
MFEM_FOREACH_THREAD(dx,x,D1Dx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t wy = (c == 1) ? sBo[qy][dy] : sBc[qy][dy];
for (int qx = 0; qx < Q1D; ++qx)
{
const int os = (dataSize*qx) + (dataSize*Q1D*qy);
const int id1 = os + ((c == 0) ? 0 : ((c == 1) ? (symmetric ? 1 : 3) :
(symmetric ? 2 : 6))); // O11, O21, O31
const int id2 = os + ((c == 0) ? 1 : ((c == 1) ? (symmetric ? 3 : 4) :
(symmetric ? 4 : 7))); // O12, O22, O32
const int id3 = os + ((c == 0) ? 2 : ((c == 1) ? (symmetric ? 4 : 5) :
(symmetric ? 5 : 8))); // O13, O23, O33
const real_t m_c = (sop[id1] * mass[qy][qx][0]) + (sop[id2] * mass[qy][qx][1]) +
(sop[id3] * mass[qy][qx][2]);
const real_t wx = (c == 0) ? sBo[qx][dx] : sBc[qx][dx];
dxyz += m_c * wx * wy * wz;
}
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1Dz)
{
MFEM_FOREACH_THREAD(dy,y,D1Dy)
{
MFEM_FOREACH_THREAD(dx,x,D1Dx)
{
Y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e) += dxyz;
}
}
}
osc += D1Dx * D1Dy * D1Dz;
} // c loop
} // qz
}); // end of element loop
}
@@ -1805,13 +1889,17 @@ inline void SmemPACurlCurlApply3D(const int d1d,
ForallWrap<3>(true, NE, device_kernel, host_kernel, Q1D, Q1D, Q1D);
}
// PA H(curl)-L2 Assemble 2D kernel
// PA H(curl)-L2 value Assemble 2D kernel
void PAHcurlL2Setup2D(const int Q1D,
const int NE,
const Array<real_t> &w,
Vector &coeff,
Vector &op);
// PA H(curl)-L2 integral Assemble 2D kernel
void PAHcurlL2IntSetup2D(const int Q1D, const int NE, const Array<real_t> &w,
Vector &coeff, const Vector &detJ, Vector &op);
// PA H(curl)-L2 Assemble 3D kernel
void PAHcurlL2Setup3D(const int NQ,
const int coeffDim,
+696
View File
@@ -62,6 +62,30 @@ void PAHcurlHdivMassApply2D(const int D1D,
const Vector &x_,
Vector &y_);
/// H(curl) test, H(div) trial
inline void
PAHcurlHdivMassApply2D(const int NE, const bool, const bool scalarCoeff,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int D1Dtest, const int Q1D)
{
return PAHcurlHdivMassApply2D(D1D, D1Dtest, Q1D, NE, scalarCoeff, false,
false, Bo_, Bc_, Bot_, Bct_, op_, x_, y_);
}
/// H(div) test, H(curl) trial
inline void
PAHdivHcurlMassApply2D(const int NE, const bool, const bool scalarCoeff,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int D1Dtest, const int Q1D)
{
return PAHcurlHdivMassApply2D(D1D, D1Dtest, Q1D, NE, scalarCoeff, true,
false, Bo_, Bc_, Bot_, Bct_, op_, x_, y_);
}
// PA H(curl)-H(div) Mass Apply 3D kernel
void PAHcurlHdivMassApply3D(const int D1D,
const int D1Dtest,
@@ -78,6 +102,30 @@ void PAHcurlHdivMassApply3D(const int D1D,
const Vector &x_,
Vector &y_);
/// H(curl) test, H(div) trial
inline void
PAHcurlHdivMassApply3D(const int NE, const bool, const bool scalarCoeff,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int D1Dtest, const int Q1D)
{
PAHcurlHdivMassApply3D(D1D, D1Dtest, Q1D, NE, scalarCoeff, false, false, Bo_,
Bc_, Bot_, Bct_, op_, x_, y_);
}
/// H(div) test, H(curl) trial
inline void
PAHdivHcurlMassApply3D(const int NE, const bool, const bool scalarCoeff,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int D1Dtest, const int Q1D)
{
PAHcurlHdivMassApply3D(D1D, D1Dtest, Q1D, NE, scalarCoeff, true, false, Bo_,
Bc_, Bot_, Bct_, op_, x_, y_);
}
// PA H(curl)-H(div) Curl Apply 3D kernel
template<int T_D1D = 0, int T_D1D_TEST = 0, int T_Q1D = 0>
inline void PAHcurlHdivApply3D(const int d1d,
@@ -816,8 +864,656 @@ inline void PAHcurlHdivApplyTranspose3D(const int d1d,
}); // end of element loop
}
namespace curlinterp
{
constexpr int NBZ3D(int ndof_o, int nquad_o, int mdq)
{
if (ndof_o <= 0 || nquad_o <= 0)
{
return 1;
}
int ndof_c = ndof_o + 1;
int nquad_c = nquad_o + 1;
// z dimension is capped at 64 on nvidia and amd gpus
int tmp =
std::min((128 + mdq * mdq * (mdq - 1) - 1) / (mdq * mdq * (mdq - 1)), 64);
int smem_req =
sizeof(mfem::real_t) *
((3 * ndof_c * ndof_c * ndof_o + 2 * 2 * mdq * mdq * mdq) * tmp +
ndof_c * nquad_o + ndof_c * nquad_c + ndof_o * nquad_o);
// assume GPU has at least 48k shared memory
return std::max(std::min(tmp, (48 * 1024 + smem_req - 1) / smem_req), 1);
}
}
template <int T_NDOF_O, int T_NQUAD_O>
void CurlInterpolatorApply3DSmem(const int ne, const int ndof_o,
const int nquad_o, const Vector &pa,
const Vector &x_, Vector &y_)
{
constexpr int mnd_o = T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
constexpr int mnq_o =
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
constexpr int mndq = std::max(mnd_o + 1, mnq_o + 1);
constexpr int tbatch = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, mndq);
MFEM_VERIFY(ndof_o <= mnd_o, "Error: H(curl) order larger than supported");
MFEM_VERIFY(nquad_o <= mnq_o, "Error: H(div) order larger than supported");
int mnq = std::max(ndof_o + 1, nquad_o + 1);
auto pa_data = pa.Read();
auto x_d = x_.Read();
auto y_d = y_.ReadWrite();
mfem::forall_2D_batch<mndq * mndq * (mndq - 1) * tbatch>(
ne, mnq * mnq * (mnq - 1), 1, tbatch, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MND_O =
T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
constexpr int MNQ_O =
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
constexpr int MNDQ = std::max(MND_O + 1, MNQ_O + 1);
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
constexpr int nbz = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, MNDQ);
int tidz = MFEM_THREAD_ID(z);
// Make mnq a local variable since capturing would result in different
// captures between host/device versions, and spuriously fails
int mnq = std::max(ndof_o + 1, nquad_o + 1);
#else
constexpr int nbz = 1;
constexpr int tidz = 0;
#endif
const int NDOF_O = T_NDOF_O ? T_NDOF_O : ndof_o;
const int NQUAD_O = T_NQUAD_O ? T_NQUAD_O : nquad_o;
const int NDOF_C = NDOF_O + 1;
const int NQUAD_C = NQUAD_O + 1;
MFEM_SHARED real_t
sBG[(MND_O + 1) * MNQ_O + (MND_O + 1) * (MNQ_O + 1) + MND_O * MNQ_O];
auto X_ = Reshape(x_d, 3 * NDOF_C * NDOF_C * NDOF_O, ne);
auto Y = Reshape(y_d, 3 * NQUAD_C * NQUAD_O * NQUAD_O, ne);
auto Gco = Reshape(sBG, NQUAD_O, NDOF_C);
auto Bcc = Reshape(sBG + NDOF_C * NQUAD_O, NQUAD_C, NDOF_C);
auto Boo =
Reshape(sBG + NDOF_C * NQUAD_O + NDOF_C * NQUAD_C, NQUAD_O, NDOF_O);
MFEM_SHARED real_t X[3][nbz][MND_O * (MND_O + 1) * (MND_O + 1)];
MFEM_SHARED real_t sm0[nbz * 2 * MNDQ * MNDQ * MNDQ];
MFEM_SHARED real_t sm1[nbz * 2 * MNDQ * MNDQ * MNDQ];
// shapes of buffers always use MNDQ to mitigate shared memory bank
// conflicts
real_t(*DDQ)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
real_t(*DQQ)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm1);
real_t(*QQQ)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
const int offset = NDOF_O * NDOF_C * NDOF_C;
const int offsetq = NQUAD_C * NQUAD_O * NQUAD_O;
MFEM_FOREACH_THREAD_DIRECT(ix, x, offset)
{
for (int dim = 0; dim < 3; ++dim)
{
X[dim][tidz][ix] = X_(ix + dim * offset, e);
}
}
// load basis functions data
if (tidz == 0)
{
auto npts = NDOF_C * NQUAD_O + NDOF_C * NQUAD_C + NDOF_O * NQUAD_O;
MFEM_FOREACH_THREAD(ix, x, npts) { sBG[ix] = pa_data[ix]; }
}
MFEM_SYNC_THREAD;
// x: Vz Bcc Gco Boo - Vy Bcc Boo Gco
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_C, NDOF_C,
NDOF_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_C; ++dx)
{
u += X[2][tidz][dx + (dy + dz * NDOF_C) * NDOF_C] * Bcc(qx, dx);
}
DDQ[0][tidz][dz][dy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_C, NDOF_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_C; ++dx)
{
u += X[1][tidz][dx + (dy + dz * NDOF_O) * NDOF_C] * Bcc(qx, dx);
}
DDQ[1][tidz][dz][dy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_C, NQUAD_O,
NDOF_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_C; ++dy)
{
u += DDQ[0][tidz][dz][dy][qx] * Gco(qy, dy);
}
DQQ[0][tidz][dz][qy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_C, NQUAD_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_O; ++dy)
{
u += DDQ[1][tidz][dz][dy][qx] * Boo(qy, dy);
}
DQQ[1][tidz][dz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_C, NQUAD_O,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_O; ++dz)
{
u += DQQ[0][tidz][dz][qy][qx] * Boo(qz, dz);
}
QQQ[0][tidz][qz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_C, NQUAD_O,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_C; ++dz)
{
u += DQQ[1][tidz][dz][qy][qx] * Gco(qz, dz);
}
Y(qx + (qy + qz * NQUAD_O) * NQUAD_C, e) =
QQQ[0][tidz][qz][qy][qx] - u;
}
MFEM_SYNC_THREAD;
// y: Vx Boo Bcc Gco - Vz Gco Bcc Boo
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
NDOF_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_O; ++dx)
{
u += X[0][tidz][dx + (dy + dz * NDOF_C) * NDOF_O] * Boo(qx, dx);
}
DDQ[0][tidz][dz][dy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
NDOF_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_C; ++dx)
{
u += X[2][tidz][dx + (dy + dz * NDOF_C) * NDOF_C] * Gco(qx, dx);
}
DDQ[1][tidz][dz][dy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_C,
NDOF_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_C; ++dy)
{
u += DDQ[0][tidz][dz][dy][qx] * Bcc(qy, dy);
}
DQQ[0][tidz][dz][qy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_C,
NDOF_O, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_C; ++dy)
{
u += DDQ[1][tidz][dz][dy][qx] * Bcc(qy, dy);
}
DQQ[1][tidz][dz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_C; ++dz)
{
u += DQQ[0][tidz][dz][qy][qx] * Gco(qz, dz);
}
QQQ[0][tidz][qz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_O; ++dz)
{
u += DQQ[1][tidz][dz][qy][qx] * Boo(qz, dz);
}
Y(qx + (qy + qz * NQUAD_C) * NQUAD_O + offsetq, e) =
QQQ[0][tidz][qz][qy][qx] - u;
}
MFEM_SYNC_THREAD;
// z: Vy Gco Boo Bcc - Vx Boo Gco Bcc
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_C; ++dx)
{
u += X[1][tidz][dx + (dy + dz * NDOF_O) * NDOF_C] * Gco(qx, dx);
}
DDQ[0][tidz][dz][dy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, dy, dz, x, NQUAD_O, NDOF_C,
NDOF_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int dx = 0; dx < NDOF_O; ++dx)
{
u += X[0][tidz][dx + (dy + dz * NDOF_C) * NDOF_O] * Boo(qx, dx);
}
DDQ[1][tidz][dz][dy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_O; ++dy)
{
u += DDQ[0][tidz][dz][dy][qx] * Boo(qy, dy);
}
DQQ[0][tidz][dz][qy][qx] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, dz, x, NQUAD_O, NQUAD_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dy = 0; dy < NDOF_C; ++dy)
{
u += DDQ[1][tidz][dz][dy][qx] * Gco(qy, dy);
}
DQQ[1][tidz][dz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_O,
NQUAD_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_C; ++dz)
{
u += DQQ[0][tidz][dz][qy][qx] * Bcc(qz, dz);
}
QQQ[0][tidz][qz][qy][qx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(qx, qy, qz, x, NQUAD_O, NQUAD_O,
NQUAD_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int dz = 0; dz < NDOF_C; ++dz)
{
u += DQQ[1][tidz][dz][qy][qx] * Bcc(qz, dz);
}
Y(qx + (qy + qz * NQUAD_O) * NQUAD_O + 2 * offsetq, e) =
QQQ[0][tidz][qz][qy][qx] - u;
}
MFEM_SYNC_THREAD;
});
}
template <int T_NDOF_O, int T_NQUAD_O>
void CurlInterpolatorTApply3DSmem(const int ne, const int ndof_o,
const int nquad_o, const Vector &pa,
const Vector &x_, Vector &y_)
{
constexpr int mnd_o = T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
constexpr int mnq_o =
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
constexpr int mndq = std::max(mnd_o + 1, mnq_o + 1);
constexpr int tbatch = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, mndq);
MFEM_VERIFY(ndof_o <= mnd_o, "Error: H(curl) order larger than supported");
MFEM_VERIFY(nquad_o <= mnq_o, "Error: H(div) order larger than supported");
int mnq = std::max(ndof_o + 1, nquad_o + 1);
auto pa_data = pa.Read();
auto x_d = x_.Read();
auto y_d = y_.ReadWrite();
mfem::forall_2D_batch<mndq * mndq * (mndq - 1) * tbatch>(
ne, mnq * mnq * (mnq - 1), 1, tbatch, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MND_O =
T_NDOF_O ? T_NDOF_O : DofQuadLimits::HCURL_MAX_D1D - 1;
constexpr int MNQ_O =
T_NQUAD_O ? T_NQUAD_O : DofQuadLimits::HDIV_MAX_D1D - 1;
constexpr int MNDQ = std::max(MND_O + 1, MNQ_O + 1);
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
constexpr int nbz = curlinterp::NBZ3D(T_NDOF_O, T_NQUAD_O, MNDQ);
int tidz = MFEM_THREAD_ID(z);
// Make mnq a local variable since capturing would result in different
// captures between host/device versions, and spuriously fails
int mnq = std::max(ndof_o + 1, nquad_o + 1);
#else
constexpr int nbz = 1;
constexpr int tidz = 0;
#endif
const int NDOF_O = T_NDOF_O ? T_NDOF_O : ndof_o;
const int NQUAD_O = T_NQUAD_O ? T_NQUAD_O : nquad_o;
const int NDOF_C = NDOF_O + 1;
const int NQUAD_C = NQUAD_O + 1;
MFEM_SHARED real_t
sBG[(MND_O + 1) * MNQ_O + (MND_O + 1) * (MNQ_O + 1) + MND_O * MNQ_O];
auto X_ = Reshape(x_d, 3 * NQUAD_C * NQUAD_O * NQUAD_O, ne);
auto Y = Reshape(y_d, 3 * NDOF_C * NDOF_C * NDOF_O, ne);
auto Gco = Reshape(sBG, NQUAD_O, NDOF_C);
auto Bcc = Reshape(sBG + NDOF_C * NQUAD_O, NQUAD_C, NDOF_C);
auto Boo =
Reshape(sBG + NDOF_C * NQUAD_O + NDOF_C * NQUAD_C, NQUAD_O, NDOF_O);
MFEM_SHARED real_t X[3][nbz][MNQ_O * MNQ_O * (MNQ_O + 1)];
MFEM_SHARED real_t sm0[nbz * 2 * MNDQ * MNDQ * MNDQ];
MFEM_SHARED real_t sm1[nbz * 2 * MNDQ * MNDQ * MNDQ];
// shapes of buffers always use MNDQ to mitigate shared memory bank
// conflicts
real_t(*QQD)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
real_t(*QDD)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm1);
real_t(*DDD)[nbz][MNDQ][MNDQ][MNDQ] =
(real_t(*)[nbz][MNDQ][MNDQ][MNDQ])(sm0);
const int offset = NDOF_O * NDOF_C * NDOF_C;
const int offsetq = NQUAD_C * NQUAD_O * NQUAD_O;
MFEM_FOREACH_THREAD_DIRECT(ix, x, offsetq)
{
for (int dim = 0; dim < 3; ++dim)
{
X[dim][tidz][ix] = X_(ix + dim * offsetq, e);
}
}
// load basis functions data
if (tidz == 0)
{
auto npts = NDOF_C * NQUAD_O + NDOF_C * NQUAD_C + NDOF_O * NQUAD_O;
MFEM_FOREACH_THREAD(ix, x, npts) { sBG[ix] = pa_data[ix]; }
}
MFEM_SYNC_THREAD;
// x: Vy Boo Bcc Gco - Vz Boo Gco Bcc
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
NQUAD_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_O; ++qz)
{
u += X[1][tidz][qx + (qy + qz * NQUAD_C) * NQUAD_O] * Gco(qz, dz);
}
QQD[0][tidz][qy][qx][dz] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_C; ++qz)
{
u += X[2][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_O] * Bcc(qz, dz);
}
QQD[1][tidz][qy][qx][dz] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_C; ++qy)
{
u += QQD[0][tidz][qy][qx][dz] * Bcc(qy, dy);
}
QDD[0][tidz][qx][dz][dy] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_O; ++qy)
{
u += QQD[1][tidz][qy][qx][dz] * Gco(qy, dy);
}
QDD[1][tidz][qx][dz][dy] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_O, NDOF_C,
NDOF_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_O; ++qx)
{
u += QDD[0][tidz][qx][dz][dy] * Boo(qx, dx);
}
DDD[0][tidz][dz][dy][dx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_O, NDOF_C,
NDOF_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_O; ++qx)
{
u += QDD[1][tidz][qx][dz][dy] * Boo(qx, dx);
}
Y(dx + (dy + dz * NDOF_C) * NDOF_O, e) =
DDD[0][tidz][dz][dy][dx] - u;
}
MFEM_SYNC_THREAD;
// y: Vz Gco Boo Bcc - Vx Bcc Boo Gco
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_O,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_C; ++qz)
{
u += X[2][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_O] * Bcc(qz, dz);
}
QQD[0][tidz][qy][qx][dz] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_C, NQUAD_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_O; ++qz)
{
u += X[0][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_C] * Gco(qz, dz);
}
QQD[1][tidz][qy][qx][dz] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_O, NDOF_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_O; ++qy)
{
u += QQD[0][tidz][qy][qx][dz] * Boo(qy, dy);
}
QDD[0][tidz][qx][dz][dy] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_O, NDOF_C,
NQUAD_C, mnq - 1, mnq, mnq)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_O; ++qy)
{
u += QQD[1][tidz][qy][qx][dz] * Boo(qy, dy);
}
QDD[1][tidz][qx][dz][dy] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_O; ++qx)
{
u += QDD[0][tidz][qx][dz][dy] * Gco(qx, dx);
}
DDD[0][tidz][dz][dy][dx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_O,
NDOF_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_C; ++qx)
{
u += QDD[1][tidz][qx][dz][dy] * Bcc(qx, dx);
}
Y(dx + (dy + dz * NDOF_O) * NDOF_C + offset, e) =
DDD[0][tidz][dz][dy][dx] - u;
}
MFEM_SYNC_THREAD;
// z: Vx Bcc Gco Boo - Vy Gco Bcc Boo
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_O, NQUAD_C,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_O; ++qz)
{
u += X[0][tidz][qx + (qy + qz * NQUAD_O) * NQUAD_C] * Boo(qz, dz);
}
QQD[0][tidz][qy][qx][dz] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dz, qx, qy, x, NDOF_O, NQUAD_O,
NQUAD_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int qz = 0; qz < NQUAD_O; ++qz)
{
u += X[1][tidz][qx + (qy + qz * NQUAD_C) * NQUAD_O] * Boo(qz, dz);
}
QQD[1][tidz][qy][qx][dz] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_O,
NQUAD_C, mnq, mnq - 1, mnq)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_O; ++qy)
{
u += QQD[0][tidz][qy][qx][dz] * Gco(qy, dy);
}
QDD[0][tidz][qx][dz][dy] = u;
}
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dy, dz, qx, x, NDOF_C, NDOF_O,
NQUAD_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qy = 0; qy < NQUAD_C; ++qy)
{
u += QQD[1][tidz][qy][qx][dz] * Bcc(qy, dy);
}
QDD[1][tidz][qx][dz][dy] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_C,
NDOF_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_C; ++qx)
{
u += QDD[0][tidz][qx][dz][dy] * Bcc(qx, dx);
}
DDD[0][tidz][dz][dy][dx] = u;
}
MFEM_SYNC_THREAD;
// threads assigned to mitigate bank conflicts
MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(dx, dy, dz, x, NDOF_C, NDOF_C,
NDOF_O, mnq, mnq, mnq - 1)
{
real_t u = 0;
for (int qx = 0; qx < NQUAD_O; ++qx)
{
u += QDD[1][tidz][qx][dz][dy] * Gco(qx, dx);
}
Y(dx + (dy + dz * NDOF_C) * NDOF_C + 2 * offset, e) =
DDD[0][tidz][dz][dy][dx] - u;
}
MFEM_SYNC_THREAD;
});
}
} // namespace internal
template <int DIM, int NDOF_O, int NQUAD_O>
CurlInterpolator::ApplyKernelType
CurlInterpolator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 3)
{
return internal::CurlInterpolatorApply3DSmem<NDOF_O, NQUAD_O>;
}
MFEM_ABORT("Bad dimension!");
}
template <int DIM, int NDOF_O, int NQUAD_O>
CurlInterpolator::ApplyKernelType
CurlInterpolator::ApplyTPAKernels::Kernel()
{
if constexpr (DIM == 3)
{
return internal::CurlInterpolatorTApply3DSmem<NDOF_O, NQUAD_O>;
}
MFEM_ABORT("Bad dimension!");
}
} // namespace mfem
/// \endcond DO_NOT_DOCUMENT
+14 -65
View File
@@ -294,61 +294,14 @@ void PAHdivMassAssembleDiagonal3D(const int D1D,
}); // end of element loop
}
void PAHdivMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo,
const Array<real_t> &Bc,
const Array<real_t> &Bot,
const Array<real_t> &Bct,
const Vector &op,
const Vector &x,
Vector &y)
{
const int id = (D1D << 4) | Q1D;
if (dim == 2)
{
switch (id)
{
case 0x22: return SmemPAHdivMassApply2D<2,2>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x33: return SmemPAHdivMassApply2D<3,3>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x44: return SmemPAHdivMassApply2D<4,4>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x55: return SmemPAHdivMassApply2D<5,5>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
default: // fallback
return PAHdivMassApply2D(D1D,Q1D,NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
}
}
else if (dim == 3)
{
switch (id)
{
case 0x23: return SmemPAHdivMassApply3D<2,3>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x34: return SmemPAHdivMassApply3D<3,4>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x45: return SmemPAHdivMassApply3D<4,5>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x56: return SmemPAHdivMassApply3D<5,6>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x67: return SmemPAHdivMassApply3D<6,7>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
case 0x78: return SmemPAHdivMassApply3D<7,8>(NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
default: // fallback
return PAHdivMassApply3D(D1D,Q1D,NE,symmetric,Bo,Bc,Bot,Bct,op,x,y);
}
}
}
void PAHdivMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_)
void PAHdivMassApply2D(const int NE, const bool symmetric, const bool,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int TestD1D, const int Q1D)
{
MFEM_VERIFY(D1D == TestD1D,
"Trial and test spaces must have same number of dofs");
auto Bo = Reshape(Bo_.Read(), Q1D, D1D-1);
auto Bc = Reshape(Bc_.Read(), Q1D, D1D);
auto Bot = Reshape(Bot_.Read(), D1D-1, Q1D);
@@ -468,18 +421,14 @@ void PAHdivMassApply2D(const int D1D,
}); // end of element loop
}
void PAHdivMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_)
void PAHdivMassApply3D(const int NE, const bool symmetric, const bool,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int D1D, const int TestD1D, const int Q1D)
{
MFEM_VERIFY(D1D == TestD1D,
"Trial and test spaces must have same number of dofs");
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Error: D1D > HDIV_MAX_D1D");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
+25 -59
View File
@@ -66,58 +66,29 @@ void PAHdivMassAssembleDiagonal3D(const int D1D,
const Vector &op_,
Vector &diag_);
void PAHdivMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo,
const Array<real_t> &Bc,
const Array<real_t> &Bot,
const Array<real_t> &Bct,
const Vector &op,
const Vector &x,
Vector &y);
// PA H(div) Mass Apply 2D kernel
void PAHdivMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_);
void PAHdivMassApply2D(const int NE, const bool symmetric,
const bool scalar_coeff, const Array<real_t> &Bo_,
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
const Array<real_t> &Bct_, const Vector &op_,
const Vector &x_, Vector &y_, const int D1D,
const int TestD1D, const int Q1D);
// PA H(div) Mass Apply 3D kernel
void PAHdivMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_);
void PAHdivMassApply3D(const int NE, const bool symmetric,
const bool scalar_coeff, const Array<real_t> &Bo_,
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
const Array<real_t> &Bct_, const Vector &op_,
const Vector &x_, Vector &y_, const int D1D,
const int TestD1D, const int Q1D);
// Shared memory PA H(div) Mass Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAHdivMassApply2D(const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
template <int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAHdivMassApply2D(
const int NE, const bool symmetric, const bool, const Array<real_t> &Bo_,
const Array<real_t> &Bc_, const Array<real_t> &Bot_,
const Array<real_t> &Bct_, const Vector &op_, const Vector &x_, Vector &y_,
const int d1d = 0, const int = 0, const int q1d = 0)
{
MFEM_CONTRACT_VAR(Bot_);
MFEM_CONTRACT_VAR(Bct_);
@@ -280,18 +251,13 @@ inline void SmemPAHdivMassApply2D(const int NE,
}
// Shared memory PA H(div) Mass Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAHdivMassApply3D(const int NE,
const bool symmetric,
const Array<real_t> &Bo_,
const Array<real_t> &Bc_,
const Array<real_t> &Bot_,
const Array<real_t> &Bct_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
template <int T_D1D = 0, int T_Q1D = 0>
inline void
SmemPAHdivMassApply3D(const int NE, const bool symmetric, const bool,
const Array<real_t> &Bo_, const Array<real_t> &Bc_,
const Array<real_t> &Bot_, const Array<real_t> &Bct_,
const Vector &op_, const Vector &x_, Vector &y_,
const int d1d = 0, const int = 0, const int q1d = 0)
{
MFEM_CONTRACT_VAR(Bot_);
MFEM_CONTRACT_VAR(Bct_);
+471
View File
@@ -14,9 +14,218 @@
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
#include "bilininteg_hcurlhdiv_kernels.hpp"
namespace mfem
{
namespace
{
void PAHcurlApplyCurl2D(const int c_dofs1D,
const int o_dofs1D,
const int NE,
const Array<real_t> &Bo_,
const Array<real_t> &Gc_,
const Vector &x_,
Vector &y_)
{
auto Bo = Reshape(Bo_.Read(), o_dofs1D, o_dofs1D);
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
auto X = Reshape(x_.Read(), 2 * c_dofs1D * o_dofs1D, NE);
auto Y = Reshape(y_.ReadWrite(), o_dofs1D, o_dofs1D, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int iy = 0; iy < c_dofs1D; ++iy)
{
for (int ix = 0; ix < o_dofs1D; ++ix)
{
const real_t xv = X(ix + iy * o_dofs1D, e);
for (int oy = 0; oy < o_dofs1D; ++oy)
{
const real_t gy = Gc(oy, iy);
for (int ox = 0; ox < o_dofs1D; ++ox)
{
Y(ox, oy, e) -= Bo(ox, ix) * gy * xv;
}
}
}
}
const int y_nd = c_dofs1D * o_dofs1D;
for (int iy = 0; iy < o_dofs1D; ++iy)
{
for (int ix = 0; ix < c_dofs1D; ++ix)
{
const real_t xv = X(y_nd + ix + iy * c_dofs1D, e);
for (int oy = 0; oy < o_dofs1D; ++oy)
{
const real_t by = Bo(oy, iy);
for (int ox = 0; ox < o_dofs1D; ++ox)
{
Y(ox, oy, e) += Gc(ox, ix) * by * xv;
}
}
}
}
});
}
void PAHcurlApplyCurlTranspose2D(const int c_dofs1D,
const int o_dofs1D,
const int NE,
const Array<real_t> &Bo_,
const Array<real_t> &Gc_,
const Vector &x_,
Vector &y_)
{
auto Bo = Reshape(Bo_.Read(), o_dofs1D, o_dofs1D);
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
auto X = Reshape(x_.Read(), o_dofs1D, o_dofs1D, NE);
auto Y = Reshape(y_.ReadWrite(), 2 * c_dofs1D * o_dofs1D, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int dy = 0; dy < c_dofs1D; ++dy)
{
for (int dx = 0; dx < o_dofs1D; ++dx)
{
real_t sum = 0.0;
for (int oy = 0; oy < o_dofs1D; ++oy)
{
const real_t gy = Gc(oy, dy);
for (int ox = 0; ox < o_dofs1D; ++ox)
{
sum -= Bo(ox, dx) * gy * X(ox, oy, e);
}
}
Y(dx + dy * o_dofs1D, e) += sum;
}
}
const int y_nd = c_dofs1D * o_dofs1D;
for (int dy = 0; dy < o_dofs1D; ++dy)
{
for (int dx = 0; dx < c_dofs1D; ++dx)
{
real_t sum = 0.0;
for (int oy = 0; oy < o_dofs1D; ++oy)
{
const real_t by = Bo(oy, dy);
for (int ox = 0; ox < o_dofs1D; ++ox)
{
sum += Gc(ox, dx) * by * X(ox, oy, e);
}
}
Y(y_nd + dx + dy * c_dofs1D, e) += sum;
}
}
});
}
void PAHdivApplyCurl2D(const int c_dofs1D,
const int o_dofs1D,
const int NE,
const Array<real_t> &Bc_,
const Array<real_t> &Gc_,
const Vector &x_,
Vector &y_)
{
auto Bc = Reshape(Bc_.Read(), c_dofs1D, c_dofs1D);
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
auto X = Reshape(x_.Read(), c_dofs1D, c_dofs1D, NE);
auto Y = Reshape(y_.ReadWrite(), 2 * c_dofs1D * o_dofs1D, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int iy = 0; iy < c_dofs1D; ++iy)
{
for (int ix = 0; ix < c_dofs1D; ++ix)
{
const real_t xv = X(ix, iy, e);
for (int oy = 0; oy < o_dofs1D; ++oy)
{
const real_t gy = Gc(oy, iy);
for (int ox = 0; ox < c_dofs1D; ++ox)
{
Y(ox + oy * c_dofs1D, e) += Bc(ox, ix) * gy * xv;
}
}
}
}
const int y_nd = c_dofs1D * o_dofs1D;
for (int iy = 0; iy < c_dofs1D; ++iy)
{
for (int ix = 0; ix < c_dofs1D; ++ix)
{
const real_t xv = X(ix, iy, e);
for (int oy = 0; oy < c_dofs1D; ++oy)
{
const real_t by = Bc(oy, iy);
for (int ox = 0; ox < o_dofs1D; ++ox)
{
Y(y_nd + ox + oy * o_dofs1D, e) -= Gc(ox, ix) * by * xv;
}
}
}
}
});
}
void PAHdivApplyCurlTranspose2D(const int c_dofs1D,
const int o_dofs1D,
const int NE,
const Array<real_t> &Bc_,
const Array<real_t> &Gc_,
const Vector &x_,
Vector &y_)
{
auto Bc = Reshape(Bc_.Read(), c_dofs1D, c_dofs1D);
auto Gc = Reshape(Gc_.Read(), o_dofs1D, c_dofs1D);
auto X = Reshape(x_.Read(), 2 * c_dofs1D * o_dofs1D, NE);
auto Y = Reshape(y_.ReadWrite(), c_dofs1D, c_dofs1D, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int dy = 0; dy < o_dofs1D; ++dy)
{
for (int dx = 0; dx < c_dofs1D; ++dx)
{
const real_t xv = X(dx + dy * c_dofs1D, e);
for (int iy = 0; iy < c_dofs1D; ++iy)
{
const real_t gy = Gc(dy, iy);
for (int ix = 0; ix < c_dofs1D; ++ix)
{
Y(ix, iy, e) += Bc(dx, ix) * gy * xv;
}
}
}
}
const int y_nd = c_dofs1D * o_dofs1D;
for (int dy = 0; dy < c_dofs1D; ++dy)
{
for (int dx = 0; dx < o_dofs1D; ++dx)
{
const real_t xv = X(y_nd + dx + dy * o_dofs1D, e);
for (int iy = 0; iy < c_dofs1D; ++iy)
{
const real_t by = Bc(dy, iy);
for (int ix = 0; ix < c_dofs1D; ++ix)
{
Y(ix, iy, e) -= Gc(dx, ix) * by * xv;
}
}
}
}
});
}
}
// Apply to x corresponding to DOFs in H^1 (domain) the (topological) gradient
// to get a dof in H(curl) (range). You can think of the range as the "test" space
// and the domain as the "trial" space, but there's no integration.
@@ -1950,4 +2159,266 @@ void IdentityInterpolator::AddMultTransposePA(const Vector &x, Vector &y) const
}
}
void CurlInterpolator::AssemblePA(const FiniteElementSpace &dom_fes,
const FiniteElementSpace &ran_fes)
{
Mesh *mesh = dom_fes.GetMesh();
dim = mesh->Dimension();
ne = dom_fes.GetNE();
pa_mode_2d = 0;
MFEM_VERIFY(ne == ran_fes.GetNE(),
"Different meshes for domain and range spaces");
if (dim == 2)
{
pa_data.SetSize(0);
const FiniteElement *dom_fel = dom_fes.GetTypicalFE();
const FiniteElement *ran_fel = ran_fes.GetTypicalFE();
const bool hcurl_to_scalar =
dynamic_cast<const VectorTensorFiniteElement*>(dom_fel) != NULL &&
dom_fel->GetDerivType() == FiniteElement::CURL &&
dynamic_cast<const TensorBasisElement*>(ran_fel) != NULL &&
ran_fel->GetRangeType() == FiniteElement::SCALAR;
const bool scalar_to_hdiv =
dynamic_cast<const TensorBasisElement*>(dom_fel) != NULL &&
dom_fel->GetRangeType() == FiniteElement::SCALAR &&
dynamic_cast<const VectorTensorFiniteElement*>(ran_fel) != NULL &&
ran_fel->GetDerivType() == FiniteElement::DIV;
MFEM_VERIFY(hcurl_to_scalar || scalar_to_hdiv,
"2D CurlInterpolator PA supports H(curl)->scalar and scalar->H(div) only.");
int closed_basis_type = -1;
int open_basis_type = -1;
if (hcurl_to_scalar)
{
const auto *trial_fec = dynamic_cast<const ND_FECollection*>(dom_fes.FEColl());
const auto *range_fec = dynamic_cast<const L2_FECollection*>(ran_fes.FEColl());
MFEM_VERIFY(trial_fec != NULL, "H(curl) domain must use ND_FECollection.");
MFEM_VERIFY(range_fec != NULL, "Scalar range must use L2_FECollection.");
MFEM_VERIFY(ran_fel->GetMapType() == FiniteElement::INTEGRAL,
"2D H(curl)->scalar CurlInterpolator PA supports integral-map scalar range spaces only.");
closed_basis_type = trial_fec->GetClosedBasisType();
open_basis_type = trial_fec->GetOpenBasisType();
MFEM_VERIFY(range_fec->GetBasisType() == open_basis_type,
"Domain/range open basis types do not match.");
pa_mode_2d = 1;
}
else
{
const auto *trial_fec = dynamic_cast<const H1_FECollection*>(dom_fes.FEColl());
const auto *range_fec = dynamic_cast<const RT_FECollection*>(ran_fes.FEColl());
MFEM_VERIFY(trial_fec != NULL, "Scalar domain must use H1_FECollection.");
MFEM_VERIFY(range_fec != NULL, "H(div) range must use RT_FECollection.");
closed_basis_type = trial_fec->GetBasisType();
open_basis_type = range_fec->GetOpenBasisType();
MFEM_VERIFY(range_fec->GetClosedBasisType() == closed_basis_type,
"Domain/range closed basis types do not match.");
pa_mode_2d = 2;
}
const int order = hcurl_to_scalar
? dynamic_cast<const VectorTensorFiniteElement*>(dom_fel)->GetOrder()
: dynamic_cast<const NodalTensorFiniteElement*>(dom_fel)->GetOrder();
c_dofs1D = order + 1;
o_dofs1D = order;
closed_dofquad_fe.reset(new H1_SegmentElement(order, closed_basis_type));
open_dofquad_fe.reset(new L2_SegmentElement(order - 1, open_basis_type));
mfem::QuadratureFunctions1D qf1d;
mfem::IntegrationRule closed_ir;
closed_ir.SetSize(c_dofs1D);
qf1d.GaussLobatto(c_dofs1D, &closed_ir);
mfem::IntegrationRule open_ir;
open_ir.SetSize(o_dofs1D);
qf1d.GaussLegendre(o_dofs1D, &open_ir);
maps_C_C = &closed_dofquad_fe->GetDofToQuad(closed_ir, DofToQuad::TENSOR);
maps_O_C = &closed_dofquad_fe->GetDofToQuad(open_ir, DofToQuad::TENSOR);
maps_O_O = &open_dofquad_fe->GetDofToQuad(open_ir, DofToQuad::TENSOR);
MFEM_VERIFY(maps_C_C->ndof == c_dofs1D && maps_C_C->nqpt == c_dofs1D, "");
MFEM_VERIFY(maps_O_C->ndof == c_dofs1D && maps_O_C->nqpt == o_dofs1D, "");
MFEM_VERIFY(maps_O_O->ndof == o_dofs1D && maps_O_O->nqpt == o_dofs1D, "");
return;
}
closed_dofquad_fe.reset();
open_dofquad_fe.reset();
maps_C_C = nullptr;
maps_O_C = nullptr;
maps_O_O = nullptr;
const VectorTensorFiniteElement *dom_el =
dynamic_cast<const VectorTensorFiniteElement *>(dom_fes.GetTypicalFE());
const VectorTensorFiniteElement *ran_el =
dynamic_cast<const VectorTensorFiniteElement *>(ran_fes.GetTypicalFE());
MFEM_VERIFY(dom_el != NULL, "Only VectorTensorFiniteElement is supported!");
MFEM_VERIFY(ran_el != NULL, "Only VectorTensorFiniteElement is supported!");
MFEM_VERIFY(dom_el->GetDerivType() == FiniteElement::CURL,
"Domain space must be H(curl)");
MFEM_VERIFY(ran_el->GetDerivType() == FiniteElement::DIV,
"Range space must be H(div)");
const int dims = dom_el->GetDim();
MFEM_VERIFY(dims == 3, "");
ndof_o = dom_el->GetOrder();
int ndof_c = ndof_o + 1;
nquad_o = ran_el->GetOrder();
int nquad_c = nquad_o + 1;
// extract the tensor product range dof locations
std::vector<real_t> qc(nquad_c);
std::vector<real_t> qo(nquad_o);
{
const IntegrationRule &ran_nodes = ran_el->GetNodes();
const Array<int> &quad_map = ran_el->GetDofMap();
for (int i = 0; i < nquad_c; ++i)
{
int idx = UnsignIndex(quad_map[i]);
qc[i] = ran_nodes.IntPoint(idx).x;
}
int offset = ndof_c * ndof_o * ndof_o;
for (int i = 0; i < nquad_o; ++i)
{
int idx = UnsignIndex(quad_map[i + offset]);
qo[i] = ran_nodes.IntPoint(idx).x;
}
}
// evaluate closed/open 1D basis (and their derivatives) at closed and
// open quads
// storage order: GCO, BCC, BOO
pa_data.SetSize(ndof_c * nquad_o + ndof_c * nquad_c + ndof_o * nquad_o);
auto ptr = pa_data.HostWrite();
auto &cbasis1d = dom_el->GetBasis1D();
auto &obasis1d = dom_el->GetOpenBasis1D();
Vector b, g;
b.SetSize(ndof_c);
g.SetSize(ndof_c);
for (int j = 0; j < nquad_o; ++j)
{
cbasis1d.Eval(qo[j], b, g);
for (int i = 0; i < ndof_c; ++i)
{
ptr[j + i * nquad_o] = g[i];
}
}
ptr += nquad_o * ndof_c;
for (int j = 0; j < nquad_c; ++j)
{
cbasis1d.Eval(qc[j], b);
for (int i = 0; i < ndof_c; ++i)
{
ptr[j + i * nquad_c] = b[i];
}
}
ptr += ndof_c * nquad_c;
b.SetSize(ndof_o);
for (int j = 0; j < nquad_o; ++j)
{
obasis1d.Eval(qo[j], b);
for (int i = 0; i < ndof_o; ++i)
{
ptr[j + i * nquad_o] = b[i];
}
}
}
CurlInterpolator::Kernels::Kernels()
{
CurlInterpolator::AddSpecialization<3, 1, 1>();
CurlInterpolator::AddSpecialization<3, 2, 2>();
CurlInterpolator::AddSpecialization<3, 3, 3>();
CurlInterpolator::AddSpecialization<3, 4, 4>();
CurlInterpolator::AddSpecialization<3, 5, 5>();
}
CurlInterpolator::CurlInterpolator() { static Kernels kernels{}; }
void CurlInterpolator::AddMultPA(const Vector &x, Vector &y) const
{
if (dim == 2)
{
MFEM_VERIFY(maps_C_C != nullptr && maps_O_C != nullptr,
"2D CurlInterpolator PA data is not assembled.");
if (pa_mode_2d == 1)
{
MFEM_VERIFY(maps_O_O != nullptr,
"2D CurlInterpolator scalar curl map is not assembled.");
PAHcurlApplyCurl2D(c_dofs1D, o_dofs1D, ne, maps_O_O->B, maps_O_C->G,
x, y);
}
else if (pa_mode_2d == 2)
{
PAHdivApplyCurl2D(c_dofs1D, o_dofs1D, ne, maps_C_C->B, maps_O_C->G,
x, y);
}
else
{
MFEM_ABORT("Unsupported 2D CurlInterpolator mode.");
}
return;
}
ApplyPAKernels::Run(dim, ndof_o, nquad_o, ne, ndof_o, nquad_o, pa_data, x, y);
}
void CurlInterpolator::AddMultTransposePA(const Vector &x, Vector &y) const
{
if (dim == 2)
{
MFEM_VERIFY(maps_C_C != nullptr && maps_O_C != nullptr,
"2D CurlInterpolator PA data is not assembled.");
if (pa_mode_2d == 1)
{
MFEM_VERIFY(maps_O_O != nullptr,
"2D CurlInterpolator scalar curl map is not assembled.");
PAHcurlApplyCurlTranspose2D(c_dofs1D, o_dofs1D, ne, maps_O_O->B,
maps_O_C->G, x, y);
}
else if (pa_mode_2d == 2)
{
PAHdivApplyCurlTranspose2D(c_dofs1D, o_dofs1D, ne, maps_C_C->B,
maps_O_C->G, x, y);
}
else
{
MFEM_ABORT("Unsupported 2D CurlInterpolator mode.");
}
return;
}
ApplyTPAKernels::Run(dim, ndof_o, nquad_o, ne, ndof_o, nquad_o, pa_data, x, y);
}
/// \cond DO_NOT_DOCUMENT
CurlInterpolator::ApplyKernelType
CurlInterpolator::ApplyPAKernels::Fallback(int DIM, int, int)
{
if (DIM == 3)
{
return internal::CurlInterpolatorApply3DSmem<0, 0>;
}
MFEM_ABORT("Bad dimension!");
}
CurlInterpolator::ApplyKernelType
CurlInterpolator::ApplyTPAKernels::Fallback(int DIM, int, int)
{
if (DIM == 3)
{
return internal::CurlInterpolatorTApply3DSmem<0, 0>;
}
MFEM_ABORT("Bad dimension!");
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
File diff suppressed because it is too large Load Diff
+2
View File
@@ -672,6 +672,8 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
const NodalTensorFiniteElement *trial_el =
dynamic_cast<const NodalTensorFiniteElement*>(trial_fel);
MFEM_VERIFY(trial_el != NULL, "Only NodalTensorFiniteElement is supported!");
MFEM_VERIFY(trial_el->GetMapType() == FiniteElement::VALUE,
"Only value map type is supported!");
const VectorTensorFiniteElement *test_el =
dynamic_cast<const VectorTensorFiniteElement*>(test_fel);
+2
View File
@@ -22,6 +22,8 @@ void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
MFEM_VERIFY(el.GetMapType() == FiniteElement::VALUE,
"Only value map type supported");
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
const auto *ir = IntRule ? IntRule : &MassIntegrator::GetRule(el, el, Trans);
@@ -0,0 +1,113 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_BILININTEG_VECTORFEMASS_KERNELS_HPP
#define MFEM_BILININTEG_VECTORFEMASS_KERNELS_HPP
#include "../../config/config.hpp"
#include "../bilininteg.hpp"
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_hcurl_kernels.hpp"
#include "bilininteg_hdiv_kernels.hpp"
#include "bilininteg_hcurlhdiv_kernels.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
namespace hcurlmass
{
constexpr int NBZ3D(int d1d, int q1d)
{
if (d1d <= 1 || q1d <= 0)
{
return 1;
}
// assume q1d >= d1d
// z dimension is capped at 64 on nvidia and amd gpus
int tmp = std::min((128 + q1d * q1d * q1d - 1) / (q1d * q1d * q1d), 64);
int smem_req =
sizeof(mfem::real_t) *
(3 * ((d1d - 1) * d1d * d1d + 2 * q1d * q1d * q1d) * tmp +
q1d * (d1d - 1) + q1d * d1d);
// assume GPU has at least 48k shared memory
return std::max(std::min(tmp, (48 * 1024 + smem_req - 1) / smem_req), 1);
}
} // namespace hcurlmass
} // namespace internal
template <FiniteElement::DerivType TrialType, FiniteElement::DerivType TestType,
int DIM, int TrialD1D, int TestD1D, int Q1D>
VectorFEMassIntegrator::ApplyKernelType
VectorFEMassIntegrator::ApplyPAKernels::Kernel()
{
constexpr bool trial_curl = (TrialType == mfem::FiniteElement::CURL);
constexpr bool trial_div = (TrialType == mfem::FiniteElement::DIV);
constexpr bool test_curl = (TestType == mfem::FiniteElement::CURL);
constexpr bool test_div = (TestType == mfem::FiniteElement::DIV);
if constexpr (DIM == 3)
{
if constexpr (trial_curl && test_curl)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
// assume TrialD1D == TestD1D
return internal::SmemPAHcurlMassApply3D<
TrialD1D, Q1D, internal::hcurlmass::NBZ3D(TrialD1D, Q1D)>;
}
else
{
return internal::PAHcurlMassApply3D;
}
}
else if constexpr (trial_div && test_div)
{
// assumes TrialD1D == TestD1D
return internal::SmemPAHdivMassApply3D<TrialD1D, Q1D>;
}
else if constexpr (trial_curl && test_div)
{
return internal::PAHdivHcurlMassApply3D;
}
else if constexpr (trial_div && test_curl)
{
return internal::PAHcurlHdivMassApply3D;
}
}
else if constexpr (DIM == 2) // 2D
{
if constexpr (trial_curl && test_curl)
{
return internal::PAHcurlMassApply2D;
}
else if constexpr (trial_div && test_div)
{
// assumes TrialD1D == TestD1D
return internal::SmemPAHdivMassApply2D<TrialD1D, Q1D>;
}
else if constexpr (trial_curl && test_div)
{
return internal::PAHdivHcurlMassApply2D;
}
else if constexpr (trial_div && test_curl)
{
return internal::PAHcurlHdivMassApply2D;
}
}
MFEM_ABORT("Unknown kernel.");
}
/// \endcond DO_NOT_DOCUMENT
}
#endif
+126 -209
View File
@@ -10,15 +10,123 @@
// CONTRIBUTING.md for details.
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_hcurl_kernels.hpp"
#include "bilininteg_hdiv_kernels.hpp"
#include "bilininteg_hcurlhdiv_kernels.hpp"
#include "bilininteg_vectorfemass_kernels.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
VectorFEMassIntegrator::ApplyKernelType
VectorFEMassIntegrator::ApplyPAKernels::Fallback(
FiniteElement::DerivType TrialType, FiniteElement::DerivType TestType,
int dim, int, int, int)
{
const bool trial_curl = (TrialType == mfem::FiniteElement::CURL);
const bool trial_div = (TrialType == mfem::FiniteElement::DIV);
const bool test_curl = (TestType == mfem::FiniteElement::CURL);
const bool test_div = (TestType == mfem::FiniteElement::DIV);
if (dim == 3)
{
if (trial_curl && test_curl)
{
return internal::PAHcurlMassApply3D;
}
else if (trial_div && test_div)
{
return internal::PAHdivMassApply3D;
}
else if (trial_curl && test_div)
{
return internal::PAHdivHcurlMassApply3D;
}
else if (trial_div && test_curl)
{
return internal::PAHcurlHdivMassApply3D;
}
}
else if (dim == 2) // 2D
{
if (trial_curl && test_curl)
{
return internal::PAHcurlMassApply2D;
}
else if (trial_div && test_div)
{
return internal::PAHdivMassApply2D;
}
else if (trial_curl && test_div)
{
return internal::PAHdivHcurlMassApply2D;
}
else if (trial_div && test_curl)
{
return internal::PAHcurlHdivMassApply2D;
}
}
MFEM_ABORT("Unknown kernel.");
}
/// \endcond DO_NOT_DOCUMENT
VectorFEMassIntegrator::Kernels::Kernels()
{
// h(curl), h(curl)
// Q = P + 1 (3D)
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 2, 2, 3>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 3, 3, 4>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 4, 4, 5>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 5, 5, 6>();
// Q = P + 2 (3D)
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 2, 2, 4>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 3, 3, 5>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 4, 4, 6>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 5, 5, 7>();
// Q = P + 4 (3D)
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 2, 2, 6>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 3, 3, 7>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 4, 4, 8>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::CURL,
FiniteElement::CURL, 3, 5, 5, 9>();
// h(div), h(div)
// Q = P (2D)
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 2, 2, 2, 2>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 2, 3, 3, 3>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 2, 4, 4, 4>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 2, 5, 5, 5>();
// Q = P + 1 (3D)
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 3, 2, 2, 3>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 3, 3, 3, 4>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 3, 4, 4, 5>();
VectorFEMassIntegrator::AddSpecialization<FiniteElement::DIV,
FiniteElement::DIV, 3, 5, 5, 6>();
}
void VectorFEMassIntegrator::Init(Coefficient *q, DiagonalMatrixCoefficient *dq,
MatrixCoefficient *mq)
{
static Kernels kernels{};
Q = q;
DQ = dq;
MQ = mq;
}
void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
@@ -67,8 +175,8 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
MFEM_VERIFY(dofs1D == mapsO->ndof + 1 && quad1D == mapsO->nqpt, "");
trial_fetype = trial_el->GetDerivType();
test_fetype = test_el->GetDerivType();
trial_fetype = static_cast<FiniteElement::DerivType>(trial_el->GetDerivType());
test_fetype = static_cast<FiniteElement::DerivType>(test_el->GetDerivType());
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
@@ -215,225 +323,34 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
if (dim == 3)
{
if (trial_curl && test_curl)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23:
return internal::SmemPAHcurlMassApply3D<2,3>(
dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x34:
return internal::SmemPAHcurlMassApply3D<3,4>(
dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x45:
return internal::SmemPAHcurlMassApply3D<4,5>(
dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x56:
return internal::SmemPAHcurlMassApply3D<5,6>(
dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
default:
return internal::SmemPAHcurlMassApply3D(
dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
}
else
{
internal::PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(3, dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (trial_curl && test_div)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
true, false, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else if (trial_div && test_curl)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
false, false, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
else // 2D
{
if (trial_curl && test_curl)
{
internal::PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(2, dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
else if ((trial_curl && test_div) || (trial_div && test_curl))
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
trial_curl, false, mapsO->B, mapsC->B,
mapsOtest->Bt, mapsCtest->Bt, pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
const bool scalar_coeff = !(DQ || MQ);
ApplyPAKernels::Run(trial_fetype, test_fetype, dim, dofs1D, dofs1Dtest,
quad1D, ne, symmetric, scalar_coeff, mapsO->B, mapsC->B,
mapsOtest->Bt, mapsCtest->Bt, pa_data, x, y, dofs1D,
dofs1Dtest, quad1D);
}
void VectorFEMassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
{
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
const bool scalar_coeff = !(DQ || MQ);
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absBo(mapsO->B);
Array<real_t> absBc(mapsC->B);
Array<real_t> absBto(mapsO->Bt);
Array<real_t> absBtc(mapsC->Bt);
Array<real_t> absBto_t(mapsOtest->Bt);
Array<real_t> absBtc_t(mapsCtest->Bt);
absBo.Abs();
absBc.Abs();
absBto.Abs();
absBtc.Abs();
absBto_t.Abs();
absBtc_t.Abs();
if (dim == 3)
{
if (trial_curl && test_curl)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23:
return internal::SmemPAHcurlMassApply3D<2,3>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x34:
return internal::SmemPAHcurlMassApply3D<3,4>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x45:
return internal::SmemPAHcurlMassApply3D<4,5>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x56:
return internal::SmemPAHcurlMassApply3D<5,6>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
default:
return internal::SmemPAHcurlMassApply3D(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
}
else
{
internal::PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(3, dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if (trial_curl && test_div)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, true, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else if (trial_div && test_curl)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, false, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
else // 2D
{
if (trial_curl && test_curl)
{
internal::PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(2, dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if ((trial_curl && test_div) || (trial_div && test_curl))
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, trial_curl, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
ApplyPAKernels::Run(trial_fetype, test_fetype, dim, dofs1D, dofs1Dtest,
quad1D, ne, symmetric, scalar_coeff, absBo, absBc,
absBto_t, absBtc_t, abs_pa_data, x, y, dofs1D,
dofs1Dtest, quad1D);
}
void VectorFEMassIntegrator::AddMultTransposePA(const Vector &x,
+4 -8
View File
@@ -542,7 +542,10 @@ void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
return;
}
#ifndef MFEM_USE_MPFR
#ifdef MFEM_USE_MPFR
MFEM_WARNING("MPFR implementation of Gauss-Jacobi quadrature not implemented yet. Falling "
"back to double precision implementation...");
#endif
const int n = np;
// common constants for Jacobi polynomials
@@ -611,13 +614,6 @@ void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
ab + 1) / ((1.0 - xi*xi)*pp*pp) / pow(2, ab);
// map nodes and weights to the interval [0,1]
}
#else // MFEM_USE_MPFR is defined
MFEM_ABORT("MPFR implementation of Gauss-Jacobi quadrature not defined yet");
#endif // MFEM_USE_MPFR
}
+4 -4
View File
@@ -94,10 +94,10 @@ void BatchedLOR_AMS::Form2DEdgeToVertex_RT(Array<int> &edge2vert)
const int iv0 = ix + iy*op1;
const int iv1 = ix1 + iy1*op1;
// Rotated gradient in 2D (-dy, dx), so flip the sign for the first
// component (c == 0).
e2v(0, iedge) = (c == 1) ? iv0 : iv1;
e2v(1, iedge) = (c == 1) ? iv1 : iv0;
// 2D curl (dy, -dx), so flip the sign for the second
// component (c == 1).
e2v(0, iedge) = (c == 0) ? iv0 : iv1;
e2v(1, iedge) = (c == 0) ? iv1 : iv0;
}
}
}
+12 -9
View File
@@ -142,8 +142,6 @@ static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
int BatchedLORAssembly::FillI(SparseMatrix &A) const
{
static constexpr int Max = 16;
const int nvdof = fes_ho.GetVSize();
const int ndof_per_el = fes_ho.GetTypicalFE()->GetDof();
@@ -165,6 +163,8 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
const auto K = dof_glob2loc_offsets_.Read();
const auto map = Reshape(sparse_mapping.Read(), nnz_per_row, ndof_per_el);
Array<int> ij_elts(dof_glob2loc_.Size() * 2);
auto d_ij_elts = Reshape(ij_elts.Write(), dof_glob2loc_.Size(), 2);
auto I = A.WriteI();
@@ -176,10 +176,10 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
const int sii = el_dof_lex(ii_el, iel_ho);
const int ii = (sii >= 0) ? sii : -1 -sii;
// Get number and list of elements containing this DOF
int i_elts[Max];
const int i_offset = K[ii];
const int i_next_offset = K[ii+1];
const int i_ne = i_next_offset - i_offset;
int *i_elts = &d_ij_elts(i_offset, 0);
for (int e_i = 0; e_i < i_ne; ++e_i)
{
const int si_E = dof_glob2loc[i_offset+e_i]; // signed
@@ -202,7 +202,7 @@ int BatchedLORAssembly::FillI(SparseMatrix &A) const
}
else // assembly required
{
int j_elts[Max];
int *j_elts = &d_ij_elts(j_offset, 1);
for (int e_j = 0; e_j < j_ne; ++e_j)
{
const int sj_E = dof_glob2loc[j_offset+e_j]; // signed
@@ -269,7 +269,8 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
mfem::forall(nvdof + 1, [=] MFEM_HOST_DEVICE (int i) { I[i] = I2[i]; });
}
static constexpr int Max = 16;
Array<int> ij_B_el(dof_glob2loc_.Size() * 4);
auto d_ij_B_el = Reshape(ij_B_el.Write(), dof_glob2loc_.Size(), 4);
mfem::forall(ndof_per_el*nel_ho, [=] MFEM_HOST_DEVICE (int i)
{
@@ -279,11 +280,13 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
const int sii = el_dof_lex(ii_el, iel_ho); // signed
const int ii = (sii >= 0) ? sii : -1 - sii;
// Get number and list of elements containing this DOF
int i_elts[Max];
int i_B[Max];
const int i_offset = K[ii];
const int i_next_offset = K[ii+1];
const int i_ne = i_next_offset - i_offset;
int *i_elts = &d_ij_B_el(i_offset, 0);
int *i_B = &d_ij_B_el(i_offset, 1);
for (int e_i = 0; e_i < i_ne; ++e_i)
{
const int si_E = dof_glob2loc[i_offset+e_i]; // signed
@@ -312,8 +315,8 @@ void BatchedLORAssembly::FillJAndData(SparseMatrix &A) const
}
else // assembly required
{
int j_elts[Max];
int j_B[Max];
int *j_elts = &d_ij_B_el(j_offset, 2);
int *j_B = &d_ij_B_el(j_offset, 3);
for (int e_j = 0; e_j < j_ne; ++e_j)
{
const int sj_E = dof_glob2loc[j_offset+e_j]; // signed
+3 -2
View File
@@ -268,8 +268,9 @@ static void Derivatives3D(const int NE,
DeviceMatrix B(BG[0], D1D, Q1D);
DeviceMatrix G(BG[1], D1D, Q1D);
MFEM_SHARED real_t sm0[3][MQ1*MQ1*MQ1];
MFEM_SHARED real_t sm1[3][MQ1*MQ1*MQ1];
constexpr int MDQ = MD1 > MQ1 ? MD1 : MQ1;
MFEM_SHARED real_t sm0[3][MD1*MD1*MDQ];
MFEM_SHARED real_t sm1[3][MD1*MQ1*MQ1];
DeviceTensor<3> X(sm0[2], D1D, D1D, D1D);
DeviceTensor<3> DDQ0(sm0[0], D1D, D1D, Q1D);
DeviceTensor<3> DDQ1(sm0[1], D1D, D1D, Q1D);
+45 -7
View File
@@ -14,7 +14,7 @@
#include "../config/config.hpp"
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#if defined(MFEM_USE_CUDA)
#include <cusparse.h>
#include <library_types.h>
#include <cuda_runtime.h>
@@ -22,7 +22,7 @@
#endif
#include "cuda.hpp"
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#if defined(MFEM_USE_HIP)
#include <hip/hip_runtime.h>
#endif
#include "hip.hpp"
@@ -45,15 +45,17 @@
#endif
#if !defined(MFEM_USE_CUDA_OR_HIP)
constexpr bool mfem_use_gpu = false;
#define MFEM_DEVICE
#define MFEM_HOST
#define MFEM_LAMBDA
// #define MFEM_HOST_DEVICE // defined in config/config.hpp
// MFEM_DEVICE_SYNC is made available for debugging purposes
#define MFEM_DEVICE_SYNC
// MFEM_STREAM_SYNC is used for UVM and MPI GPU-Aware kernels
#define MFEM_STREAM_SYNC
#endif
#if !defined(MFEM_USE_CUDA_OR_HIP_LANG)
#define MFEM_DEVICE
#define MFEM_HOST
#define MFEM_LAMBDA
// #define MFEM_HOST_DEVICE // defined in config/config.hpp
#define MFEM_LAUNCH_BOUNDS(...)
#endif
@@ -66,6 +68,23 @@ constexpr bool mfem_use_gpu = false;
#define MFEM_THREAD_SIZE(k) 1
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) MFEM_FOREACH_THREAD(i,k,N)
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
// Example (3,2,1) block:
// 0 (0,0), 1 (1,0), 2 (2,0)
// 3 (1,0), 4 (1,1), 5 (2,1)
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
for (int iz = 0; iz < SZ; ++iz) \
for (int iy = 0; iy < SY; ++iy) \
for (int ix = 0; ix < SX; ++ix)
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
// contiguous in x. This intentionally offsets threads within the block to avoid
// shared memory bank conflicts.
// Example (3,2,1) block assigned to work on (2,2,1) items:
// 0 (0,0), 1 (1,0), 2 (N/A)
// 3 (1,0), 4 (1,1), 5 (N/A)
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
OY, OZ) \
MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ)
#endif
// 'double' and 'float' atomicAdd implementation for previous versions of CUDA
@@ -109,4 +128,23 @@ MFEM_HOST_DEVICE T AtomicAdd(T &add, const T val)
#endif
}
namespace mfem::internal
{
#if defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
static constexpr bool can_compile_kernels = false;
#else
static constexpr bool can_compile_kernels = true;
#endif
template <bool can_compile_kernels = can_compile_kernels>
void RequireKernelCompilation()
{
static_assert(
can_compile_kernels,
"The calling function needs to be compiled with CUDA/HIP language!");
}
}
#endif // MFEM_BACKENDS_HPP
+30 -9
View File
@@ -18,14 +18,8 @@
// CUDA block size used by MFEM.
#define MFEM_CUDA_BLOCKS 256
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#if defined(MFEM_USE_CUDA)
#define MFEM_USE_CUDA_OR_HIP
constexpr bool mfem_use_gpu = true;
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
// Define a CUDA error check macro, MFEM_GPU_CHECK(x), where x returns/is of
@@ -40,6 +34,15 @@ constexpr bool mfem_use_gpu = true;
} \
} while (0)
// Macros defined only when compiling with CUDA language
#if defined(__CUDACC__)
#define MFEM_USE_CUDA_OR_HIP_LANG
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
// Define the MFEM inner threading macros
#if defined(__CUDA_ARCH__)
#define MFEM_SHARED __shared__
@@ -49,13 +52,31 @@ constexpr bool mfem_use_gpu = true;
#define MFEM_THREAD_SIZE(k) blockDim.k
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=threadIdx.k; i<N; i+=blockDim.k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) if(const int i=threadIdx.k; i<N)
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
// Example (3,2,1) block:
// 0 (0,0), 1 (1,0), 2 (2,0)
// 3 (1,0), 4 (1,1), 5 (2,1)
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
if (int ix = threadIdx.k % (SX), iy = threadIdx.k / (SX), iz = iy / (SY); \
(iy %= (SY)), (threadIdx.k < (SX) * (SY) * (SZ)))
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
// contiguous in x. This intentionally offsets threads within the block to avoid
// shared memory bank conflicts.
// Example (3,2,1) block assigned to work on (2,2,1) items:
// 0 (0,0), 1 (1,0), 2 (N/A)
// 3 (1,0), 4 (1,1), 5 (N/A)
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
OY, OZ) \
if (int ix = threadIdx.k % (OX), iy = threadIdx.k / (OX), iz = iy / (OY); \
(ix < (SX)) && ((iy %= (OY)) < (SY)) && (iz < (SZ)))
#endif // defined(__CUDA_ARCH__)
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#endif // defined(__CUDACC__)
#endif // defined(MFEM_USE_CUDA)
namespace mfem
{
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#if defined(MFEM_USE_CUDA)
// Function used by the macro MFEM_GPU_CHECK.
void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
const char *file, int line);
+1 -1
View File
@@ -171,7 +171,7 @@ void mfem_error(const char *msg)
#ifdef MFEM_USE_EXCEPTIONS
if (mfem_error_action == MFEM_ERROR_THROW)
{
throw ErrorException(msg);
throw ErrorException(msg ? msg : "");
}
#endif
+2 -10
View File
@@ -15,7 +15,7 @@
#include "../config/config.hpp"
#include <iomanip>
#include <sstream>
#ifdef MFEM_USE_HIP
#if defined(MFEM_USE_HIP)
#include <hip/hip_runtime.h>
#endif
@@ -153,21 +153,13 @@ void mfem_warning(const char *msg = NULL);
// Additional abort functions for HIP
#if defined(MFEM_USE_HIP)
#ifndef __HIP_DEVICE_COMPILE__
template<typename T>
__host__ void abort_msg(T & msg)
{
MFEM_ABORT(msg);
}
#else
#if defined(__HIP_DEVICE_COMPILE__)
template<typename T>
__device__ void abort_msg(T & msg)
{
abort();
}
#endif
#endif
// Abort inside a device kernel
#if defined(__CUDA_ARCH__)
+6
View File
@@ -1044,6 +1044,8 @@ inline void ForallWrap(const bool use_dev, const int N,
const int X=0, const int Y=0, const int Z=0,
const int G=0)
{
internal::RequireKernelCompilation();
MFEM_CONTRACT_VAR(X);
MFEM_CONTRACT_VAR(Y);
MFEM_CONTRACT_VAR(Z);
@@ -1276,6 +1278,9 @@ inline void hypre_forall_cpu(int N, lambda &&body)
template<typename lambda>
inline void hypre_forall_gpu(int N, lambda &&body)
{
internal::RequireKernelCompilation();
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
#if defined(HYPRE_USING_CUDA)
CuWrap1D(N, body);
#elif defined(HYPRE_USING_HIP)
@@ -1283,6 +1288,7 @@ inline void hypre_forall_gpu(int N, lambda &&body)
#else
#error Unknown HYPRE GPU backend!
#endif
#endif
}
#endif
+31 -8
View File
@@ -18,14 +18,8 @@
// HIP block size used by MFEM.
#define MFEM_HIP_BLOCKS 256
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#if defined(MFEM_USE_HIP)
#define MFEM_USE_CUDA_OR_HIP
constexpr bool mfem_use_gpu = true;
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__ __device__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
// Define a HIP error check macro, MFEM_GPU_CHECK(x), where x returns/is of
@@ -40,6 +34,15 @@ constexpr bool mfem_use_gpu = true;
} \
} while (0)
// Macros defined only when compiling with HIP language
#if defined(__HIP__)
#define MFEM_USE_CUDA_OR_HIP_LANG
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
#define MFEM_LAMBDA __host__ __device__
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
// Define the MFEM inner threading macros
#if defined(__HIP_DEVICE_COMPILE__)
#define MFEM_SHARED __shared__
@@ -51,8 +54,28 @@ constexpr bool mfem_use_gpu = true;
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) \
if(const int i=hipThreadIdx_ ##k; i<N)
// Assigns a thread block shaped (SX,SY,SZ) contiguous in x.
// Example (3,2,1) block:
// 0 (0,0), 1 (1,0), 2 (2,0)
// 3 (1,0), 4 (1,1), 5 (2,1)
#define MFEM_FOREACH_THREAD_DIRECT_3D(ix, iy, iz, k, SX, SY, SZ) \
if (int ix = hipThreadIdx_##k % (SX), iy = hipThreadIdx_##k / (SX), \
iz = iy / (SY); \
(iy %= (SY)), (hipThreadIdx_##k < (SX) * (SY) * (SZ)))
// Assigns a thread block shaped (OX,OY,OZ) to work on items (SX,SY,SZ),
// contiguous in x. This intentionally offsets threads within the block to avoid
// shared memory bank conflicts.
// Example (3,2,1) block assigned to work on (2,2,1) items:
// 0 (0,0), 1 (1,0), 2 (N/A)
// 3 (1,0), 4 (1,1), 5 (N/A)
#define MFEM_FOREACH_THREAD_DIRECT_3D_OFFSET(ix, iy, iz, k, SX, SY, SZ, OX, \
OY, OZ) \
if (int ix = hipThreadIdx_##k % (OX), iy = hipThreadIdx_##k / (OX), \
iz = iy / (OY); \
(ix < (SX)) && ((iy %= (OY)) < (SY)) && (iz < (SZ)))
#endif // defined(__HIP_DEVICE_COMPILE__)
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
#endif // defined(__HIP__)
#endif // defined(MFEM_USE_HIP)
namespace mfem
{
+2 -2
View File
@@ -550,10 +550,10 @@ void reduce(int N, T &res, B &&body, const R &reducer, bool use_dev,
int num_mp = Device::NumMultiprocessors(Device::GetId());
#if defined(MFEM_USE_CUDA)
// good value of mp_sat found experimentally on Lassen
// good value of mp_sat found experimentally on Lassen (V100)
constexpr int mp_sat = 8;
#elif defined(MFEM_USE_HIP)
// good value of mp_sat found experimentally on Tuolumne
// good value of mp_sat found experimentally on Tuolumne (MI300A)
constexpr int mp_sat = 4;
#else
num_mp = 1;
+7 -1
View File
@@ -15,6 +15,10 @@
#include "backends.hpp"
#include "forall.hpp"
#if defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
#error "This header requires compilation with CUDA/HIP language!"
#else
#ifdef MFEM_USE_CUDA
#include <cub/device/device_scan.cuh>
#include <cub/device/device_select.cuh>
@@ -406,4 +410,6 @@ void CopyUnique(bool use_dev, InputIt d_in, OutputIt d_out,
#undef MFEM_CUB_NAMESPACE
#endif
#endif // defined(MFEM_USE_CUDA_OR_HIP) && !defined(MFEM_USE_CUDA_OR_HIP_LANG)
#endif // MFEM_SCAN_HPP
+13
View File
@@ -13,6 +13,7 @@
#include "native.hpp"
#include "gpu_blas.hpp"
#include "magma.hpp"
#include "../../general/reducers.hpp"
namespace mfem
{
@@ -119,4 +120,16 @@ void BatchedLinAlgBase::MultTranspose(const DenseTensor &A, const Vector &x,
AddMult(A, x, y, 1.0, 0.0, Op::T);
}
void VerifyBatchedLUInfo(const Array<int> &info_array, const char *message)
{
static Array<int> workspace;
int status = 0;
const int *d_info = info_array.Read();
mfem::reduce(
info_array.Size(), status,
[=] MFEM_HOST_DEVICE (int i, int &r) { r |= d_info[i]; },
BOrReducer<int> {}, true, workspace);
MFEM_VERIFY(status == 0, message);
}
}
+3
View File
@@ -141,6 +141,9 @@ public:
virtual ~BatchedLinAlgBase() { }
};
/// Check that all batched LU info values are zero.
void VerifyBatchedLUInfo(const Array<int> &info_array, const char *message);
} // namespace mfem
#endif
+6 -3
View File
@@ -126,7 +126,8 @@ void GPUBlasBatchedLinAlg::LUFactor(DenseTensor &A, Array<int> &P) const
const blasStatus_t status = MFEM_GPUBLAS_PREFIX(getrfBatched)(
GPUBlas::Handle(), n, d_A_ptrs, n, P.Write(),
info_array.Write(), n_mat);
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
}
void GPUBlasBatchedLinAlg::LUSolve(
@@ -189,12 +190,14 @@ void GPUBlasBatchedLinAlg::Invert(DenseTensor &A) const
status = MFEM_GPUBLAS_PREFIX(getrfBatched)(
GPUBlas::Handle(), n, d_LU_ptrs, n, P.Write(),
info_array.Write(), n_mat);
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
status = MFEM_GPUBLAS_PREFIX(getriBatched)(
GPUBlas::Handle(), n, d_LU_ptrs, n, P.ReadWrite(), d_A_ptrs, n,
info_array.Write(), n_mat);
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "");
MFEM_VERIFY(status == MFEM_BLAS_SUCCESS, "GPU BLAS error.");
VerifyBatchedLUInfo(info_array, "Batch matrix inversion failed");
}
#endif
+6 -3
View File
@@ -99,7 +99,8 @@ void MagmaBatchedLinAlg::LUFactor(DenseTensor &A, Array<int> &P) const
const magma_int_t status = MFEM_MAGMA_PREFIX(getrf_batched)(
n, n, d_A_ptrs, n, d_P_ptrs,
info_array.Write(), n_mat, Magma::Queue());
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
}
void MagmaBatchedLinAlg::LUSolve(
@@ -169,12 +170,14 @@ void MagmaBatchedLinAlg::Invert(DenseTensor &A) const
status = MFEM_MAGMA_PREFIX(getrf_batched)(
n, n, d_LU_ptrs, n, d_P_ptrs, info_array.Write(), n_mat,
Magma::Queue());
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
VerifyBatchedLUInfo(info_array, "Batch LU factorization failed");
status = MFEM_MAGMA_PREFIX(getri_outofplace_batched)(
n, d_LU_ptrs, n, d_P_ptrs, d_A_ptrs, n, info_array.Write(),
n_mat, Magma::Queue());
MFEM_VERIFY(status == MAGMA_SUCCESS, "");
MFEM_VERIFY(status == MAGMA_SUCCESS, "MAGMA error.");
VerifyBatchedLUInfo(info_array, "Batch matrix inversion failed");
}
} // namespace mfem
+11 -7
View File
@@ -246,6 +246,10 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
const int nrows_i = (A_i)?A_i->Height():0;
const int nrows = std::max(nrows_r, nrows_i);
const int ncols_r = (A_r)?A_r->Width():0;
const int ncols_i = (A_i)?A_i->Width():0;
const int ncols = std::max(ncols_r, ncols_i);
const int *I_r = (A_r)?A_r->GetI():NULL;
const int *I_i = (A_i)?A_i->GetI():NULL;
@@ -280,7 +284,7 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
J[I[i] + j] = J_r[I_r[i] + j];
D[I[i] + j] = D_r[I_r[i] + j];
J[I[i+nrows] + off_i + j] = J_r[I_r[i] + j] + nrows;
J[I[i+nrows] + off_i + j] = J_r[I_r[i] + j] + ncols;
D[I[i+nrows] + off_i + j] = factor*D_r[I_r[i] + j];
}
}
@@ -289,7 +293,7 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
const int off_r = (I_r)?(I_r[i+1] - I_r[i]):0;
for (int j=0; j<I_i[i+1] - I_i[i]; j++)
{
J[I[i] + off_r + j] = J_i[I_i[i] + j] + nrows;
J[I[i] + off_r + j] = J_i[I_i[i] + j] + ncols;
D[I[i] + off_r + j] = -D_i[I_i[i] + j];
J[I[i+nrows] + j] = J_i[I_i[i] + j];
@@ -892,12 +896,12 @@ ComplexHypreParMatrix::getColStartStop(const HypreParMatrix * A_r,
HYPRE_BigInt loc_start_stop[2];
offd_col_start_stop = new HYPRE_BigInt[2 * num_recv_procs];
const HYPRE_BigInt * row_part = (A_r) ? A_r->RowPart() :
((A_i) ? A_i->RowPart() : NULL);
const HYPRE_BigInt * col_part = (A_r) ? A_r->ColPart() :
((A_i) ? A_i->ColPart() : NULL);
int row_part_ind = (HYPRE_AssumedPartitionCheck()) ? 0 : myid_;
loc_start_stop[0] = row_part[row_part_ind];
loc_start_stop[1] = row_part[row_part_ind+1];
int col_part_ind = (HYPRE_AssumedPartitionCheck()) ? 0 : myid_;
loc_start_stop[0] = col_part[col_part_ind];
loc_start_stop[1] = col_part[col_part_ind+1];
MPI_Request * req = new MPI_Request[send_procs.size()+recv_procs.size()];
MPI_Status * stat = new MPI_Status[send_procs.size()+recv_procs.size()];
+40 -20
View File
@@ -87,6 +87,9 @@ CuDSSSolver::CuDSSSolver(MPI_Comm comm_) : mpi_comm(comm_)
CuDSSSolver::~CuDSSSolver()
{
// Sync the stream to make sure any pending asynchronous operations have
// completed.
MFEM_STREAM_SYNC;
// Destroy the system Matrix, RHS vector and solution vector
if (Ac)
{
@@ -99,7 +102,6 @@ CuDSSSolver::~CuDSSSolver()
MFEM_CUDSS_CHECK(cudssDataDestroy(handle, solverData));
MFEM_CUDSS_CHECK(cudssConfigDestroy(solverConfig));
MFEM_CUDSS_CHECK(cudssDestroy(handle));
handle = nullptr;
@@ -125,6 +127,9 @@ void CuDSSSolver::InitCuDSS()
// Create the cuDSS handle
MFEM_CUDSS_CHECK(cudssCreate(&handle));
// Set CuDSS to use MFEM's default stream of 0.
MFEM_CUDSS_CHECK(cudssSetStream(handle, 0));
#ifdef MFEM_USE_OPENMP
// NOTE: Set the threading layer library name to NULL so that cuDSS picks
// it from the environment variable "CUDSS_THREADING_LIB"
@@ -251,27 +256,42 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
Ac = std::make_unique<cudssMatrix_t>();
// Create empty RHS and solution vectors
SetNumRHS(1);
// Allocate device memory for csr values
CuMemAlloc(&csr_values_d, nnz * sizeof(real_t));
}
if (cuDSSObjectInitialized && !reorder_reuse)
{
MFEM_STREAM_SYNC;
MFEM_CUDSS_CHECK(cudssMatrixDestroy(*Ac));
}
// Allocate device memory for csr values. Unless reuse is specified, the
// nnz may be different, so we will free and reallocate.
if (csr_values_d == NULL || !reorder_reuse)
{
if (csr_values_d != NULL) { CuMemFree(csr_values_d); }
CuMemAlloc(&csr_values_d, nnz * sizeof(real_t));
}
CuMemcpyDtoD(csr_values_d, csr_values, nnz * sizeof(real_t));
// We copy and store the I and J arrays, since the CuDSS matrix object
// technically needs these to be valid, so we protect against the caller
// destroying the original matrix.
if (!cuDSSObjectInitialized || !reorder_reuse)
{
if (csr_offsets_d != NULL) { CuMemFree(csr_offsets_d); }
CuMemAlloc(&csr_offsets_d, (n_loc + 1) * sizeof(int));
if (csr_columns_d != NULL) { CuMemFree(csr_columns_d); }
CuMemAlloc(&csr_columns_d, nnz * sizeof(int));
CuMemcpyDtoD(csr_offsets_d, csr_offsets, (n_loc + 1) * sizeof(int));
CuMemcpyDtoD(csr_columns_d, csr_columns, nnz * sizeof(int));
}
// New cuDSS CSR matrix object and analysis or reuse the one from a previous
// matrix
if (!cuDSSObjectInitialized || !reorder_reuse)
{
if (reorder_reuse) // !cuDSSObjectInitialized && reorder_reuse
{
// NOTE: For CuDSS solver to reuse the reordering (skipping analysis
// phase), it needs to access the I and J arrays of the **initial**
// matrix. Therefore, we need to copy and keep I and J in device memory.
CuMemAlloc(&csr_offsets_d, (n_loc + 1) * sizeof(int));
CuMemAlloc(&csr_columns_d, nnz * sizeof(int));
CuMemcpyDtoD(csr_offsets_d, csr_offsets, (n_loc + 1) * sizeof(int));
CuMemcpyDtoD(csr_columns_d, csr_columns, nnz * sizeof(int));
#if CUDSS_VERSION >= 800
MFEM_CUDSS_CHECK(
cudssMatrixCreateCsr(
@@ -288,21 +308,17 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
}
else // !reorder_reuse
{
if (cuDSSObjectInitialized)
{
MFEM_CUDSS_CHECK(cudssMatrixDestroy(*Ac));
}
#if CUDSS_VERSION >= 800
MFEM_CUDSS_CHECK(
cudssMatrixCreateCsr(
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_INT_T, CUDSS_REAL_T,
mat_type, mview, CUDSS_BASE_ZERO));
#else
MFEM_CUDSS_CHECK(
cudssMatrixCreateCsr(
Ac.get(), n_global, n_global, nnz, csr_offsets, NULL,
csr_columns, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
Ac.get(), n_global, n_global, nnz, csr_offsets_d, NULL,
csr_columns_d, csr_values_d, CUDSS_INT_T, CUDSS_REAL_T,
mat_type, mview, CUDSS_BASE_ZERO));
#endif
}
@@ -326,6 +342,9 @@ void CuDSSSolver::SetMatrixCuDSS(int *csr_offsets, int *csr_columns,
// Factorization
MFEM_CUDSS_CHECK(cudssExecute(handle, CUDSS_PHASE_FACTORIZATION, solverConfig,
solverData, *Ac, yc, xc));
// In serial, the factorization can execute asynchronously.
MFEM_STREAM_SYNC;
}
void CuDSSSolver::SetOperator(const Operator &op)
@@ -360,6 +379,7 @@ void CuDSSSolver::SetNumRHS(int nrhs_) const
if (nrhs > 0)
{
// Destroy the previous RHS vector and solution vector
MFEM_STREAM_SYNC;
MFEM_CUDSS_CHECK(cudssMatrixDestroy(xc));
MFEM_CUDSS_CHECK(cudssMatrixDestroy(yc));
}
+1 -2
View File
@@ -157,8 +157,7 @@ private:
mutable int nrhs = 0; // the number of the RHSs
int nnz = 0; // the number of non zeros
// copy and keep the I and J arrays in device memory when skipping analysis
// phase
// copy and keep the I and J arrays in device memory
void *csr_offsets_d = NULL; // copy and keep I in device
void *csr_columns_d = NULL; // copy and keep J in device
void *csr_values_d = NULL; // copy and keep csr data in device
+38
View File
@@ -1136,6 +1136,17 @@ private:
public:
DenseTensor() : ni(0), nj(0), nk(0) { }
DenseTensor(const DenseTensor &other)
: tdata(other.tdata), ni(other.ni), nj(other.nj), nk(other.nk) { }
DenseTensor(DenseTensor &&other)
: tdata(std::move(other.tdata)), ni(other.ni), nj(other.nj), nk(other.nk)
{
// Reset other; other.tdata is reset in Array<T> move constructror.
other.Mk.ClearExternalData();
other.ni = other.nj = other.nk = 0;
}
DenseTensor(int i, int j, int k) : tdata(i*j*k), ni(i), nj(j), nk(k) { }
DenseTensor(real_t *d, int i, int j, int k)
@@ -1144,6 +1155,33 @@ public:
DenseTensor(int i, int j, int k, MemoryType mt)
: tdata(i*j*k, mt), ni(i), nj(j), nk(k) { }
DenseTensor &operator=(const DenseTensor &other)
{
if (this == &other) { return *this; }
Mk.ClearExternalData();
tdata = other.tdata;
ni = other.ni;
nj = other.nj;
nk = other.nk;
return *this;
}
DenseTensor &operator=(DenseTensor &&other)
{
if (this == &other) { return *this; }
Mk.ClearExternalData();
tdata = std::move(other.tdata);
ni = other.ni;
nj = other.nj;
nk = other.nk;
// Reset other; other.tdata is reset in Array<T> move assignment.
other.Mk.ClearExternalData();
other.ni = other.nj = other.nk = 0;
return *this;
}
int SizeI() const { return ni; }
int SizeJ() const { return nj; }
int SizeK() const { return nk; }
+4
View File
@@ -5842,6 +5842,10 @@ void HypreAMS::MakeGradientAndInterpolation(
{
grad->AddTraceFaceInterpolator(new GradientInterpolator);
}
else if (dynamic_cast<const RT_FECollection *>(edge_fec))
{
grad->AddDomainInterpolator(new CurlInterpolator);
}
else
{
grad->AddDomainInterpolator(new GradientInterpolator);
+1
View File
@@ -810,6 +810,7 @@ MINIAPPS_SUBDIRS = dpg/util hooke/operators hooke/preconditioners \
hooke/materials hooke/kernels
FORMAT_FILES += $(foreach dir,$(TESTS_SUBDIRS),tests/$(dir)/*.?pp)
FORMAT_FILES += $(foreach dir,$(UNIT_TESTS_SUBDIRS),tests/unit/$(dir)/*.?pp)
FORMAT_FILES += tests/unit/fem/specializations/*.?pp
FORMAT_FILES += $(foreach dir,$(MINIAPPS_SUBDIRS),miniapps/$(dir)/*.?pp)
FORMAT_FILES += config/cmake/config.hpp.in config/config.hpp.in mfem*.hpp
FORMAT_EXCLUDE = general/tinyxml2.cpp tests/unit/catch.hpp
+101 -5
View File
@@ -667,9 +667,84 @@ void Mesh::GetEdgeTransformation(int EdgeNo,
}
EdTr->SetFE(edge_el);
}
else
else // L2 Nodes (e.g., periodic mesh), go through the face containing the edge
{
MFEM_ABORT("Not implemented.");
// Search for a face that contains this edge
GetEdgeFaceTable();
Array<int> faces_e;
edge_face->GetRow(EdgeNo, faces_e);
MFEM_VERIFY(faces_e.Size() > 0, "Edge not found in any face!");
const int face_no = faces_e[0];
// Get edge local index and orientation
Array<int> edges_f, oris_f;
GetFaceEdges(face_no, edges_f, oris_f);
const int local_idx = edges_f.Find(EdgeNo);
MFEM_ASSERT(local_idx >= 0, "Edge not found on the face!");
const int edge_ori = oris_f[local_idx] > 0 ? 0 : 1;
// Get face information
const FaceInfo &face_info = faces_info[face_no];
// Get transformation from face to edge
IntegrationPointTransformation LocEdge;
int edge_info = EncodeFaceInfo(local_idx, edge_ori);
Element::Type face_type = GetFaceElementType(face_no);
switch (face_type)
{
case Element::TRIANGLE:
GetLocalSegToTriTransformation(LocEdge.Transf, edge_info);
break;
case Element::QUADRILATERAL:
GetLocalSegToQuadTransformation(LocEdge.Transf, edge_info);
break;
default:
MFEM_ABORT("Unsupported face type for edge transformation!");
}
// Get edge element
const int order = Nodes->FESpace()->GetElementOrder(face_info.Elem1No);
const L2_FECollection *l2_fec = dynamic_cast<const L2_FECollection*>
(Nodes->FESpace()->FEColl());
if (l2_fec)
{
// L2 elements do not have a defined trace space
if (!EdgeTransfElement || EdgeTransfElement->GetOrder() != order
|| EdgeTransfElement->GetBasisType() != l2_fec->GetBasisType())
{
EdgeTransfElement = make_unique<L2_SegmentElement>(
order, l2_fec->GetBasisType());
}
edge_el = EdgeTransfElement.get();
}
else
{
MFEM_ABORT("Unsupported finite element collection.");
}
// Map edge nodes to face reference space
IntegrationRule face_ir(edge_el->GetDof());
LocEdge.Transform(edge_el->GetNodes(), face_ir);
// Then, map from face to element
IntegrationPointTransformation Loc1;
GetLocalFaceTransformation(face_type,
GetElementType(face_info.Elem1No),
Loc1.Transf, face_info.Elem1Inf);
IntegrationRule elem_ir(edge_el->GetDof());
Loc1.Transf.ElementNo = face_info.Elem1No;
Loc1.Transf.ElementType = ElementTransformation::ELEMENT;
Loc1.Transf.mesh = this;
Loc1.Transform(face_ir, elem_ir);
// Finally, get the physical coordinates
Nodes->GetVectorValues(Loc1.Transf, elem_ir, pm);
EdTr->SetFE(edge_el);
}
}
}
@@ -1824,8 +1899,8 @@ void Mesh::Init()
void Mesh::InitTables()
{
el_to_edge =
el_to_face = el_to_el = bel_to_edge = face_edge = edge_vertex = NULL;
el_to_edge = el_to_face = el_to_el = bel_to_edge = NULL;
face_edge = edge_face = edge_vertex = NULL;
face_to_elem = NULL;
}
@@ -1848,6 +1923,7 @@ void Mesh::DestroyTables()
}
delete face_edge;
delete edge_face;
delete edge_vertex;
delete face_to_elem;
@@ -1921,6 +1997,7 @@ void Mesh::ResetLazyData()
{
delete el_to_el; el_to_el = NULL;
delete face_edge; face_edge = NULL;
delete edge_face; edge_face = NULL;
delete face_to_elem; face_to_elem = NULL;
delete edge_vertex; edge_vertex = NULL;
DeleteGeometricFactors();
@@ -2845,6 +2922,7 @@ void Mesh::ReorderElements(const Array<int> &ordering, bool reorder_vertices)
// boundary element ordering
// - el_to_el - no need to rebuild
// - face_edge - no need to rebuild
// - edge_face - no need to rebuild
// - edge_vertex - no need to rebuild
// - geom_factors - no need to rebuild
@@ -4598,8 +4676,9 @@ Mesh::Mesh(const Mesh &mesh, bool copy_nodes)
// Do NOT copy the element-to-element Table, el_to_el
el_to_el = NULL;
// Do NOT copy the face-to-edge Table, face_edge
// Do NOT copy the face-to-edge Table, face_edge and edge_face
face_edge = NULL;
edge_face = NULL;
face_to_elem = NULL;
// Copy the edge-to-vertex Table, edge_vertex
@@ -8094,6 +8173,22 @@ Table *Mesh::GetFaceEdgeTable() const
return (face_edge);
}
Table *Mesh::GetEdgeFaceTable() const
{
if (edge_face)
{
return edge_face;
}
if (Dim != 3)
{
return NULL;
}
edge_face = Transpose(*GetFaceEdgeTable());
return edge_face;
}
Table *Mesh::GetEdgeVertexTable() const
{
if (edge_vertex)
@@ -11452,6 +11547,7 @@ void Mesh::Swap(Mesh& other, bool non_geometry)
mfem::Swap(bel_to_edge, other.bel_to_edge);
mfem::Swap(be_to_face, other.be_to_face);
mfem::Swap(face_edge, other.face_edge);
mfem::Swap(edge_face, other.edge_face);
mfem::Swap(face_to_elem, other.face_to_elem);
mfem::Swap(edge_vertex, other.edge_vertex);
+8 -1
View File
@@ -250,16 +250,18 @@ protected:
Table *bel_to_edge; // for 3D only
// Note that the following tables are owned by this class and should not be
// deleted by the caller. Of these three tables, only face_edge and
// deleted by the caller. Of these four tables, only face_edge, edge_face and
// edge_vertex are returned by access functions.
mutable Table *face_to_elem; // Used by FindFaceNeighbors, not returned.
mutable Table *face_edge; // Returned by GetFaceEdgeTable().
mutable Table *edge_face; // Returned by GetEdgeFaceTable().
mutable Table *edge_vertex; // Returned by GetEdgeVertexTable().
IsoparametricTransformation Transformation, Transformation2;
IsoparametricTransformation BdrTransformation;
IsoparametricTransformation FaceTransformation, EdgeTransformation;
FaceElementTransformations FaceElemTr;
mutable std::unique_ptr<L2_SegmentElement> EdgeTransfElement;
// refinement embeddings for forward compatibility with NCMesh
mutable CoarseFineTransformations CoarseFineTr;
@@ -1731,6 +1733,11 @@ public:
/// @note The returned object should NOT be deleted by the caller.
Table *GetFaceEdgeTable() const;
/// Returns the edge-to-face Table (3D)
///
/// @note The returned object should NOT be deleted by the caller.
Table *GetEdgeFaceTable() const;
/// Returns the edge-to-vertex Table (3D)
///
/// @note The returned object should NOT be deleted by the caller.
+7
View File
@@ -4866,6 +4866,13 @@ void ParMesh::Print(std::ostream &os, const std::string &comments) const
return;
}
if (pncmesh && pncmesh->using_scaling)
{
// For nodes scaling, we write the file in the format MFEM NC mesh v1.1.
Printer(os, "", comments);
return;
}
const Array<int>* s2l_face;
if (!pncmesh)
{
+138 -62
View File
@@ -28,6 +28,48 @@ namespace mfem
using namespace bin_io;
static int GetHexEdgeSplit(const int* nodes, int v1, int v2);
static bool SameSplitScale(real_t a, real_t b)
{
#ifdef MFEM_USE_DOUBLE
constexpr real_t rel_tol = 1.0e-8;
#else
constexpr real_t rel_tol = 1.0e-5;
#endif
return std::abs(a - b) <= rel_tol *
std::max(real_t(1.0), std::max(std::abs(a), std::abs(b)));
}
static real_t DirectedHexEdgeScale(const int* nodes, const Refinement &ref,
int v0, int v1)
{
const int dir = GetHexEdgeSplit(nodes, v0, v1);
static const int split_edges[3][4][2] =
{
{{0, 1}, {3, 2}, {4, 5}, {7, 6}},
{{1, 2}, {0, 3}, {5, 6}, {4, 7}},
{{0, 4}, {1, 5}, {2, 6}, {3, 7}}
};
for (int i = 0; i < 4; i++)
{
const int a = nodes[split_edges[dir][i][0]];
const int b = nodes[split_edges[dir][i][1]];
if (a == v0 && b == v1)
{
return ref.s[dir];
}
if (a == v1 && b == v0)
{
return 1.0 - ref.s[dir];
}
}
MFEM_ABORT("Shared face edge does not match the refinement direction.");
return 0.0;
}
ParNCMesh::ParNCMesh(MPI_Comm comm, const NCMesh &ncmesh,
const int *partitioning)
: NCMesh(ncmesh)
@@ -1555,7 +1597,7 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
ElementNeighborProcessors(elem, ranks);
for (int j = 0; j < ranks.Size(); j++)
{
send_ref[ranks[j]].AddRefinement(elem, ref.GetType());
send_ref[ranks[j]].AddRefinement(elem, ref);
}
}
@@ -1576,8 +1618,8 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
for (int i = 0; i < refinements.Size(); i++)
{
const Refinement &ref = refinements[i];
CheckRefinement(leaf_elements[ref.index], ref.GetType(), refinements,
elemToRef, conflicts);
CheckRefinement(leaf_elements[ref.index], ref, refinements, elemToRef,
conflicts);
}
// Receive (ghost layer) refinements from all neighbors
@@ -1593,7 +1635,9 @@ bool ParNCMesh::AnisotropicConflict(const Array<Refinement> &refinements,
// check the ghost refinements
for (int i = 0; i < msg.Size(); i++)
{
CheckRefinement(msg.elements[i], msg.values[i], refinements, elemToRef,
Refinement ghost_ref(msg.elements[i], msg.values[i].ref_type);
ghost_ref.SetScaleForType(msg.values[i].scale);
CheckRefinement(msg.elements[i], ghost_ref, refinements, elemToRef,
conflicts);
}
}
@@ -1749,7 +1793,7 @@ int FindHexFace(const int* no, int vn1, int vn2, int vn3, int vn4)
// Assumption: v1 and v2 are indices of hex vertices connected by an edge.
// The return value is {0,1,2} denoting split {X,Y,Z}.
int GetHexEdgeSplit(const int* nodes, int v1, int v2)
static int GetHexEdgeSplit(const int* nodes, int v1, int v2)
{
Array<int> v(2);
v[0] = v1;
@@ -1780,7 +1824,8 @@ int GetHexEdgeSplit(const int* nodes, int v1, int v2)
return edgeDir[edge];
}
void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
void ParNCMesh::CheckRefAnisoFace(const Refinement &ref, int elem,
int vn1, int vn2, int vn3, int vn4,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
std::set<int> &conflicts)
@@ -1798,11 +1843,11 @@ void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
if (elemToRef.count(nghbIndex) > 0)
{
const int refIndex = elemToRef.at(nghbIndex);
const Refinement& ref = refinements[refIndex];
const Refinement& nghb_ref = refinements[refIndex];
bool refDir[3];
for (int i=0; i<3; ++i)
refDir[i] = ref.s[i] > real_t{0};
refDir[i] = nghb_ref.s[i] > real_t{0};
const int localFace = FindHexFace(nghb.node, vn1, vn2, vn3, vn4);
const int faceDir = GetHexFaceDir(localFace);
@@ -1834,30 +1879,50 @@ void ParNCMesh::CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
MFEM_ASSERT(cnt == 2 && hexSplitOnFace >= 0, "");
const int edgeSplit = GetHexEdgeSplit(nghb.node, vn1, vn2);
if (edgeSplit != hexSplitOnFace) { conflicts.insert(refIndex); }
if (edgeSplit != hexSplitOnFace)
{
conflicts.insert(refIndex);
}
else
{
const real_t elem_scale =
DirectedHexEdgeScale(elements[elem].node, ref, vn1, vn2);
const real_t nghb_scale =
DirectedHexEdgeScale(nghb.node, nghb_ref, vn1, vn2);
if (!SameSplitScale(elem_scale, nghb_scale))
{
conflicts.insert(refIndex);
}
}
}
}
// The else case is that the neighbor is not refined, so there is no need to
// check for conflicts.
}
void ParNCMesh::CheckRefIsoFace(int elem, int vn1, int vn2, int vn3, int vn4,
void ParNCMesh::CheckRefIsoFace(const Refinement &ref, int elem,
int vn1, int vn2, int vn3, int vn4,
int en1, int en2, int en3, int en4,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
std::set<int> &conflicts)
{
CheckRefAnisoFace(elem, vn1, vn2, en2, en4, refinements, elemToRef, conflicts);
CheckRefAnisoFace(elem, en4, en2, vn3, vn4, refinements, elemToRef, conflicts);
CheckRefAnisoFace(elem, vn4, vn1, en1, en3, refinements, elemToRef, conflicts);
CheckRefAnisoFace(elem, en3, en1, vn2, vn3, refinements, elemToRef, conflicts);
CheckRefAnisoFace(ref, elem, vn1, vn2, en2, en4, refinements, elemToRef,
conflicts);
CheckRefAnisoFace(ref, elem, en4, en2, vn3, vn4, refinements, elemToRef,
conflicts);
CheckRefAnisoFace(ref, elem, vn4, vn1, en1, en3, refinements, elemToRef,
conflicts);
CheckRefAnisoFace(ref, elem, en3, en1, vn2, vn3, refinements, elemToRef,
conflicts);
}
void ParNCMesh::CheckRefinement(int elem, char ref_type,
void ParNCMesh::CheckRefinement(int elem, const Refinement &ref,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
std::set<int> &conflicts)
{
const char ref_type = ref.GetType();
const Element &el = elements[elem];
MFEM_ASSERT(el.geom == Geometry::CUBE && el.ref_type == 0,
"Element must be an unrefined hexahedron");
@@ -1868,46 +1933,46 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
// This follows the logic of NCMesh::RefineElement().
if (ref_type == Refinement::X) // split along X axis
{
CheckRefAnisoFace(elem, no[0], no[1], no[5], no[4], refinements,
CheckRefAnisoFace(ref, elem, no[0], no[1], no[5], no[4], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[2], no[3], no[7], no[6], refinements,
CheckRefAnisoFace(ref, elem, no[2], no[3], no[7], no[6], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[4], no[5], no[6], no[7], refinements,
CheckRefAnisoFace(ref, elem, no[4], no[5], no[6], no[7], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[3], no[2], no[1], no[0], refinements,
CheckRefAnisoFace(ref, elem, no[3], no[2], no[1], no[0], refinements,
elemToRef, conflicts);
}
else if (ref_type == Refinement::Y) // split along Y axis
{
CheckRefAnisoFace(elem, no[1], no[2], no[6], no[5], refinements,
CheckRefAnisoFace(ref, elem, no[1], no[2], no[6], no[5], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[3], no[0], no[4], no[7], refinements,
CheckRefAnisoFace(ref, elem, no[3], no[0], no[4], no[7], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[5], no[6], no[7], no[4], refinements,
CheckRefAnisoFace(ref, elem, no[5], no[6], no[7], no[4], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[0], no[3], no[2], no[1], refinements,
CheckRefAnisoFace(ref, elem, no[0], no[3], no[2], no[1], refinements,
elemToRef, conflicts);
}
else if (ref_type == Refinement::Z) // split along Z axis
{
CheckRefAnisoFace(elem, no[4], no[0], no[1], no[5], refinements,
CheckRefAnisoFace(ref, elem, no[4], no[0], no[1], no[5], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[5], no[1], no[2], no[6], refinements,
CheckRefAnisoFace(ref, elem, no[5], no[1], no[2], no[6], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[6], no[2], no[3], no[7], refinements,
CheckRefAnisoFace(ref, elem, no[6], no[2], no[3], no[7], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[7], no[3], no[0], no[4], refinements,
CheckRefAnisoFace(ref, elem, no[7], no[3], no[0], no[4], refinements,
elemToRef, conflicts);
}
else if (ref_type == Refinement::XY) // XY split
{
CheckRefAnisoFace(elem, no[0], no[1], no[5], no[4], refinements,
CheckRefAnisoFace(ref, elem, no[0], no[1], no[5], no[4], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[1], no[2], no[6], no[5], refinements,
CheckRefAnisoFace(ref, elem, no[1], no[2], no[6], no[5], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[2], no[3], no[7], no[6], refinements,
CheckRefAnisoFace(ref, elem, no[2], no[3], no[7], no[6], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[3], no[0], no[4], no[7], refinements,
CheckRefAnisoFace(ref, elem, no[3], no[0], no[4], no[7], refinements,
elemToRef, conflicts);
const int mid01 = GetMidEdgeNode(no[0], no[1]);
@@ -1920,20 +1985,20 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
const int mid67 = GetMidEdgeNode(no[6], no[7]);
const int mid74 = GetMidEdgeNode(no[7], no[4]);
CheckRefIsoFace(elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
CheckRefIsoFace(ref, elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
mid30, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
CheckRefIsoFace(ref, elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
mid74, refinements, elemToRef, conflicts);
}
else if (ref_type == Refinement::XZ) // XZ split
{
CheckRefAnisoFace(elem, no[3], no[2], no[1], no[0], refinements,
CheckRefAnisoFace(ref, elem, no[3], no[2], no[1], no[0], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[2], no[6], no[5], no[1], refinements,
CheckRefAnisoFace(ref, elem, no[2], no[6], no[5], no[1], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[6], no[7], no[4], no[5], refinements,
CheckRefAnisoFace(ref, elem, no[6], no[7], no[4], no[5], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[7], no[3], no[0], no[4], refinements,
CheckRefAnisoFace(ref, elem, no[7], no[3], no[0], no[4], refinements,
elemToRef, conflicts);
const int mid01 = GetMidEdgeNode(no[0], no[1]);
@@ -1946,9 +2011,9 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
const int mid26 = GetMidEdgeNode(no[2], no[6]);
const int mid37 = GetMidEdgeNode(no[3], no[7]);
CheckRefIsoFace(elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
CheckRefIsoFace(ref, elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
mid04, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
CheckRefIsoFace(ref, elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
mid26, refinements, elemToRef, conflicts);
}
else if (ref_type == Refinement::YZ) // YZ split
@@ -1963,18 +2028,18 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
const int mid26 = GetMidEdgeNode(no[2], no[6]);
const int mid37 = GetMidEdgeNode(no[3], no[7]);
CheckRefAnisoFace(elem, no[4], no[0], no[1], no[5], refinements,
CheckRefAnisoFace(ref, elem, no[4], no[0], no[1], no[5], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[0], no[3], no[2], no[1], refinements,
CheckRefAnisoFace(ref, elem, no[0], no[3], no[2], no[1], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[3], no[7], no[6], no[2], refinements,
CheckRefAnisoFace(ref, elem, no[3], no[7], no[6], no[2], refinements,
elemToRef, conflicts);
CheckRefAnisoFace(elem, no[7], no[4], no[5], no[6], refinements,
CheckRefAnisoFace(ref, elem, no[7], no[4], no[5], no[6], refinements,
elemToRef, conflicts);
CheckRefIsoFace(elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
CheckRefIsoFace(ref, elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
mid15, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
CheckRefIsoFace(ref, elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
mid37, refinements, elemToRef, conflicts);
}
else if (ref_type == Refinement::XYZ) // XYZ split
@@ -1994,17 +2059,17 @@ void ParNCMesh::CheckRefinement(int elem, char ref_type,
const int mid26 = GetMidEdgeNode(no[2], no[6]);
const int mid37 = GetMidEdgeNode(no[3], no[7]);
CheckRefIsoFace(elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
CheckRefIsoFace(ref, elem, no[3], no[2], no[1], no[0], mid23, mid12, mid01,
mid30, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
CheckRefIsoFace(ref, elem, no[0], no[1], no[5], no[4], mid01, mid15, mid45,
mid04, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
CheckRefIsoFace(ref, elem, no[1], no[2], no[6], no[5], mid12, mid26, mid56,
mid15, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
CheckRefIsoFace(ref, elem, no[2], no[3], no[7], no[6], mid23, mid37, mid67,
mid26, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
CheckRefIsoFace(ref, elem, no[3], no[0], no[4], no[7], mid30, mid04, mid74,
mid37, refinements, elemToRef, conflicts);
CheckRefIsoFace(elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
CheckRefIsoFace(ref, elem, no[4], no[5], no[6], no[7], mid45, mid56, mid67,
mid74, refinements, elemToRef, conflicts);
}
else
@@ -2053,7 +2118,7 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
ElementNeighborProcessors(elem, ranks);
for (int j = 0; j < ranks.Size(); j++)
{
send_ref[ranks[j]].AddRefinement(elem, ref.GetType());
send_ref[ranks[j]].AddRefinement(elem, ref);
}
}
@@ -2063,8 +2128,9 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
// do local refinements
for (int i = 0; i < refinements.Size(); i++)
{
const Refinement &ref = refinements[i];
NCMesh::RefineElement(leaf_elements[ref.index], ref.GetType());
Refinement ref_i = refinements[i];
ref_i.index = leaf_elements[refinements[i].index];
NCMesh::RefineElement(ref_i);
}
// receive (ghost layer) refinements from all neighbors
@@ -2080,7 +2146,9 @@ void ParNCMesh::Refine(const Array<Refinement> &refinements)
// do the ghost refinements
for (int i = 0; i < msg.Size(); i++)
{
NCMesh::RefineElement(msg.elements[i], msg.values[i]);
Refinement ghost_ref(msg.elements[i], msg.values[i].ref_type);
ghost_ref.SetScaleForType(msg.values[i].scale);
NCMesh::RefineElement(ghost_ref);
}
}
@@ -2556,7 +2624,8 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
for (int i = 0; i < rank_neighbors.Size(); i++)
{
int elem = rank_neighbors[i];
msg.AddElementRank(elem, new_ranks[elements[elem].index]);
const Element &el = elements[elem];
msg.AddElement(elem, new_ranks[el.index], el.attribute);
}
msg.Isend(rank, MyComm);
@@ -2579,7 +2648,9 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
{
int ghost_index = elements[msg.elements[i]].index;
MFEM_ASSERT(element_type[ghost_index] == 2, "");
new_ranks[ghost_index] = msg.values[i];
const ElementRankAndAttribute &value = msg.values[i];
new_ranks[ghost_index] = value.rank;
elements[msg.elements[i]].attribute = value.attribute;
}
}
@@ -2650,7 +2721,7 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
if ((element_type[el.index] & 1) || el.rank != rank)
{
msg.AddElementRank(elem, el.rank);
msg.AddElement(elem, el.rank, el.attribute);
}
// NOTE: we skip 'ghosts' that are of the receiver's rank because
// they are not really ghosts and would get sent multiple times,
@@ -2702,10 +2773,12 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
for (int i = 0; i < msg.Size(); i++)
{
int elem_rank = msg.values[i];
elements[msg.elements[i]].rank = elem_rank;
const ElementRankAndAttribute &value = msg.values[i];
Element &el = elements[msg.elements[i]];
el.rank = value.rank;
el.attribute = value.attribute;
if (elem_rank == MyRank) { received_elements++; }
if (value.rank == MyRank) { received_elements++; }
}
// save the ranks we received from, for later use in RecvRebalanceDofs
@@ -2741,7 +2814,10 @@ void ParNCMesh::RedistributeElements(Array<int> &new_ranks, int target_elements,
for (int i = 0; i < msg.Size(); i++)
{
elements[msg.elements[i]].rank = msg.values[i];
const ElementRankAndAttribute &value = msg.values[i];
Element &el = elements[msg.elements[i]];
el.rank = value.rank;
el.attribute = value.attribute;
}
// save the ranks we received from, for later use in RecvRebalanceDofs
+42 -15
View File
@@ -497,11 +497,27 @@ protected: // implementation
/** Used by ParNCMesh::Refine() to inform neighbors about refinements at
* the processor boundary. This keeps their ghost layers synchronized.
*/
class NeighborRefinementMessage : public ElementValueMessage<char, false,
VarMessageTag::NEIGHBOR_REFINEMENT_VM>
struct NeighborRefinement
{
char ref_type;
real_t scale[3];
};
class NeighborRefinementMessage
: public ElementValueMessage<NeighborRefinement, false,
VarMessageTag::NEIGHBOR_REFINEMENT_VM>
{
public:
void AddRefinement(int elem, char ref_type) { Add(elem, ref_type); }
void AddRefinement(int elem, const Refinement &ref)
{
NeighborRefinement data{};
data.ref_type = ref.GetType();
for (int i = 0; i < 3; i++)
{
data.scale[i] = ref.s[i];
}
Add(elem, data);
}
typedef std::map<int, NeighborRefinementMessage> Map;
};
@@ -515,26 +531,36 @@ protected: // implementation
typedef std::map<int, NeighborDerefinementMessage> Map;
};
/** Used in Step 2 of Rebalance() to synchronize new rank assignments in
* the ghost layer.
struct ElementRankAndAttribute
{
int rank;
int attribute;
};
/** Used in RedistributeElements() to synchronize new rank assignments and
* element attributes in the ghost layer.
*/
class NeighborElementRankMessage : public ElementValueMessage<int, false,
class NeighborElementRankMessage :
public ElementValueMessage<ElementRankAndAttribute, false,
VarMessageTag::NEIGHBOR_ELEMENT_RANK_VM>
{
public:
void AddElementRank(int elem, int rank) { Add(elem, rank); }
void AddElement(int elem, int rank, int attribute)
{ Add(elem, {rank, attribute}); }
typedef std::map<int, NeighborElementRankMessage> Map;
};
/** Used by Rebalance() to send elements and their ranks. Note that
/** Used by Rebalance() to send elements, ranks, and attributes. Note that
* RefTypes == true which means the refinement hierarchy will be recreated
* on the receiving side.
*/
class RebalanceMessage : public ElementValueMessage<int, true,
class RebalanceMessage :
public ElementValueMessage<ElementRankAndAttribute, true,
VarMessageTag::REBALANCE_VM>
{
public:
void AddElementRank(int elem, int rank) { Add(elem, rank); }
void AddElement(int elem, int rank, int attribute)
{ Add(elem, {rank, attribute}); }
typedef std::map<int, RebalanceMessage> Map;
};
@@ -602,7 +628,8 @@ protected: // implementation
/** For the face with ordered vertices vn* and neighboring element @a elem,
check whether the other neighboring element (if it exists) is marked for
a horizontal refinement conflicting with a vertical split. */
void CheckRefAnisoFace(int elem, int vn1, int vn2, int vn3, int vn4,
void CheckRefAnisoFace(const Refinement &ref, int elem,
int vn1, int vn2, int vn3, int vn4,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
std::set<int> &conflicts);
@@ -611,7 +638,8 @@ protected: // implementation
neighboring element @a elem, check whether the other neighboring element
(if it exists) is marked for a refinement conflicting with an isotropic
refinement of the face. */
void CheckRefIsoFace(int elem, int vn1, int vn2, int vn3, int vn4,
void CheckRefIsoFace(const Refinement &ref, int elem,
int vn1, int vn2, int vn3, int vn4,
int en1, int en2, int en3, int en4,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
@@ -622,9 +650,8 @@ protected: // implementation
const std::map<int, int> &elemToRef,
std::set<int> &conflicts);
/** Check whether the refinement of the element with index @a elem and type
@a ref_type would cause a conflict. */
void CheckRefinement(int elem, char ref_type,
/// Check whether the input refinement would cause a conflict.
void CheckRefinement(int elem, const Refinement &ref,
const Array<Refinement> &refinements,
const std::map<int, int> &elemToRef,
std::set<int> &conflicts);
+2
View File
@@ -52,6 +52,8 @@ endif
.SUFFIXES:
.SUFFIXES: .o .cpp .mk
.PHONY: all lib-common clean clean-build clean-exec
# Keeping the *.o files fixes an issue with the MacOS version of 'make'.
.PRECIOUS: %.o
# Remove built-in rules
%: %.cpp
+5
View File
@@ -151,6 +151,10 @@ if (MFEM_USE_MPI)
MAIN phpref.cpp
LIBRARIES mfem)
add_mfem_miniapp(pref321
MAIN pref321.cpp
LIBRARIES mfem)
# Add parallel tests.
if (MFEM_ENABLE_TESTING)
set(PARALLEL_TESTS
@@ -160,6 +164,7 @@ if (MFEM_USE_MPI)
fit-node-position
pminimal-surface
phpref
pref321
)
# Meshing miniapps that return MFEM_SKIP_RETURN_VALUE in some cases:
set(SKIP_TESTS)
+3 -1
View File
@@ -24,7 +24,7 @@ SEQ_MINIAPPS = mobius-strip klein-bottle toroid trimmer twist mesh-explorer\
shaper extruder mesh-optimizer minimal-surface polar-nc reflector\
ref321 mesh-quality hpref
PAR_MINIAPPS = pmesh-optimizer pminimal-surface pmesh-fitting fit-node-position\
phpref mesh-bounding-boxes
phpref pref321 mesh-bounding-boxes
ifeq ($(MFEM_USE_MPI),NO)
MINIAPPS = $(SEQ_MINIAPPS)
else
@@ -99,6 +99,8 @@ hpref-test-seq: hpref
@$(call mfem-test,$<,, Serial hp-refinement)
phpref-test-par: phpref
@$(call mfem-test,$<, $(RUN_MPI), Parallel hp-refinement)
pref321-test-par: pref321
@$(call mfem-test,$<, $(RUN_MPI), Parallel 3:1 refinement)
mesh-bounding-boxes-test-par: mesh-bounding-boxes
@$(call mfem-test,$<, $(RUN_MPI), Parallel bounding boxes)
ref321-test-seq: ref321
+336
View File
@@ -0,0 +1,336 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
//
// -----------------------------------------------------------------
// 3:1 Refinement Miniapp: Parallel 3:1 anisotropic mesh refinements
// -----------------------------------------------------------------
//
// This miniapp performs random 3:1 refinements of a quadrilateral or hexahedral
// mesh. A diffusion equation is solved in an H1 finite element space defined on
// the refined mesh, and its continuity is verified across local and shared
// faces.
//
// Compile with: make pref321
//
// Sample runs: mpirun -np 4 pref321 -mm -dim 2 -o 2 -r 100
// mpirun -np 4 pref321 -mm -dim 3 -o 2 -r 100
// mpirun -np 4 pref321 -m ../../data/star.mesh -o 2 -r 100
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
real_t CheckH1Continuity(ParGridFunction &x);
// Find the two children of parent element `elem` after its refinement in one
// direction.
void FindChildren(const Mesh &mesh, int elem, Array<int> &children)
{
const CoarseFineTransformations &cf = mesh.ncmesh->GetRefinementTransforms();
MFEM_ASSERT(mesh.GetNE() == cf.embeddings.Size(), "");
// Note that row `elem` of the table constructed by cf.MakeCoarseToFineTable
// is an alternative to this global loop, but constructing the table is also
// a global operation with global storage.
for (int i = 0; i < mesh.GetNE(); i++)
{
const int p = cf.embeddings[i].parent;
if (p == elem)
{
children.Append(i);
}
}
}
// Refine 3:1 via 2 refinements with scalings 2/3 and 1/2.
void Refine31(Mesh &mesh, int elem, char type)
{
Array<Refinement> refs; // Refinement is defined in ncmesh.hpp
refs.Append(Refinement(elem, type, 2.0 / 3.0));
mesh.GeneralRefinement(refs);
// Find the elements with parent `elem`
Array<int> children;
FindChildren(mesh, elem, children);
MFEM_ASSERT(children.Size() == 2, "");
const int elem1 = children[0];
refs.SetSize(0);
refs.Append(Refinement(elem1, type)); // Default scaling of 0.5
mesh.GeneralRefinement(refs);
}
// Randomly select elements for 3:1 refinements in random directions.
void TestAnisoRefRandom(int num_refs, int dim, ParMesh &mesh, int myid,
int seed = 0)
{
std::mt19937 gen(seed);
for (int i = 0; i < num_refs; i++)
{
const int elem = gen() % mesh.GetNE();
const int t = gen() % dim;
auto type = t == 0 ? Refinement::X :
(t == 1 ? Refinement::Y : Refinement::Z);
// In 3D, check for conflicts in the parallel refinements.
if (dim == 3)
{
std::set<int> conflicts; // Indices in refs of conflicting elements
Array<Refinement> refs;
refs.Append(Refinement(elem, type));
const bool conflict = mesh.AnisotropicConflict(refs, conflicts);
if (conflict)
{
if (myid == 0)
cout << "Anisotropic conflict on iteration " << i
<< ", retrying\n";
i--;
continue;
}
}
Refine31(mesh, elem, type);
}
mesh.EnsureNodes();
mesh.SetScaledNCMesh();
}
int main(int argc, char *argv[])
{
Mpi::Init(argc, argv);
Hypre::Init();
const int num_procs = Mpi::WorldSize();
const int myid = Mpi::WorldRank();
// 1. Parse command-line options.
const char *mesh_file = "../../data/star.mesh";
int order = 1;
bool visualization = true;
bool makeMesh = false;
int num_refs = 1;
int tdim = 2; // Mesh dimension for Cartesian meshes.
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&makeMesh, "-mm", "--make-mesh", "-no-mm",
"--no-make-mesh", "Create Cartesian mesh");
args.AddOption(&tdim, "-dim", "--dimension", "Dimension for Cartesian mesh");
args.AddOption(&num_refs, "-r", "--refs", "Number of 3:1 refinements");
args.Parse();
if (!args.Good())
{
if (myid == 0)
{
args.PrintUsage(cout);
}
return 1;
}
if (myid == 0)
{
args.PrintOptions(cout);
}
// 2. Create or read the serial mesh on all ranks, then apply the same
// deterministic 3:1 refinement sequence before partitioning it.
Mesh mesh;
if (makeMesh)
{
mesh = tdim == 3 ? Mesh::MakeCartesian3D(2, 2, 2, Element::HEXAHEDRON) :
Mesh::MakeCartesian2D(2, 2, Element::QUADRILATERAL);
}
else
{
mesh = Mesh::LoadFromFile(mesh_file, 1, 1);
}
const int dim = mesh.Dimension();
mesh.EnsureNCMesh();
mesh.SetScaledNCMesh();
// 3. Partition the refined serial mesh.
ParMesh pmesh(MPI_COMM_WORLD, mesh);
mesh.Clear();
TestAnisoRefRandom(num_refs, dim, pmesh, myid, myid);
// 4. Define a parallel H1 finite element space and report its global size.
H1_FECollection fec(order, dim);
ParFiniteElementSpace fespace(&pmesh, &fec);
if (myid == 0)
{
cout << "Number of finite element unknowns: "
<< fespace.GlobalTrueVSize() << endl;
}
// 5. Assemble and solve the Poisson problem, following ex1p.
ParGridFunction x(&fespace);
x = 0.0;
ParLinearForm b(&fespace);
ConstantCoefficient one(1.0);
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
ParBilinearForm a(&fespace);
a.AddDomainIntegrator(new DiffusionIntegrator());
a.Assemble();
OperatorPtr A;
Vector B, X;
Array<int> ess_tdof_list;
if (pmesh.bdr_attributes.Size())
{
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
ess_bdr = 0;
pmesh.MarkExternalBoundaries(ess_bdr);
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
HypreBoomerAMG M;
CGSolver cg(MPI_COMM_WORLD);
cg.SetPreconditioner(M);
cg.SetOperator(*A);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
cg.Mult(B, X);
a.RecoverFEMSolution(X, b, x);
// 6. Verify the continuity of the solution in H1 over local and shared
// faces and compute the global maximum jump.
const real_t h1err = CheckH1Continuity(x);
if (myid == 0)
{
cout << "Error of H1 continuity: " << h1err << endl;
}
MFEM_VERIFY(h1err < 1.0e-7, "H1 discontinuity found");
// 7. Save the refined mesh and the solution in parallel. This output can
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_name << "sol." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh.Print(mesh_ofs);
ofstream sol_ofs(sol_name.str().c_str());
sol_ofs.precision(8);
x.Save(sol_ofs);
}
// 8. Send the parallel solution to GLVis.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << pmesh << x << flush;
}
return 0;
}
real_t CheckH1Continuity(ParGridFunction &x)
{
const ParFiniteElementSpace *pfes = x.ParFESpace();
ParMesh *pmesh = pfes->GetParMesh();
const int dim = pmesh->Dimension();
real_t errorMax = 0.0;
// Shared-face values require face-neighbor data.
x.ExchangeFaceNbrData();
// First handle faces for which both elements are local to this rank.
for (int f = 0; f < pmesh->GetNumFaces(); f++)
{
const auto info = pmesh->GetFaceInformation(f);
if (!info.IsLocal())
{
continue;
}
FaceElementTransformations *FT = pmesh->GetFaceElementTransformations(f);
const int faceOrder = dim == 3 ? pfes->GetFaceOrder(f) :
pfes->GetEdgeOrder(f);
const IntegrationRule &ir = IntRules.Get(FT->FaceGeom, 2 * faceOrder);
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &fip = ir.IntPoint(i);
IntegrationPoint ip1, ip2;
FT->Loc1.Transform(fip, ip1);
FT->Loc2.Transform(fip, ip2);
const real_t v1 = x.GetValue(*FT->Elem1, ip1);
const real_t v2 = x.GetValue(*FT->Elem2, ip2);
errorMax = std::max(errorMax, std::abs(v1 - v2));
}
}
// Then check partition interfaces. Conforming shared faces are handled on
// the lower-rank side, while shared slave nonconforming faces are handled
// only on the slave side and therefore do not need additional filtering.
for (int sf = 0; sf < pmesh->GetNSharedFaces(); sf++)
{
const int f = pmesh->GetSharedFace(sf);
const auto info = pmesh->GetFaceInformation(f);
if (!info.IsShared())
{
continue;
}
FaceElementTransformations *FT = pmesh->GetSharedFaceTransformations(sf);
const int faceOrder = dim == 3 ? pfes->GetFaceOrder(f) :
pfes->GetEdgeOrder(f);
const IntegrationRule &ir = IntRules.Get(FT->FaceGeom, 2 * faceOrder);
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &fip = ir.IntPoint(i);
IntegrationPoint ip1, ip2;
FT->Loc1.Transform(fip, ip1);
FT->Loc2.Transform(fip, ip2);
const real_t v1 = x.GetValue(*FT->Elem1, ip1);
const real_t v2 = x.GetValue(*FT->Elem2, ip2);
errorMax = std::max(errorMax, std::abs(v1 - v2));
}
}
MPI_Allreduce(MPI_IN_PLACE, &errorMax, 1, MFEM_MPI_REAL_T, MPI_MAX,
pmesh->GetComm());
return errorMax;
}
+5 -13
View File
@@ -71,22 +71,14 @@ void Refine31(Mesh & mesh, int elem, char type)
mesh.GeneralRefinement(refs);
}
// Deterministic, somewhat random integer generator
int MyRand(int & s)
{
s++;
const double a = 1000 * sin(s * 1.1234 * M_PI);
return int(std::abs(a));
}
// Randomly select elements for 3:1 refinements in random directions.
void TestAnisoRefRandom(int iter, int dim, Mesh & mesh)
void TestAnisoRefRandom(int num_refs, int dim, Mesh & mesh)
{
int seed = 0;
for (int i = 0; i < iter; i++)
std::mt19937 gen(1);
for (int i = 0; i < num_refs; i++)
{
const int elem = MyRand(seed) % mesh.GetNE();
const int t = MyRand(seed) % dim;
const auto elem = gen() % mesh.GetNE();
const auto t = gen() % dim;
auto type = t == 0 ? Refinement::X :
(t == 1 ? Refinement::Y : Refinement::Z);
Refine31(mesh, elem, type);
+1 -1
View File
@@ -68,7 +68,7 @@ multidomain-test-par: multidomain
multidomain_nd-test-par: multidomain_nd
@$(call mfem-test,$<, $(RUN_MPI), Multidomain ND miniapp,-tf 0.001)
multidomain_rt-test-par: multidomain_rt
@$(call mfem-test,$<, $(RUN_MPI), Multidomain RT iniapp,-tf 0.001)
@$(call mfem-test,$<, $(RUN_MPI), Multidomain RT miniapp,-tf 0.001)
# Generate an error message if the MFEM library is not built and exit
$(MFEM_LIB_FILE):
+5 -1
View File
@@ -32,7 +32,11 @@
// Custom benchmark arguments generator
static void CustomArguments(bm::Benchmark *b) noexcept
{
constexpr int MAX_NDOFS = 16 * 1024 * (mfem_use_gpu ? 1024 : 8);
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
constexpr int MAX_NDOFS = 16 * 1024 * 1024;
#else
constexpr int MAX_NDOFS = 16 * 1024 * 8;
#endif
const auto orders = { 7, 6, 5, 4, 3, 2, 1 };
+16
View File
@@ -39,6 +39,7 @@ set(UNIT_TESTS_SRCS
dfem/test_divergence.cpp
dfem/test_lvector_interface.cpp
dfem/test_mass.cpp
dfem/test_tuple.cpp
general/test_array.cpp
general/test_scan.cpp
general/test_arrays_by_name.cpp
@@ -140,6 +141,7 @@ set(UNIT_TESTS_SRCS
fem/test_lor_batched.cpp
fem/test_lor_dg.cpp
fem/test_lor.cpp
fem/test_mixedsesqform.cpp
fem/test_nonlinearform.cpp
fem/test_operatorjacobismoother.cpp
fem/test_oscillation.cpp
@@ -168,6 +170,20 @@ set(UNIT_TESTS_SRCS
fem/test_transfer.cpp
fem/test_var_order.cpp
fem/test_white_noise.cpp
fem/specializations/test_diffusion_integ.cpp
fem/specializations/test_mass_integ.cpp
fem/specializations/test_convection_integ.cpp
fem/specializations/test_vecmass_integ.cpp
fem/specializations/test_curlcurl_integ.cpp
fem/specializations/test_vecdiffusion_integ.cpp
fem/specializations/test_dgtrace_integ.cpp
fem/specializations/test_dgdiffusion_integ.cpp
fem/specializations/test_dgmassinv.cpp
fem/specializations/test_qinterp_det.cpp
fem/specializations/test_qinterp_eval.cpp
fem/specializations/test_qinterp_grad.cpp
fem/specializations/test_qinterp_tensoreval.cpp
fem/specializations/test_qinterp_eval_hdiv.cpp
enzyme/compatibility.cpp
# The following are tested separately (keep the comment as a reminder).
# This list can be updated using (in bash):
+379
View File
@@ -0,0 +1,379 @@
MFEM NC mesh v1.0
# NCMesh supported geometry types:
# SEGMENT = 1
# TRIANGLE = 2
# SQUARE = 3
# TETRAHEDRON = 4
# CUBE = 5
# PRISM = 6
# PYRAMID = 7
dimension
3
rank
0
# rank attr geom ref_type nodes/children
elements
75
0 1 5 0 0 1 5 4 16 17 21 20
0 1 5 0 16 17 21 20 32 33 37 36
-1 1 5 7 59 60 61 62 63 64 65 66
0 1 5 0 1 2 6 5 17 18 22 21
-1 1 5 7 27 28 29 30 31 32 33 34
0 1 5 0 21 22 26 25 37 38 42 41
-1 1 5 7 43 44 45 46 47 48 49 50
0 1 5 0 4 5 9 8 20 21 25 24
0 1 5 0 8 9 13 12 24 25 29 28
0 1 5 0 24 25 29 28 40 41 45 44
0 1 5 0 9 10 14 13 25 26 30 29
-1 1 5 7 67 68 69 70 71 72 73 74
0 1 5 0 41 42 46 45 57 58 62 61
0 1 5 0 40 41 45 44 56 57 61 60
0 1 5 0 36 37 41 40 52 53 57 56
-1 1 5 7 35 36 37 38 39 40 41 42
0 1 5 0 32 33 37 36 48 49 53 52
0 1 5 0 33 34 38 37 49 50 54 53
0 1 5 0 34 35 39 38 50 51 55 54
0 1 5 0 38 39 43 42 54 55 59 58
0 1 5 0 42 43 47 46 58 59 63 62
0 1 5 0 26 27 31 30 42 43 47 46
0 1 5 0 10 11 15 14 26 27 31 30
0 1 5 0 6 7 11 10 22 23 27 26
-1 1 5 7 51 52 53 54 55 56 57 58
0 1 5 0 18 19 23 22 34 35 39 38
0 1 5 0 2 3 7 6 18 19 23 22
0 1 5 0 5 94 208 99 74 209 214 212
0 1 5 0 94 6 97 208 209 96 210 214
0 1 5 0 208 97 10 98 214 210 103 211
0 1 5 0 99 208 98 9 212 214 211 104
0 1 5 0 74 209 214 212 21 86 213 102
0 1 5 0 209 96 210 214 86 22 100 213
0 1 5 0 214 210 103 211 213 100 26 101
0 1 5 0 212 214 211 104 102 213 101 25
0 1 5 0 37 89 269 107 156 270 275 273
0 1 5 0 89 38 105 269 270 159 271 275
0 1 5 0 269 105 42 106 275 271 144 272
0 1 5 0 107 269 106 41 273 275 272 143
0 1 5 0 156 270 275 273 53 157 274 153
0 1 5 0 270 159 271 275 157 54 158 274
0 1 5 0 275 271 144 272 274 158 58 139
0 1 5 0 273 275 272 143 153 274 139 57
0 1 5 0 20 70 330 111 83 331 336 334
0 1 5 0 70 21 102 330 331 82 332 336
0 1 5 0 330 102 25 110 336 332 109 333
0 1 5 0 111 330 110 24 334 336 333 114
0 1 5 0 83 331 336 334 36 78 335 113
0 1 5 0 331 82 332 336 78 37 107 335
0 1 5 0 336 332 109 333 335 107 41 112
0 1 5 0 334 336 333 114 113 335 112 40
0 1 5 0 22 198 387 100 91 388 393 391
0 1 5 0 198 23 199 387 388 201 389 393
0 1 5 0 387 199 27 186 393 389 189 390
0 1 5 0 100 387 186 26 391 393 390 108
0 1 5 0 91 388 393 391 38 170 392 105
0 1 5 0 388 201 389 393 170 39 176 392
0 1 5 0 393 389 189 390 392 176 43 177
0 1 5 0 391 393 390 108 105 392 177 42
0 1 5 0 17 84 444 69 81 445 450 448
0 1 5 0 84 18 85 444 445 90 446 450
0 1 5 0 444 85 22 86 450 446 91 447
0 1 5 0 69 444 86 21 448 450 447 82
0 1 5 0 81 445 450 448 33 87 449 77
0 1 5 0 445 90 446 450 87 34 88 449
0 1 5 0 450 446 91 447 449 88 38 89
0 1 5 0 448 450 447 82 77 449 89 37
0 1 5 0 25 101 497 121 109 498 503 501
0 1 5 0 101 26 133 497 498 108 499 503
0 1 5 0 497 133 30 134 503 499 138 500
0 1 5 0 121 497 134 29 501 503 500 129
0 1 5 0 109 498 503 501 41 106 502 126
0 1 5 0 498 108 499 503 106 42 136 502
0 1 5 0 503 499 138 500 502 136 46 137
0 1 5 0 501 503 500 129 126 502 137 45
# attr geom nodes
boundary
72
1 3 4 5 1 0
1 3 0 1 17 16
1 3 4 0 16 20
1 3 16 17 33 32
1 3 20 16 32 36
1 3 5 6 2 1
1 3 1 2 18 17
1 3 8 9 5 4
1 3 8 4 20 24
1 3 12 13 9 8
1 3 13 12 28 29
1 3 12 8 24 28
1 3 29 28 44 45
1 3 28 24 40 44
1 3 13 14 10 9
1 3 14 13 29 30
1 3 46 45 61 62
1 3 57 58 62 61
1 3 45 44 60 61
1 3 44 40 56 60
1 3 56 57 61 60
1 3 40 36 52 56
1 3 52 53 57 56
1 3 32 33 49 48
1 3 36 32 48 52
1 3 48 49 53 52
1 3 33 34 50 49
1 3 49 50 54 53
1 3 34 35 51 50
1 3 35 39 55 51
1 3 50 51 55 54
1 3 39 43 59 55
1 3 54 55 59 58
1 3 43 47 63 59
1 3 47 46 62 63
1 3 58 59 63 62
1 3 27 31 47 43
1 3 31 30 46 47
1 3 14 15 11 10
1 3 11 15 31 27
1 3 15 14 30 31
1 3 10 11 7 6
1 3 7 11 27 23
1 3 18 19 35 34
1 3 19 23 39 35
1 3 6 7 3 2
1 3 2 3 19 18
1 3 3 7 23 19
2 3 99 208 94 5
2 3 208 97 6 94
2 3 98 10 97 208
2 3 9 98 208 99
2 3 53 157 274 153
2 3 157 54 158 274
2 3 274 158 58 139
2 3 153 274 139 57
2 3 111 20 83 334
2 3 24 111 334 114
2 3 334 83 36 113
2 3 114 334 113 40
2 3 23 199 389 201
2 3 199 27 189 389
2 3 201 389 176 39
2 3 389 189 43 176
2 3 17 84 445 81
2 3 84 18 90 445
2 3 81 445 87 33
2 3 445 90 34 87
2 3 30 134 500 138
2 3 134 29 129 500
2 3 138 500 137 46
2 3 500 129 45 137
# vert_id p1 p2
vertex_parents
102
69 17 21
70 20 21
74 5 21
77 33 37
78 36 37
81 17 33
82 21 37
83 20 36
84 17 18
85 18 22
86 21 22
87 33 34
88 34 38
89 37 38
90 18 34
91 22 38
94 5 6
96 6 22
97 6 10
98 9 10
99 5 9
100 22 26
101 25 26
102 21 25
103 10 26
104 9 25
105 38 42
106 41 42
107 37 41
108 26 42
109 25 41
110 24 25
111 20 24
112 40 41
113 36 40
114 24 40
121 25 29
126 41 45
129 29 45
133 26 30
134 29 30
136 42 46
137 45 46
138 30 46
139 57 58
143 41 57
144 42 58
153 53 57
156 37 53
157 53 54
158 54 58
159 38 54
170 38 39
176 39 43
177 42 43
186 26 27
189 27 43
198 22 23
199 23 27
201 23 39
208 97 99
209 74 96
210 96 103
211 103 104
212 74 104
213 100 102
214 209 211
269 105 107
270 156 159
271 144 159
272 143 144
273 143 156
274 153 158
275 270 272
330 102 111
331 82 83
332 82 109
333 109 114
334 83 114
335 107 113
336 331 333
387 100 199
388 91 201
389 189 201
390 108 189
391 91 108
392 105 176
393 388 390
444 69 85
445 81 90
446 90 91
447 82 91
448 81 82
449 77 88
450 445 447
497 121 133
498 108 109
499 108 138
500 129 138
501 109 129
502 126 136
503 498 500
# root element orientation
root_state
27
0
1
1
15
15
6
6
22
15
8
12
10
10
18
18
13
7
22
22
15
16
16
16
7
8
6
21
# top-level node coordinates
coordinates
64
3
0 0 0
0.33333333 0 0
0.66666667 0 0
1 0 0
0 0.33333333 0
0.33333333 0.33333333 0
0.66666667 0.33333333 0
1 0.33333333 0
0 0.66666667 0
0.33333333 0.66666667 0
0.66666667 0.66666667 0
1 0.66666667 0
0 1 0
0.33333333 1 0
0.66666667 1 0
1 1 0
0 0 0.33333333
0.33333333 0 0.33333333
0.66666667 0 0.33333333
1 0 0.33333333
0 0.33333333 0.33333333
0.33333333 0.33333333 0.33333333
0.66666667 0.33333333 0.33333333
1 0.33333333 0.33333333
0 0.66666667 0.33333333
0.33333333 0.66666667 0.33333333
0.66666667 0.66666667 0.33333333
1 0.66666667 0.33333333
0 1 0.33333333
0.33333333 1 0.33333333
0.66666667 1 0.33333333
1 1 0.33333333
0 0 0.66666667
0.33333333 0 0.66666667
0.66666667 0 0.66666667
1 0 0.66666667
0 0.33333333 0.66666667
0.33333333 0.33333333 0.66666667
0.66666667 0.33333333 0.66666667
1 0.33333333 0.66666667
0 0.66666667 0.66666667
0.33333333 0.66666667 0.66666667
0.66666667 0.66666667 0.66666667
1 0.66666667 0.66666667
0 1 0.66666667
0.33333333 1 0.66666667
0.66666667 1 0.66666667
1 1 0.66666667
0 0 1
0.33333333 0 1
0.66666667 0 1
1 0 1
0 0.33333333 1
0.33333333 0.33333333 1
0.66666667 0.33333333 1
1 0.33333333 1
0 0.66666667 1
0.33333333 0.66666667 1
0.66666667 0.66666667 1
1 0.66666667 1
0 1 1
0.33333333 1 1
0.66666667 1 1
1 1 1
mfem_mesh_end
+555
View File
@@ -0,0 +1,555 @@
MFEM NC mesh v1.0
# NCMesh supported geometry types:
# SEGMENT = 1
# TRIANGLE = 2
# SQUARE = 3
# TETRAHEDRON = 4
# CUBE = 5
# PRISM = 6
# PYRAMID = 7
dimension
3
rank
0
# rank attr geom ref_type nodes/children
elements
258
0 1 4 0 21 0 5 1
0 1 4 0 21 0 1 17
0 1 4 0 21 0 17 16
0 1 4 0 21 0 4 5
0 1 4 0 21 0 20 4
0 1 4 0 21 0 16 20
0 1 4 0 22 1 6 2
0 1 4 0 22 1 2 18
0 1 4 0 22 1 18 17
0 1 4 0 22 1 5 6
0 1 4 0 22 1 21 5
0 1 4 0 22 1 17 21
0 1 4 0 23 2 7 3
0 1 4 0 23 2 3 19
0 1 4 0 23 2 19 18
0 1 4 0 23 2 6 7
0 1 4 0 23 2 22 6
0 1 4 0 23 2 18 22
0 1 4 0 25 4 9 5
0 1 4 0 25 4 5 21
0 1 4 0 25 4 21 20
0 1 4 0 25 4 8 9
0 1 4 0 25 4 24 8
0 1 4 0 25 4 20 24
-1 1 4 7 170 171 172 173 174 175 176 177
0 1 4 0 26 5 6 22
0 1 4 0 26 5 22 21
-1 1 4 7 162 163 164 165 166 167 168 169
0 1 4 0 26 5 25 9
0 1 4 0 26 5 21 25
0 1 4 0 27 6 11 7
0 1 4 0 27 6 7 23
0 1 4 0 27 6 23 22
0 1 4 0 27 6 10 11
0 1 4 0 27 6 26 10
0 1 4 0 27 6 22 26
0 1 4 0 29 8 13 9
0 1 4 0 29 8 9 25
0 1 4 0 29 8 25 24
0 1 4 0 29 8 12 13
0 1 4 0 29 8 28 12
0 1 4 0 29 8 24 28
0 1 4 0 30 9 14 10
0 1 4 0 30 9 10 26
0 1 4 0 30 9 26 25
0 1 4 0 30 9 13 14
0 1 4 0 30 9 29 13
0 1 4 0 30 9 25 29
0 1 4 0 31 10 15 11
0 1 4 0 31 10 11 27
0 1 4 0 31 10 27 26
0 1 4 0 31 10 14 15
0 1 4 0 31 10 30 14
0 1 4 0 31 10 26 30
0 1 4 0 37 16 21 17
0 1 4 0 37 16 17 33
0 1 4 0 37 16 33 32
0 1 4 0 37 16 20 21
0 1 4 0 37 16 36 20
0 1 4 0 37 16 32 36
0 1 4 0 38 17 22 18
-1 1 4 7 226 227 228 229 230 231 232 233
-1 1 4 7 234 235 236 237 238 239 240 241
0 1 4 0 38 17 21 22
0 1 4 0 38 17 37 21
0 1 4 0 38 17 33 37
0 1 4 0 39 18 23 19
0 1 4 0 39 18 19 35
0 1 4 0 39 18 35 34
0 1 4 0 39 18 22 23
0 1 4 0 39 18 38 22
0 1 4 0 39 18 34 38
0 1 4 0 41 20 25 21
0 1 4 0 41 20 21 37
0 1 4 0 41 20 37 36
0 1 4 0 41 20 24 25
-1 1 4 7 202 203 204 205 206 207 208 209
-1 1 4 7 194 195 196 197 198 199 200 201
0 1 4 0 42 21 26 22
0 1 4 0 42 21 22 38
0 1 4 0 42 21 38 37
0 1 4 0 42 21 25 26
0 1 4 0 42 21 41 25
0 1 4 0 42 21 37 41
-1 1 4 7 210 211 212 213 214 215 216 217
-1 1 4 7 218 219 220 221 222 223 224 225
0 1 4 0 43 22 39 38
0 1 4 0 43 22 26 27
0 1 4 0 43 22 42 26
0 1 4 0 43 22 38 42
0 1 4 0 45 24 29 25
0 1 4 0 45 24 25 41
0 1 4 0 45 24 41 40
0 1 4 0 45 24 28 29
0 1 4 0 45 24 44 28
0 1 4 0 45 24 40 44
0 1 4 0 46 25 30 26
0 1 4 0 46 25 26 42
0 1 4 0 46 25 42 41
-1 1 4 7 250 251 252 253 254 255 256 257
-1 1 4 7 242 243 244 245 246 247 248 249
0 1 4 0 46 25 41 45
0 1 4 0 47 26 31 27
0 1 4 0 47 26 27 43
0 1 4 0 47 26 43 42
0 1 4 0 47 26 30 31
0 1 4 0 47 26 46 30
0 1 4 0 47 26 42 46
0 1 4 0 53 32 37 33
0 1 4 0 53 32 33 49
0 1 4 0 53 32 49 48
0 1 4 0 53 32 36 37
0 1 4 0 53 32 52 36
0 1 4 0 53 32 48 52
0 1 4 0 54 33 38 34
0 1 4 0 54 33 34 50
0 1 4 0 54 33 50 49
0 1 4 0 54 33 37 38
0 1 4 0 54 33 53 37
0 1 4 0 54 33 49 53
0 1 4 0 55 34 39 35
0 1 4 0 55 34 35 51
0 1 4 0 55 34 51 50
0 1 4 0 55 34 38 39
0 1 4 0 55 34 54 38
0 1 4 0 55 34 50 54
0 1 4 0 57 36 41 37
0 1 4 0 57 36 37 53
0 1 4 0 57 36 53 52
0 1 4 0 57 36 40 41
0 1 4 0 57 36 56 40
0 1 4 0 57 36 52 56
0 1 4 0 58 37 42 38
0 1 4 0 58 37 38 54
-1 1 4 7 178 179 180 181 182 183 184 185
0 1 4 0 58 37 41 42
0 1 4 0 58 37 57 41
-1 1 4 7 186 187 188 189 190 191 192 193
0 1 4 0 59 38 43 39
0 1 4 0 59 38 39 55
0 1 4 0 59 38 55 54
0 1 4 0 59 38 42 43
0 1 4 0 59 38 58 42
0 1 4 0 59 38 54 58
0 1 4 0 61 40 45 41
0 1 4 0 61 40 41 57
0 1 4 0 61 40 57 56
0 1 4 0 61 40 44 45
0 1 4 0 61 40 60 44
0 1 4 0 61 40 56 60
0 1 4 0 62 41 46 42
0 1 4 0 62 41 42 58
0 1 4 0 62 41 58 57
0 1 4 0 62 41 45 46
0 1 4 0 62 41 61 45
0 1 4 0 62 41 57 61
0 1 4 0 63 42 47 43
0 1 4 0 63 42 43 59
0 1 4 0 63 42 59 58
0 1 4 0 63 42 46 47
0 1 4 0 63 42 62 46
0 1 4 0 63 42 58 62
0 1 4 0 26 125 132 126
0 1 4 0 125 5 115 128
0 1 4 0 132 115 9 133
0 1 4 0 126 128 133 10
0 1 4 0 125 133 132 126
0 1 4 0 125 133 126 128
0 1 4 0 125 133 128 115
0 1 4 0 125 133 115 132
0 1 4 0 26 125 126 127
0 1 4 0 125 5 128 95
0 1 4 0 126 128 10 129
0 1 4 0 127 95 129 6
0 1 4 0 125 129 126 127
0 1 4 0 125 129 127 95
0 1 4 0 125 129 95 128
0 1 4 0 125 129 128 126
0 1 4 0 58 305 308 309
0 1 4 0 305 37 283 262
0 1 4 0 308 283 54 284
0 1 4 0 309 262 284 53
0 1 4 0 305 284 308 309
0 1 4 0 305 284 309 262
0 1 4 0 305 284 262 283
0 1 4 0 305 284 283 308
0 1 4 0 58 305 309 311
0 1 4 0 305 37 262 297
0 1 4 0 309 262 53 298
0 1 4 0 311 297 298 57
0 1 4 0 305 298 309 311
0 1 4 0 305 298 311 297
0 1 4 0 305 298 297 262
0 1 4 0 305 298 262 309
0 1 4 0 41 213 217 219
0 1 4 0 213 20 191 220
0 1 4 0 217 191 36 222
0 1 4 0 219 220 222 40
0 1 4 0 213 222 217 219
0 1 4 0 213 222 219 220
0 1 4 0 213 222 220 191
0 1 4 0 213 222 191 217
0 1 4 0 41 213 219 218
0 1 4 0 213 20 220 124
0 1 4 0 219 220 40 221
0 1 4 0 218 124 221 24
0 1 4 0 213 221 219 218
0 1 4 0 213 221 218 124
0 1 4 0 213 221 124 220
0 1 4 0 213 221 220 219
0 1 4 0 43 230 231 232
0 1 4 0 230 22 141 110
0 1 4 0 231 141 27 140
0 1 4 0 232 110 140 23
0 1 4 0 230 140 231 232
0 1 4 0 230 140 232 110
0 1 4 0 230 140 110 141
0 1 4 0 230 140 141 231
0 1 4 0 43 230 232 233
0 1 4 0 230 22 110 211
0 1 4 0 232 110 23 204
0 1 4 0 233 211 204 39
0 1 4 0 230 204 232 233
0 1 4 0 230 204 233 211
0 1 4 0 230 204 211 110
0 1 4 0 230 204 110 232
0 1 4 0 38 193 195 196
0 1 4 0 193 17 93 197
0 1 4 0 195 93 18 198
0 1 4 0 196 197 198 34
0 1 4 0 193 198 195 196
0 1 4 0 193 198 196 197
0 1 4 0 193 198 197 93
0 1 4 0 193 198 93 195
0 1 4 0 38 193 196 199
0 1 4 0 193 17 197 184
0 1 4 0 196 197 34 200
0 1 4 0 199 184 200 33
0 1 4 0 193 200 196 199
0 1 4 0 193 200 199 184
0 1 4 0 193 200 184 197
0 1 4 0 193 200 197 196
0 1 4 0 46 247 253 252
0 1 4 0 247 25 239 150
0 1 4 0 253 239 45 238
0 1 4 0 252 150 238 29
0 1 4 0 247 238 253 252
0 1 4 0 247 238 252 150
0 1 4 0 247 238 150 239
0 1 4 0 247 238 239 253
0 1 4 0 46 247 252 248
0 1 4 0 247 25 150 165
0 1 4 0 252 150 29 168
0 1 4 0 248 165 168 30
0 1 4 0 247 168 252 248
0 1 4 0 247 168 248 165
0 1 4 0 247 168 165 150
0 1 4 0 247 168 150 252
# attr geom nodes
boundary
144
1 2 0 5 1
1 2 0 1 17
1 2 0 17 16
1 2 0 4 5
1 2 0 20 4
1 2 0 16 20
1 2 1 6 2
1 2 1 2 18
1 2 1 18 17
1 2 1 5 6
1 2 2 7 3
1 2 23 3 7
1 2 2 3 19
1 2 23 19 3
1 2 2 19 18
1 2 2 6 7
1 2 4 9 5
1 2 4 8 9
1 2 4 24 8
1 2 4 20 24
1 2 6 11 7
1 2 27 7 11
1 2 27 23 7
1 2 6 10 11
1 2 8 13 9
1 2 8 12 13
1 2 29 13 12
1 2 8 28 12
1 2 29 12 28
1 2 8 24 28
1 2 9 14 10
1 2 9 13 14
1 2 30 14 13
1 2 30 13 29
1 2 10 15 11
1 2 31 11 15
1 2 31 27 11
1 2 10 14 15
1 2 31 15 14
1 2 31 14 30
1 2 16 17 33
1 2 16 33 32
1 2 16 36 20
1 2 16 32 36
1 2 39 19 23
1 2 18 19 35
1 2 39 35 19
1 2 18 35 34
1 2 45 29 28
1 2 24 44 28
1 2 45 28 44
1 2 24 40 44
1 2 47 27 31
1 2 47 43 27
1 2 47 31 30
1 2 47 30 46
1 2 32 33 49
1 2 32 49 48
1 2 53 48 49
1 2 32 52 36
1 2 32 48 52
1 2 53 52 48
1 2 33 34 50
1 2 33 50 49
1 2 54 49 50
1 2 54 53 49
1 2 55 35 39
1 2 34 35 51
1 2 55 51 35
1 2 34 51 50
1 2 55 50 51
1 2 55 54 50
1 2 57 52 53
1 2 36 56 40
1 2 36 52 56
1 2 57 56 52
1 2 59 39 43
1 2 59 55 39
1 2 59 54 55
1 2 59 58 54
1 2 61 56 57
1 2 61 45 44
1 2 40 60 44
1 2 61 44 60
1 2 40 56 60
1 2 61 60 56
1 2 62 57 58
1 2 62 46 45
1 2 62 45 61
1 2 62 61 57
1 2 63 43 47
1 2 63 59 43
1 2 63 58 59
1 2 63 47 46
1 2 63 46 62
1 2 63 62 58
2 2 5 115 128
2 2 115 9 133
2 2 128 133 10
2 2 133 128 115
2 2 5 128 95
2 2 128 10 129
2 2 95 129 6
2 2 129 95 128
2 2 58 309 308
2 2 308 284 54
2 2 309 53 284
2 2 284 308 309
2 2 58 311 309
2 2 309 298 53
2 2 311 57 298
2 2 298 309 311
2 2 20 191 220
2 2 191 36 222
2 2 220 222 40
2 2 222 220 191
2 2 20 220 124
2 2 220 40 221
2 2 124 221 24
2 2 221 124 220
2 2 43 232 231
2 2 231 140 27
2 2 232 23 140
2 2 140 231 232
2 2 43 233 232
2 2 232 204 23
2 2 233 39 204
2 2 204 232 233
2 2 17 93 197
2 2 93 18 198
2 2 197 198 34
2 2 198 197 93
2 2 17 197 184
2 2 197 34 200
2 2 184 200 33
2 2 200 184 197
2 2 46 252 253
2 2 253 238 45
2 2 252 29 238
2 2 238 253 252
2 2 46 248 252
2 2 252 168 29
2 2 248 30 168
2 2 168 252 248
# vert_id p1 p2
vertex_parents
54
93 17 18
95 5 6
110 22 23
115 5 9
124 20 24
125 5 26
126 10 26
127 6 26
128 5 10
129 6 10
132 9 26
133 9 10
140 23 27
141 22 27
150 25 29
165 25 30
168 29 30
184 17 33
191 20 36
193 17 38
195 18 38
196 34 38
197 17 34
198 18 34
199 33 38
200 33 34
204 23 39
211 22 39
213 20 41
217 36 41
218 24 41
219 40 41
220 20 40
221 24 40
222 36 40
230 22 43
231 27 43
232 23 43
233 39 43
238 29 45
239 25 45
247 25 46
248 30 46
252 29 46
253 45 46
262 37 53
283 37 54
284 53 54
297 37 57
298 53 57
305 37 58
308 54 58
309 53 58
311 57 58
# top-level node coordinates
coordinates
64
3
0 0 0
0.33333333 0 0
0.66666667 0 0
1 0 0
0 0.33333333 0
0.33333333 0.33333333 0
0.66666667 0.33333333 0
1 0.33333333 0
0 0.66666667 0
0.33333333 0.66666667 0
0.66666667 0.66666667 0
1 0.66666667 0
0 1 0
0.33333333 1 0
0.66666667 1 0
1 1 0
0 0 0.33333333
0.33333333 0 0.33333333
0.66666667 0 0.33333333
1 0 0.33333333
0 0.33333333 0.33333333
0.33333333 0.33333333 0.33333333
0.66666667 0.33333333 0.33333333
1 0.33333333 0.33333333
0 0.66666667 0.33333333
0.33333333 0.66666667 0.33333333
0.66666667 0.66666667 0.33333333
1 0.66666667 0.33333333
0 1 0.33333333
0.33333333 1 0.33333333
0.66666667 1 0.33333333
1 1 0.33333333
0 0 0.66666667
0.33333333 0 0.66666667
0.66666667 0 0.66666667
1 0 0.66666667
0 0.33333333 0.66666667
0.33333333 0.33333333 0.66666667
0.66666667 0.33333333 0.66666667
1 0.33333333 0.66666667
0 0.66666667 0.66666667
0.33333333 0.66666667 0.66666667
0.66666667 0.66666667 0.66666667
1 0.66666667 0.66666667
0 1 0.66666667
0.33333333 1 0.66666667
0.66666667 1 0.66666667
1 1 0.66666667
0 0 1
0.33333333 0 1
0.66666667 0 1
1 0 1
0 0.33333333 1
0.33333333 0.33333333 1
0.66666667 0.33333333 1
1 0.33333333 1
0 0.66666667 1
0.33333333 0.66666667 1
0.66666667 0.66666667 1
1 0.66666667 1
0 1 1
0.33333333 1 1
0.66666667 1 1
1 1 1
mfem_mesh_end
+274
View File
@@ -0,0 +1,274 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../unit_tests.hpp"
#include "mfem.hpp"
#ifndef MFEM_USE_MPI
#include "../../../fem/dfem/tuple.hpp"
#endif
using namespace mfem;
using namespace mfem::future;
namespace tuple_test
{
// A payload that is not a scalar, mimicking what dFEM kernels actually store.
using vec3 = tensor<real_t, 3>;
using tuple3 = tuple<real_t, int, vec3>;
// mfem::future::tuple is no longer an aggregate: it derives from tuple_leaf
// bases so that it can be defined for an arbitrary number of elements. These
// checks pin down the properties that the aggregate used to provide for free
// and that device kernels (which capture tuples by value) depend on.
static_assert(std::is_trivially_copyable<tuple3>::value,
"tuple must be trivially copyable to be captured by value in device kernels");
static_assert(std::is_trivially_destructible<tuple3>::value,
"tuple must be trivially destructible");
static_assert(std::is_trivially_default_constructible<tuple3>::value,
"tuple must be trivially default constructible");
static_assert(std::is_trivially_copy_assignable<tuple3>::value,
"tuple must be trivially copy assignable");
static_assert(sizeof(tuple3) == sizeof(real_t) + sizeof(int) + sizeof(vec3) +
(alignof(real_t) - sizeof(int)),
"tuple must not be larger than the sum of its (padded) members");
// Size and element types, both through mfem::future and through the std
// specializations that drive structured bindings.
static_assert(tuple_size<tuple3>::value == 3, "");
static_assert(std::tuple_size<tuple3>::value == 3, "");
static_assert(std::is_same<tuple_element<0, tuple3>::type, real_t>::value, "");
static_assert(std::is_same<tuple_element<1, tuple3>::type, int>::value, "");
static_assert(std::is_same<tuple_element<2, tuple3>::type, vec3>::value, "");
static_assert(std::is_same<std::tuple_element_t<0, tuple3>, real_t>::value, "");
static_assert(std::is_same<std::tuple_element_t<2, tuple3>, vec3>::value, "");
// get must preserve the value category and constness of its argument.
static_assert(std::is_same<decltype(get<1>(std::declval<tuple3&>())),
int&>::value, "get on an lvalue must return an lvalue reference");
static_assert(std::is_same<decltype(get<1>(std::declval<const tuple3&>())),
const int&>::value,
"get on a const lvalue must return a const lvalue reference");
static_assert(std::is_same<decltype(get<1>(std::declval<tuple3&&>())),
int&&>::value, "get on an rvalue must return an rvalue reference");
static_assert(std::is_same<decltype(get<1>(std::declval<const tuple3&&>())),
const int&&>::value,
"get on a const rvalue must return a const rvalue reference");
// += and -= must return a reference, not a copy of the whole tuple.
using tuple2 = tuple<real_t, vec3>;
static_assert(std::is_same<decltype(std::declval<tuple2&>() +=
std::declval<const tuple2&>()), tuple2&>::value,
"operator+= must return a reference");
static_assert(std::is_same<decltype(std::declval<tuple2&>() -=
std::declval<const tuple2&>()), tuple2&>::value,
"operator-= must return a reference");
// The element-wise constructor must stay implicit, so that the
// copy-list-initialization forms that worked with the aggregate keep working.
static_assert(std::is_convertible<int, tuple<int>>::value,
"tuple's element-wise constructor must not be explicit");
// Constructing from an incompatible type must SFINAE out rather than hard-error,
// so that the constructor does not poison type traits.
struct not_a_number { };
static_assert(!std::is_constructible<tuple<int, int>, int, not_a_number>::value,
"");
static_assert(!std::is_constructible<tuple<int, int>, int>::value,
"arity mismatch must not be constructible");
// Usable at compile time.
constexpr tuple<int, real_t> const_tuple {2, 3.0};
static_assert(get<0>(const_tuple) == 2, "");
// Copy-list-initialization in a return statement (broken by an explicit ctor).
tuple<int, real_t> returns_braced_init_list() { return {7, 8.0}; }
} // namespace tuple_test
using namespace tuple_test;
TEST_CASE("dFEM tuple structured bindings", "[dFEM]")
{
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
SECTION("binding by reference writes through")
{
auto &[a, b, c] = t;
a = 10.0;
b = 20;
c(0) = 30.0;
REQUIRE(get<0>(t) == 10.0_r);
REQUIRE(get<1>(t) == 20);
REQUIRE(get<2>(t)(0) == 30.0_r);
}
SECTION("binding by value copies")
{
auto [a, b, c] = t;
a = 10.0;
b = 20;
c(0) = 30.0;
REQUIRE(get<0>(t) == 1.0_r);
REQUIRE(get<1>(t) == 2);
REQUIRE(get<2>(t)(0) == 3.0_r);
}
SECTION("binding to const")
{
const auto &[a, b, c] = t;
REQUIRE(a == 1.0_r);
REQUIRE(b == 2);
REQUIRE(c(2) == 5.0_r);
static_assert(std::is_same<decltype(a), const real_t>::value, "");
static_assert(std::is_same<decltype(c), const vec3>::value, "");
}
SECTION("the bindings alias the tuple storage")
{
auto &[a, b, c] = t;
REQUIRE(&a == &get<0>(t));
REQUIRE(&b == &get<1>(t));
REQUIRE(&c == &get<2>(t));
}
}
TEST_CASE("dFEM tuple construction", "[dFEM]")
{
SECTION("copy-list-initialization")
{
tuple<int, real_t> a = {1, 2.0};
REQUIRE(get<0>(a) == 1);
REQUIRE(get<1>(a) == 2.0_r);
const auto b = returns_braced_init_list();
REQUIRE(get<0>(b) == 7);
REQUIRE(get<1>(b) == 8.0_r);
}
SECTION("direct initialization and CTAD")
{
tuple c {1, 2.0_r, vec3{{1.0, 2.0, 3.0}}};
static_assert(std::is_same<decltype(c), tuple<int, real_t, vec3>>::value,
"CTAD must decay the arguments");
REQUIRE(get<1>(c) == 2.0_r);
}
SECTION("make_tuple")
{
const auto d = make_tuple(1, 2.0_r);
static_assert(std::is_same<decltype(d), const tuple<int, real_t>>::value, "");
REQUIRE(get<0>(d) == 1);
}
SECTION("copy and move construction preserve values")
{
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
tuple3 copy(t);
tuple3 moved(std::move(t));
REQUIRE(get<1>(copy) == 2);
REQUIRE(get<2>(moved)(1) == 4.0_r);
}
SECTION("value initialization zeroes trivial members")
{
tuple<int, real_t> z {};
REQUIRE(get<0>(z) == 0);
REQUIRE(get<1>(z) == 0.0_r);
}
}
TEST_CASE("dFEM tuple arithmetic", "[dFEM]")
{
const tuple2 x {1.0, vec3{{1.0, 2.0, 3.0}}};
const tuple2 y {2.0, vec3{{4.0, 5.0, 6.0}}};
SECTION("element-wise binary operators")
{
const auto sum = x + y;
REQUIRE(get<0>(sum) == 3.0_r);
REQUIRE(get<1>(sum)(2) == 9.0_r);
const auto diff = y - x;
REQUIRE(get<0>(diff) == 1.0_r);
REQUIRE(get<1>(diff)(0) == 3.0_r);
}
SECTION("compound assignment mutates in place and returns a reference")
{
tuple2 z = x;
auto &ref = (z += y);
REQUIRE(&ref == &z);
REQUIRE(get<0>(z) == 3.0_r);
REQUIRE(get<1>(z)(1) == 7.0_r);
auto &ref2 = (z -= y);
REQUIRE(&ref2 == &z);
REQUIRE(get<0>(z) == 1.0_r);
REQUIRE(get<1>(z)(1) == 2.0_r);
}
SECTION("scalar operators and unary minus")
{
const auto scaled = 2.0_r * x;
REQUIRE(get<0>(scaled) == 2.0_r);
REQUIRE(get<1>(scaled)(2) == 6.0_r);
const auto halved = x / 2.0_r;
REQUIRE(get<0>(halved) == 0.5_r);
const auto negated = -x;
REQUIRE(get<0>(negated) == -1.0_r);
REQUIRE(get<1>(negated)(0) == -1.0_r);
}
SECTION("apply")
{
const auto s = apply([](const real_t &a, const vec3 &b) { return a + b(0); },
x);
REQUIRE(s == 2.0_r);
}
}
// The tuples are captured by value in device kernels, so exercise a round trip
// through device memory: construct, mutate through structured bindings and read
// back on the device.
TEST_CASE("dFEM tuple on device", "[dFEM][GPU]")
{
Vector res(4);
auto d_res = res.Write();
forall(1, [=] MFEM_HOST_DEVICE (int)
{
tuple3 t {1.0, 2, vec3{{3.0, 4.0, 5.0}}};
auto &[a, b, c] = t;
a += static_cast<real_t>(b);
c(0) = a;
tuple2 u {get<0>(t), get<2>(t)};
u += tuple2 {1.0, vec3{{1.0, 1.0, 1.0}}};
d_res[0] = get<0>(u);
d_res[1] = get<1>(u)(0);
d_res[2] = get<1>(u)(1);
d_res[3] = static_cast<real_t>(get<1>(t));
tuple2 v1{0_r, vec3{0_r, 0_r, 0_r}};
tuple2 v2{0_r, vec3{0_r, 0_r, 0_r}};
[[maybe_unused]] auto v = v1 + v2;
});
res.HostRead();
REQUIRE(std::as_const(res)(0) == 4.0_r);
REQUIRE(std::as_const(res)(1) == 4.0_r);
REQUIRE(std::as_const(res)(2) == 5.0_r);
REQUIRE(std::as_const(res)(3) == 2.0_r);
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_convection_kernels.hpp"
TEST_CASE("Convection Kernel Specializations", "[Specializations]")
{
using namespace mfem;
ConvectionIntegrator::AddSpecialization<2, 2, 4>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_hcurl_kernels.hpp"
TEST_CASE("CurlCurl Kernel Specializations", "[Specializations]")
{
using namespace mfem;
CurlCurlIntegrator::AddSpecialization<3, 2, 4>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_dgdiffusion_kernels.hpp"
TEST_CASE("DGDiffusion Kernel Specializations", "[Specializations]")
{
using namespace mfem;
DGDiffusionIntegrator::AddSpecialization<2, 2, 4>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/dgmassinv_kernels.hpp"
TEST_CASE("DGMassInverse Kernel Specializations", "[Specializations]")
{
using namespace mfem;
DGMassInverse::CGKernels::Specialization<2, 1, 2>::Add();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_dgtrace_kernels.hpp"
TEST_CASE("DGTrace Kernel Specializations", "[Specializations]")
{
using namespace mfem;
DGTraceIntegrator::AddSpecialization<2, 2, 3>();
}
@@ -0,0 +1,15 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_diffusion_kernels.hpp"
TEST_CASE("Diffusion Kernel Specializations", "[Specializations]")
{
using namespace mfem;
DiffusionIntegrator::AddSpecialization<2, 1, 5>();
DiffusionIntegrator::AddSimplexSpecialization<2, 2, 3>();
}
@@ -0,0 +1,15 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_mass_kernels.hpp"
TEST_CASE("Mass Kernel Specializations", "[Specializations]")
{
using namespace mfem;
MassIntegrator::AddSpecialization<2, 1, 3>();
MassIntegrator::AddSimplexSpecialization<2, 2, 4>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/qinterp/det.hpp"
TEST_CASE("QInterp Det Kernel Specializations", "[Specializations]")
{
using namespace mfem;
QuadratureInterpolator::AddDetSpecializations<2, 3, 2, 2>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/qinterp/eval.hpp"
TEST_CASE("QInterp Eval Kernel Specializations", "[Specializations]")
{
using namespace mfem;
QuadratureInterpolator::AddEvalSpecializations<2, 1, 1, 2>();
}
@@ -0,0 +1,19 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/qinterp/eval_hdiv.hpp"
TEST_CASE("QInterp Eval HDiv Kernel Specializations", "[Specializations]")
{
using namespace mfem;
QuadratureInterpolator::TensorEvalHDivKernels::Specialization<
2, QVectorLayout::byNODES, QuadratureInterpolator::PHYSICAL_VALUES, 2,
4>::Add();
QuadratureInterpolator::TensorEvalHDivKernels::Specialization<
2, QVectorLayout::byNODES, QuadratureInterpolator::PHYSICAL_MAGNITUDES, 2,
4>::Add();
}
@@ -0,0 +1,21 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/qinterp/grad.hpp"
TEST_CASE("QInterp Grad Kernel Specializations", "[Specializations]")
{
using namespace mfem;
QuadratureInterpolator::AddGradSpecializations<
2, QVectorLayout::byNODES, false, 1, 3, 3, 1>();
QuadratureInterpolator::AddGradSpecializations<
2, QVectorLayout::byNODES, true, 1, 3, 3, 1>();
QuadratureInterpolator::AddCollocatedGradSpecializations<
2, QVectorLayout::byNODES, false, 1, 2, 1>();
QuadratureInterpolator::AddCollocatedGradSpecializations<
2, QVectorLayout::byNODES, true, 1, 2, 1>();
}
@@ -0,0 +1,15 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/qinterp/eval.hpp"
TEST_CASE("QInterp TensorEval Kernel Specializations", "[Specializations]")
{
using namespace mfem;
QuadratureInterpolator::AddTensorEvalSpecializations<
2, QVectorLayout::byNODES, 1, 3, 3, 2>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp"
TEST_CASE("VectorDiffusion Kernel Specializations", "[Specializations]")
{
using namespace mfem;
VectorDiffusionIntegrator::AddSpecialization<2, 2, 2, 3>();
}
@@ -0,0 +1,14 @@
/// Tests which make sure adding user-defined kernel specializations work.
/// These tests are compile/link-only tests
#include "mfem.hpp"
#include "unit_tests.hpp"
#include "fem/integ/bilininteg_vecmass_pa.hpp"
TEST_CASE("Vector Mass Kernel Specializations", "[Specializations]")
{
using namespace mfem;
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 2, 4>::Add();
}
+77
View File
@@ -3451,4 +3451,81 @@ TEST_CASE("2D Bilinear Scalar Weak Curl Cross Integrators",
}
}
TEST_CASE("2D Bilinear Scalar Curl Integrator PartialAssembly",
"[MixedScalarCurlIntegrator]"
"[BilinearFormIntegrator]"
"[NonlinearFormIntegrator]"
"[GPU]")
{
int order = 2, n = 1, dim = 2;
double tol = 1e-9;
Mesh mesh = Mesh::MakeCartesian2D(n, n, Element::QUADRILATERAL, 1, 2.0, 3.0);
VectorFunctionCoefficient F2_coef(dim, F2);
FunctionCoefficient q2_coef(q2);
SECTION("Operators on ND")
{
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
GridFunction f_nd(&fespace_nd); f_nd.ProjectCoefficient(F2_coef);
for (int map_type = (int)FiniteElement::VALUE;
map_type <= (int)FiniteElement::INTEGRAL; map_type++)
{
SECTION("Mapping ND to L2 (" +
MapTypeName((FiniteElement::MapType)map_type) + ")")
{
L2_FECollection fec_l2(order - 1, dim,
BasisType::GaussLegendre,
(FiniteElement::MapType)map_type);
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
Vector tmp_l2(fespace_l2.GetNDofs());
Vector tmp_l2_pa(fespace_l2.GetNDofs());
SECTION("Without Coefficient")
{
MixedBilinearForm blf_fa(&fespace_nd, &fespace_l2);
blf_fa.AddDomainIntegrator(new MixedScalarCurlIntegrator());
blf_fa.Assemble();
blf_fa.Finalize();
blf_fa.Mult(f_nd, tmp_l2);
MixedBilinearForm blf_pa(&fespace_nd, &fespace_l2);
blf_pa.SetAssemblyLevel(mfem::AssemblyLevel::PARTIAL);
blf_pa.AddDomainIntegrator(new MixedScalarCurlIntegrator());
blf_pa.Assemble();
blf_pa.Mult(f_nd, tmp_l2_pa);
tmp_l2_pa -= tmp_l2;
REQUIRE(tmp_l2_pa.Normlinf() < tol);
}
SECTION("With Scalar Coefficient")
{
MixedBilinearForm blf_fa(&fespace_nd, &fespace_l2);
blf_fa.AddDomainIntegrator(
new MixedScalarCurlIntegrator(q2_coef));
blf_fa.Assemble();
blf_fa.Finalize();
blf_fa.Mult(f_nd, tmp_l2);
MixedBilinearForm blf_pa(&fespace_nd, &fespace_l2);
blf_pa.SetAssemblyLevel(mfem::AssemblyLevel::PARTIAL);
blf_pa.AddDomainIntegrator(new MixedScalarCurlIntegrator(q2_coef));
blf_pa.Assemble();
blf_pa.Mult(f_nd, tmp_l2_pa);
tmp_l2_pa -= tmp_l2;
REQUIRE(tmp_l2_pa.Normlinf() < tol);
}
}
}
}
}
} // namespace bilininteg_2d
+234
View File
@@ -1069,4 +1069,238 @@ TEST_CASE("Exact Sequence Properties: d(df)=0",
}
}
template <class A, class B>
static void TestCurl(FiniteElementSpace &dom_fes, FiniteElementSpace &ran_fes,
A coeff, B dcoeff)
{
real_t tol = 1e-10;
DiscreteLinearOperator CurlFA(&dom_fes, &ran_fes);
CurlFA.AddDomainInterpolator(new CurlInterpolator());
CurlFA.Assemble();
CurlFA.Finalize();
SparseMatrix &Curl = CurlFA.SpMat();
GridFunction x(&dom_fes), y_fa(&ran_fes), y(&ran_fes);
x.ProjectCoefficient(coeff);
y.ProjectCoefficient(dcoeff);
REQUIRE(x.Size() == Curl.Width());
REQUIRE(y_fa.Size() == Curl.Height());
Curl.Mult(x, y_fa);
y_fa -= y;
REQUIRE(y_fa.Normlinf() < tol);
}
template<class Coeff, class TCoeff>
static void CompareCurlPA(FiniteElementSpace& dom_fes,
FiniteElementSpace &ran_fes,
Coeff coeff, TCoeff tcoeff)
{
real_t tol = 1e-10;
DiscreteLinearOperator CurlFA(&dom_fes, &ran_fes);
CurlFA.AddDomainInterpolator(new CurlInterpolator());
CurlFA.Assemble();
CurlFA.Finalize();
DiscreteLinearOperator CurlPA(&dom_fes, &ran_fes);
CurlPA.AddDomainInterpolator(new CurlInterpolator());
CurlPA.SetAssemblyLevel(AssemblyLevel::PARTIAL);
CurlPA.Assemble();
SparseMatrix &Curl = CurlFA.SpMat();
GridFunction x(&dom_fes), y_fa(&ran_fes), y_pa(&ran_fes);
x.ProjectCoefficient(coeff);
REQUIRE(x.Size() == Curl.Width());
REQUIRE(y_fa.Size() == Curl.Height());
REQUIRE(x.Size() == CurlPA.Width());
REQUIRE(y_pa.Size() == CurlPA.Height());
Curl.Mult(x, y_fa);
CurlPA.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE(y_pa.Normlinf() < tol);
// transpose
y_fa.ProjectCoefficient(tcoeff);
GridFunction x_fa(&dom_fes), x_pa(&dom_fes);
Curl.MultTranspose(y_fa, x_fa);
CurlPA.MultTranspose(y_fa, x_pa);
x_pa -= x_fa;
REQUIRE(x_pa.Normlinf() < tol);
}
TEST_CASE("Partial Assemble Linear Interpolator",
"[CurlInterpolator]"
"[GPU]")
{
constexpr int maxOrder = 3;
auto order = GENERATE_COPY(range(1, maxOrder + 1));
CAPTURE(order);
auto dim = GENERATE(2, 3);
CAPTURE(dim);
int n = 3;
Mesh mesh;
switch (dim)
{
case 2:
mesh =
Mesh::MakeCartesian2D(n, n, Element::QUADRILATERAL, true, 2.0, 3.0);
break;
case 3:
mesh = Mesh::MakeCartesian3D(n, n, n, Element::HEXAHEDRON, 2.0, 3.0, 5.0);
break;
}
// domain spaces
H1_FECollection fec_h1(order, dim);
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
// range spaces
RT_FECollection fec_rt(order - 1, dim);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
L2_FECollection fec_l2(order - 1, dim, BasisType::GaussLegendre,
FiniteElement::INTEGRAL);
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
switch (dim)
{
case 2:
{
FunctionCoefficient coeff([](const Vector &x)
{ return sin(2 * M_PI * x[1] / 3) - cos(2 * M_PI * x[0] / 2); });
VectorFunctionCoefficient vcoeff(2, [](const Vector &x, Vector &y)
{
y.SetSize(2);
y[0] = -cos(2 * M_PI * x[1] / 3);
y[1] = sin(2 * M_PI * x[0] / 2);
});
// out of plane H1 -> in-plane RT
SECTION("H1 to RT")
{
CompareCurlPA(fespace_h1, fespace_rt, coeff, vcoeff);
}
// in-plane ND -> out of plane L2
SECTION("ND to L2")
{
CompareCurlPA(fespace_nd, fespace_l2, vcoeff, coeff);
}
break;
}
case 3:
{
VectorFunctionCoefficient coeff(3, [](const Vector &x, Vector &y)
{
y.SetSize(3);
y[0] = sin(2 * M_PI * x[2] / 5) - cos(2 * M_PI * x[1] / 3);
y[1] = sin(2 * M_PI * x[0] / 2) - cos(2 * M_PI * x[2] / 5);
y[2] = sin(2 * M_PI * x[1] / 3) - cos(2 * M_PI * x[0] / 2);
});
CompareCurlPA(fespace_nd, fespace_rt, coeff, coeff);
break;
}
}
}
TEST_CASE("Curl Linear Interpolator",
"[CurlInterpolator]"
"[GPU]")
{
int order = 2;
auto type = (Element::Type)GENERATE(range((int)Element::TRIANGLE,
(int)Element::PYRAMID + 1));
CAPTURE(type);
int n = 3;
Mesh mesh;
int dim;
if (type < (int)Element::TETRAHEDRON)
{
dim = 2;
mesh = Mesh::MakeCartesian2D(n, n, (Element::Type)type, 1, 2.0, 3.0);
}
else
{
dim = 3;
mesh = Mesh::MakeCartesian3D(n, n, n, (Element::Type)type,
2.0, 3.0, 5.0);
}
// domain spaces
H1_FECollection fec_h1(order, dim);
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
// range spaces
RT_FECollection fec_rt(order - 1, dim);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
L2_FECollection fec_l2(order - 1, dim, BasisType::GaussLegendre,
FiniteElement::INTEGRAL);
FiniteElementSpace fespace_l2(&mesh, &fec_l2);
switch (dim)
{
case 2:
{
// out of plane H1 -> in-plane RT
SECTION("H1 to RT")
{
FunctionCoefficient coeff([](const Vector &x)
{
return 1 - 2 * x[0] + 3 * x[1];
});
VectorFunctionCoefficient dcoeff(2, [](const Vector &x, Vector &y)
{
y.SetSize(2);
// d Ez/dy
y[0] = 3;
// -d Ez/dx
y[1] = 2;
});
TestCurl(fespace_h1, fespace_rt, coeff, dcoeff);
}
// in-plane ND -> out of plane L2
SECTION("ND to L2")
{
VectorFunctionCoefficient coeff(2, [](const Vector &x, Vector &y)
{
y.SetSize(2);
y[0] = 1 - 2 * x[0] + 3 * x[1];
y[1] = 2 * (1 - 2 * x[0] + 3 * x[1]);
});
FunctionCoefficient dcoeff([](const Vector &x)
{ return 2 * (-2) - 3; });
TestCurl(fespace_nd, fespace_l2, coeff, dcoeff);
}
break;
}
case 3:
{
VectorFunctionCoefficient coeff(3, [](const Vector &x, Vector &y)
{
y.SetSize(3);
y[0] = 1 + 2 * x[0] - 3 * x[1] + 4 * x[2];
y[1] = 4 + 3 * x[0] - 2 * x[1] + 1 * x[2];
y[2] = 2 - 1 * x[0] + 4 * x[1] - 3 * x[2];
});
VectorFunctionCoefficient dcoeff(3, [](const Vector &x, Vector &y)
{
y.SetSize(3);
y[0] = 4 - 1;
y[1] = 4 + 1;
y[2] = 3 + 3;
});
TestCurl(fespace_nd, fespace_rt, coeff, dcoeff);
break;
}
}
}
} // namespace lin_interp
+8 -1
View File
@@ -214,7 +214,14 @@ TEST_CASE("LOR AMS", "[LOR][BatchedLOR][AMS][Parallel][GPU]")
ParFiniteElementSpace vert_fespace(edge_fespace.GetParMesh(), &vert_fec);
ParDiscreteLinearOperator grad(&vert_fespace, &edge_fespace);
grad.AddDomainInterpolator(new GradientInterpolator);
if (space_type == RT)
{
grad.AddDomainInterpolator(new CurlInterpolator);
}
else
{
grad.AddDomainInterpolator(new GradientInterpolator);
}
grad.Assemble();
grad.Finalize();
std::unique_ptr<HypreParMatrix> G(grad.ParallelAssemble());
+323
View File
@@ -0,0 +1,323 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "mfem.hpp"
#include "unit_tests.hpp"
using namespace mfem;
namespace
{
constexpr real_t a_coef = 1.0;
constexpr real_t b_coef = 2.0;
constexpr real_t c_coef = 3.0;
constexpr real_t omega_val = 10.0;
real_t V_exact_fn(const Vector &x)
{
return a_coef*x[0] + b_coef*x[1] + c_coef*x[2];
}
} // namespace
TEST_CASE("Mixed Sesquilinear Form", "[MixedSesquilinearForm]")
{
const bool cross = GENERATE(false, true);
const auto conv = GENERATE(ComplexOperator::HERMITIAN,
ComplexOperator::BLOCK_SYMMETRIC);
CAPTURE(cross, int(conv));
Mesh mesh = Mesh::MakeCartesian3D(10, 10, 1, Element::HEXAHEDRON);
H1_FECollection fec_h1(1, mesh.Dimension());
FiniteElementSpace fespace_h1(&mesh, &fec_h1);
ND_FECollection fec_nd(1, mesh.Dimension());
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
Array<int> dbc_bdr(mesh.bdr_attributes.Max());
dbc_bdr = 1;
Array<int> ess_tdof_list_h1;
Array<int> ess_tdof_list_nd;
fespace_h1.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_h1);
fespace_nd.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_nd);
ConstantCoefficient omega(omega_val);
ConstantCoefficient neg_omega(-omega_val);
ConstantCoefficient half(0.5);
FunctionCoefficient V_exact_real(V_exact_fn);
const real_t den = cross ? 2.0*omega_val : omega_val;
Vector A_vec({a_coef/den, b_coef/den, c_coef/den});
VectorConstantCoefficient A_exact_imag(A_vec);
A_vec *= (cross ? -1.0 : 0.0);
VectorConstantCoefficient A_exact_real(A_vec);
ComplexGridFunction V(&fespace_h1);
ComplexGridFunction A(&fespace_nd);
V = 0.0;
A = 0.0;
V.real().ProjectBdrCoefficient(V_exact_real, dbc_bdr);
A.real().ProjectBdrCoefficientTangent(A_exact_real, dbc_bdr);
A.imag().ProjectBdrCoefficientTangent(A_exact_imag, dbc_bdr);
ComplexLinearForm b_h1(&fespace_h1);
b_h1 = 0.0;
b_h1.Assemble();
ComplexLinearForm b_nd(&fespace_nd);
b_nd = 0.0;
b_nd.Assemble();
// Add integrators to the blocks
MixedSesquilinearForm a_h1_nd(&fespace_h1, &fespace_nd, conv);
a_h1_nd.AddDomainIntegrator(cross ? new MixedVectorGradientIntegrator(half)
: new MixedVectorGradientIntegrator,
cross ? new MixedVectorGradientIntegrator(half)
: nullptr);
a_h1_nd.Assemble();
MixedSesquilinearForm a_nd_h1(&fespace_nd, &fespace_h1, conv);
a_nd_h1.AddDomainIntegrator(
cross ? new MixedVectorWeakDivergenceIntegrator(neg_omega) : nullptr,
new MixedVectorWeakDivergenceIntegrator(neg_omega));
a_nd_h1.Assemble();
SesquilinearForm a_h1(&fespace_h1, conv);
a_h1.AddDomainIntegrator(new DiffusionIntegrator, nullptr);
a_h1.Assemble();
SesquilinearForm a_nd(&fespace_nd, conv);
a_nd.AddDomainIntegrator(new CurlCurlIntegrator, nullptr);
a_nd.AddDomainIntegrator(nullptr, new VectorFEMassIntegrator(omega));
a_nd.Assemble();
// Set block offsets (doubled for real+imag)
mfem::Array<int> bOffsets(3);
bOffsets[0] = 0;
bOffsets[1] = 2 * fespace_h1.GetTrueVSize();
bOffsets[2] = 2 * fespace_nd.GetTrueVSize();
bOffsets.PartialSum();
OperatorPtr A_h1, A_nd, A_h1_nd, A_nd_h1;
BlockVector trueX(bOffsets), trueRHS(bOffsets);
Vector B_h1, X_h1, B_nd, X_nd;
trueX = 0.0;
trueRHS = 0.0;
// Form the diagonal entries
a_h1.FormLinearSystem(ess_tdof_list_h1, V, b_h1, A_h1, X_h1, B_h1);
a_nd.FormLinearSystem(ess_tdof_list_nd, A, b_nd, A_nd, X_nd, B_nd);
trueX.GetBlock(0) = X_h1;
trueX.GetBlock(1) = X_nd;
trueRHS.GetBlock(0) += B_h1;
trueRHS.GetBlock(1) += B_nd;
// Form the off-diagonal entries
a_h1_nd.FormRectangularLinearSystem(ess_tdof_list_h1, ess_tdof_list_nd, V, b_nd,
A_h1_nd, X_h1, B_nd);
a_nd_h1.FormRectangularLinearSystem(ess_tdof_list_nd, ess_tdof_list_h1, A, b_h1,
A_nd_h1, X_nd, B_h1);
trueRHS.GetBlock(0) += B_h1;
trueRHS.GetBlock(1) += B_nd;
auto *Ah1 = A_h1.As<ComplexSparseMatrix>();
auto *And = A_nd.As<ComplexSparseMatrix>();
auto *Ah1nd = A_h1_nd.As<ComplexSparseMatrix>();
auto *Andh1 = A_nd_h1.As<ComplexSparseMatrix>();
BlockOperator blockOp(bOffsets);
blockOp.SetBlock(0, 0, Ah1);
blockOp.SetBlock(1, 1, And);
blockOp.SetBlock(0, 1, Andh1);
blockOp.SetBlock(1, 0, Ah1nd);
SparseMatrix *Sh1 = Ah1->GetSystemMatrix();
SparseMatrix *Snd = And->GetSystemMatrix();
GSSmoother smoothSh1(*Sh1), smoothSnd(*Snd);
BlockDiagonalPreconditioner P(bOffsets);
P.SetDiagonalBlock(0, &smoothSh1);
P.SetDiagonalBlock(1, &smoothSnd);
GMRESSolver gmres;
gmres.SetOperator(blockOp);
gmres.SetPreconditioner(P);
gmres.SetAbsTol(1e-10);
gmres.SetMaxIter(2000);
gmres.SetKDim(200);
gmres.SetPrintLevel(2);
gmres.Mult(trueRHS, trueX);
delete Sh1;
delete Snd;
V = trueX.GetBlock(0);
A = trueX.GetBlock(1);
// Check solution
ConstantCoefficient zero(0.0);
real_t err_V = V.ComputeL2Error(V_exact_real, zero);
real_t err_A = A.ComputeL2Error(A_exact_real, A_exact_imag);
REQUIRE(err_V == MFEM_Approx(0.0, 1e-5));
REQUIRE(err_A == MFEM_Approx(0.0, 1e-5));
}
#ifdef MFEM_USE_MPI
#ifdef MFEM_USE_SUPERLU
TEST_CASE("Parallel Mixed Sesquilinear Form",
"[MixedSesquilinearForm][Parallel]")
{
// See the serial test above for the manufactured solution
const bool cross = GENERATE(false, true);
const auto conv = GENERATE(ComplexOperator::HERMITIAN,
ComplexOperator::BLOCK_SYMMETRIC);
CAPTURE(cross, int(conv));
Mesh mesh = Mesh::MakeCartesian3D(10, 10, 1, Element::HEXAHEDRON);
ParMesh par_mesh(MPI_COMM_WORLD, mesh);
H1_FECollection fec_h1(1, mesh.Dimension());
ParFiniteElementSpace fespace_h1(&par_mesh, &fec_h1);
ND_FECollection fec_nd(1, mesh.Dimension());
ParFiniteElementSpace fespace_nd(&par_mesh, &fec_nd);
Array<int> dbc_bdr(par_mesh.bdr_attributes.Max());
dbc_bdr = 1;
Array<int> ess_tdof_list_h1;
Array<int> ess_tdof_list_nd;
fespace_h1.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_h1);
fespace_nd.GetEssentialTrueDofs(dbc_bdr, ess_tdof_list_nd);
ConstantCoefficient omega(omega_val);
ConstantCoefficient neg_omega(-omega_val);
ConstantCoefficient half(0.5);
FunctionCoefficient V_exact_real(V_exact_fn);
const real_t den = cross ? 2.0*omega_val : omega_val;
Vector A_vec({a_coef/den, b_coef/den, c_coef/den});
VectorConstantCoefficient A_exact_imag(A_vec);
A_vec *= (cross ? -1.0 : 0.0);
VectorConstantCoefficient A_exact_real(A_vec);
ParComplexGridFunction V(&fespace_h1);
ParComplexGridFunction A(&fespace_nd);
V = 0.0;
A = 0.0;
V.real().ProjectBdrCoefficient(V_exact_real, dbc_bdr);
A.real().ProjectBdrCoefficientTangent(A_exact_real, dbc_bdr);
A.imag().ProjectBdrCoefficientTangent(A_exact_imag, dbc_bdr);
ParComplexLinearForm b_h1(&fespace_h1);
b_h1 = 0.0;
b_h1.Assemble();
ParComplexLinearForm b_nd(&fespace_nd);
b_nd = 0.0;
b_nd.Assemble();
// Add integrators to the blocks
ParMixedSesquilinearForm a_h1_nd(&fespace_h1, &fespace_nd, conv);
a_h1_nd.AddDomainIntegrator(cross ? new MixedVectorGradientIntegrator(half)
: new MixedVectorGradientIntegrator,
cross ? new MixedVectorGradientIntegrator(half)
: nullptr);
a_h1_nd.Assemble();
ParMixedSesquilinearForm a_nd_h1(&fespace_nd, &fespace_h1, conv);
a_nd_h1.AddDomainIntegrator(
cross ? new MixedVectorWeakDivergenceIntegrator(neg_omega) : nullptr,
new MixedVectorWeakDivergenceIntegrator(neg_omega));
a_nd_h1.Assemble();
ParSesquilinearForm a_h1(&fespace_h1, conv);
a_h1.AddDomainIntegrator(new DiffusionIntegrator, nullptr);
a_h1.Assemble();
ParSesquilinearForm a_nd(&fespace_nd, conv);
a_nd.AddDomainIntegrator(new CurlCurlIntegrator, nullptr);
a_nd.AddDomainIntegrator(nullptr, new VectorFEMassIntegrator(omega));
a_nd.Assemble();
mfem::Array2D<const mfem::HypreParMatrix *> h_blocks;
h_blocks.SetSize(2, 2);
h_blocks = nullptr;
// Set block offsets
mfem::Array<int> bOffsets(3);
bOffsets[0] = 0;
bOffsets[1] = 2 * fespace_h1.TrueVSize();
bOffsets[2] = 2 * fespace_nd.TrueVSize();
bOffsets.PartialSum();
OperatorPtr A_h1, A_nd, A_h1_nd, A_nd_h1;
BlockVector trueX(bOffsets), trueRHS(bOffsets);
Vector B_h1, X_h1, B_nd, X_nd;
trueX = 0.0;
trueRHS = 0.0;
// Form the diagonal entries
a_h1.FormLinearSystem(ess_tdof_list_h1, V, b_h1, A_h1, X_h1, B_h1);
a_nd.FormLinearSystem(ess_tdof_list_nd, A, b_nd, A_nd, X_nd, B_nd);
trueX.GetBlock(0) = X_h1;
trueX.GetBlock(1) = X_nd;
trueRHS.GetBlock(0) += B_h1;
trueRHS.GetBlock(1) += B_nd;
// Form the off-diagonal entries
a_h1_nd.FormRectangularLinearSystem(ess_tdof_list_h1, ess_tdof_list_nd, V, b_nd,
A_h1_nd, X_h1, B_nd);
a_nd_h1.FormRectangularLinearSystem(ess_tdof_list_nd, ess_tdof_list_h1, A, b_h1,
A_nd_h1, X_nd, B_h1);
trueRHS.GetBlock(0) += B_h1;
trueRHS.GetBlock(1) += B_nd;
h_blocks(0,0) = A_h1.As<ComplexHypreParMatrix>()->GetSystemMatrix();
h_blocks(1,1) = A_nd.As<ComplexHypreParMatrix>()->GetSystemMatrix();
h_blocks(0,1) = A_nd_h1.As<ComplexHypreParMatrix>()->GetSystemMatrix();
h_blocks(1,0) = A_h1_nd.As<ComplexHypreParMatrix>()->GetSystemMatrix();
OperatorHandle op(HypreParMatrixFromBlocks(h_blocks));
SuperLURowLocMatrix S_op(*op);
SuperLUSolver superlu(MPI_COMM_WORLD);
superlu.SetPrintStatistics(false);
superlu.SetSymmetricPattern(false);
superlu.SetOperator(S_op);
superlu.Mult(trueRHS, trueX);
trueX.GetBlock(0).SyncAliasMemory(trueX);
trueX.GetBlock(1).SyncAliasMemory(trueX);
V.Distribute(trueX.GetBlock(0));
A.Distribute(trueX.GetBlock(1));
// Check solution
ConstantCoefficient zero(0.0);
real_t err_Vr = V.real().ComputeL2Error(V_exact_real);
real_t err_Vi = V.imag().ComputeL2Error(zero);
real_t err_Ar = A.real().ComputeL2Error(A_exact_real);
real_t err_Ai = A.imag().ComputeL2Error(A_exact_imag);
REQUIRE(err_Vr == MFEM_Approx(0.0, 1e-5));
REQUIRE(err_Vi == MFEM_Approx(0.0, 1e-5));
REQUIRE(err_Ar == MFEM_Approx(0.0, 1e-5));
REQUIRE(err_Ai == MFEM_Approx(0.0, 1e-5));
}
#endif
#endif
+416
View File
@@ -750,4 +750,420 @@ TEST_CASE("Hcurl/Hdiv Mixed PA Coefficient",
}
}
TEST_CASE("3D Bilinear VectorFE Integrators PartialAssembly",
"[BilinearFormIntegrator]"
"[PartialAssembly]"
"[GPU]")
{
auto order = GENERATE(1, 2);
CAPTURE(order);
dimension = 3;
FunctionCoefficient q3_coeff(coeffFunction);
VectorFunctionCoefficient F3_coeff(dimension, vectorCoeffFunction);
MatrixFunctionCoefficient M3_coeff(dimension, asymmetricMatrixCoeffFunction);
auto mesh_fname =
GENERATE("../../data/fichera-amr.mesh", "../../data/fichera-q2.mesh");
CAPTURE(mesh_fname);
Mesh mesh(mesh_fname);
REQUIRE(mesh.Dimension() == dimension);
REQUIRE(mesh.SpaceDimension() == dimension);
SECTION("RT to RT Scalar Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
BilinearForm bfa(&fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to RT Diagonal Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
BilinearForm bfa(&fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to RT Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
BilinearForm bfa(&fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to ND Scalar Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to ND Diagonal Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to ND Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to RT Scalar Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to RT Diagonal Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to RT Matrix Coeff")
{
RT_FECollection fec_rt(order - 1, dimension);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
MixedBilinearForm bfa(&fespace_nd, &fespace_rt);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_nd, &fespace_rt);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_rt), y_pa(&fespace_rt);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to ND Scalar Coeff")
{
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
BilinearForm bfa(&fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(q3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to ND Diagonal Matrix Coeff")
{
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
BilinearForm bfa(&fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(F3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("ND to ND Matrix Coeff")
{
ND_FECollection fec_nd(order, dimension);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
BilinearForm bfa(&fespace_nd);
bfa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bfa.Assemble();
bfa.Finalize();
BilinearForm bpa(&fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new VectorFEMassIntegrator(M3_coeff));
bpa.Assemble();
GridFunction x(&fespace_nd), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
}
TEST_CASE("3D Bilinear Weak Curl Integrators Partial Assembly",
"[MixedVectorWeakCurlIntegrator]"
"[BilinearFormIntegrator]"
"[PartialAssembly]"
"[GPU]")
{
auto order = GENERATE(1, 2);
CAPTURE(order);
int dim = 3;
FunctionCoefficient q3_coeff(coeffFunction);
VectorFunctionCoefficient F3_coeff(dim, vectorCoeffFunction);
auto mesh_fname =
GENERATE("../../data/fichera-amr.mesh", "../../data/ball-nurbs.mesh");
CAPTURE(mesh_fname);
Mesh mesh(mesh_fname);
REQUIRE(mesh.Dimension() == dim);
REQUIRE(mesh.SpaceDimension() == dim);
// convert nurbs into piecewise-quadratic curved mesh
if (mesh.NURBSext)
{
mesh.UniformRefinement();
mesh.SetCurvature(2);
}
SECTION("RT to ND No Coeff")
{
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
RT_FECollection fec_rt(order - 1, dim);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator);
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator);
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
REQUIRE(bfa.Height() == y_fa.Size());
REQUIRE(bfa.Width() == x.Size());
REQUIRE(bpa.Height() == y_fa.Size());
REQUIRE(bpa.Width() == x.Size());
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to ND Scalar Coeff")
{
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
RT_FECollection fec_rt(order - 1, dim);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(q3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(q3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
SECTION("RT to ND Diagonal Matrix Coeff")
{
ND_FECollection fec_nd(order, dim);
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
RT_FECollection fec_rt(order - 1, dim);
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
MixedBilinearForm bfa(&fespace_rt, &fespace_nd);
bfa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(F3_coeff));
bfa.Assemble();
bfa.Finalize();
MixedBilinearForm bpa(&fespace_rt, &fespace_nd);
bpa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
bpa.AddDomainIntegrator(new MixedVectorWeakCurlIntegrator(F3_coeff));
bpa.Assemble();
GridFunction x(&fespace_rt), y_fa(&fespace_nd), y_pa(&fespace_nd);
x.Randomize(1234);
bfa.Mult(x, y_fa);
bpa.Mult(x, y_pa);
y_pa -= y_fa;
REQUIRE( y_pa.Normlinf() == MFEM_Approx(0_r) );
}
}
} // namespace pa_coeff
+105 -8
View File
@@ -17,12 +17,63 @@ using namespace mfem;
namespace project_bdr
{
void Func_3D_lin(const Vector &x, Vector &v)
TEST_CASE("3D ProjectBdrCoefficient",
"[GridFunction]"
"[NCMesh]")
{
v.SetSize(3);
v[0] = 1.234 * x[0] - 2.357 * x[1] + 3.572 * x[2];
v[1] = 2.537 * x[0] + 4.321 * x[1] - 1.234 * x[2];
v[2] = -2.572 * x[0] + 1.321 * x[1] + 3.234 * x[2];
const char *mesh_file = GENERATE("data/hex-nc-cross.mesh",
"data/tet-nc-cross.mesh");
CAPTURE(mesh_file);
constexpr int order = 3;
constexpr real_t freq = 5.0;
// Attributes
Array<int> bdr_attr(2);
bdr_attr = 0;
bdr_attr[1] = 1;
// Coefficient
FunctionCoefficient coeff([&](const Vector &x)
{
return cos(freq * M_PI * x[0])
* cos(freq * M_PI * x[1])
* cos(freq * M_PI * x[2]);
});
// Vertex-based mesh
Mesh mesh_v(mesh_file, 1, 1);
H1_FECollection fec_v(order, mesh_v.Dimension());
FiniteElementSpace fes_v(&mesh_v, &fec_v);
GridFunction gf_v(&fes_v);
gf_v = 0.0;
gf_v.ProjectBdrCoefficient(coeff, bdr_attr);
// Nodal mesh
Mesh mesh_n(mesh_file, 1, 1);
mesh_n.SetCurvature(order, true);
H1_FECollection fec_n(order, mesh_n.Dimension());
FiniteElementSpace fes_n(&mesh_n, &fec_n);
GridFunction gf_n(&fes_n);
gf_n = 0.0;
gf_n.ProjectBdrCoefficient(coeff, bdr_attr);
gf_n -= gf_v;
REQUIRE(gf_n.Norml2() == MFEM_Approx(0.0));
}
void Func_lin(const Vector &x, Vector &v)
{
const int dim = x.Size();
v.SetSize(dim);
v[0] = 1.234 * x[0] - 2.357 * x[1];
v[1] = 2.537 * x[0] + 4.321 * x[1];
if (dim == 3)
{
v[0] += 3.572 * x[2];
v[1] -= 1.234 * x[2];
v[2] = -2.572 * x[0] + 1.321 * x[1] + 3.234 * x[2];
}
}
TEST_CASE("3D ProjectBdrCoefficientNormal Vector",
@@ -41,7 +92,7 @@ TEST_CASE("3D ProjectBdrCoefficientNormal Vector",
Mesh mesh = Mesh::MakeCartesian3D(
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
VectorFunctionCoefficient funcCoef(dim, Func_lin);
SECTION("3D GetVectorValue tests for element type " +
std::to_string(type))
@@ -133,7 +184,7 @@ TEST_CASE("3D ProjectBdrCoefficientNormal Scalar",
Mesh mesh = Mesh::MakeCartesian3D(
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
VectorFunctionCoefficient funcCoef(dim, Func_lin);
SECTION("3D GetVectorValue tests for element type " +
std::to_string(type))
@@ -227,7 +278,7 @@ TEST_CASE("3D ProjectBdrCoefficientTangent",
Mesh mesh = Mesh::MakeCartesian3D(
n, n, n, (Element::Type)type, 2.0, 3.0, 5.0);
VectorFunctionCoefficient funcCoef(dim, Func_3D_lin);
VectorFunctionCoefficient funcCoef(dim, Func_lin);
SECTION("3D GetVectorValue tests for element type " +
std::to_string(type))
@@ -305,4 +356,50 @@ TEST_CASE("3D ProjectBdrCoefficientTangent",
}
}
TEST_CASE("ProjectBdrCoefficientTangent with IntegratedGLL",
"[GridFunction]"
"[VectorGridFunctionCoefficient]")
{
const int dim = GENERATE(2, 3);
CAPTURE(dim);
Mesh mesh = (dim == 2) ?
Mesh::MakeCartesian2D(1, 2, Element::QUADRILATERAL,
true, 2.0, 5.0) :
Mesh::MakeCartesian3D(1, 1, 2, Element::HEXAHEDRON,
2.0, 3.0, 5.0);
mesh.EnsureNodes();
mesh.EnsureNCMesh(false);
VectorFunctionCoefficient func_coef(dim, Func_lin);
Array<int> all_bdr(mesh.bdr_attributes.Max());
all_bdr = 1;
for (int order = 1; order <= 4; order++)
{
CAPTURE(order);
ND_FECollection nd_fec(order, dim, BasisType::GaussLobatto,
BasisType::IntegratedGLL);
FiniteElementSpace nd_fespace(&mesh, &nd_fec);
GridFunction volume_projection(&nd_fespace);
GridFunction boundary_projection(&nd_fespace);
volume_projection.ProjectCoefficient(func_coef);
boundary_projection = 0.0;
boundary_projection.ProjectBdrCoefficientTangent(func_coef, all_bdr);
Array<int> ess_vdofs;
nd_fespace.GetEssentialVDofs(all_bdr, ess_vdofs);
real_t max_error = 0.0;
for (int i = 0; i < ess_vdofs.Size(); i++)
{
if (ess_vdofs[i])
{
max_error = std::max(max_error, std::abs(
boundary_projection[i] -
volume_projection[i]));
}
}
REQUIRE(max_error == MFEM_Approx(0.0));
}
}
} // namespace project_bdr
+19 -2
View File
@@ -164,12 +164,29 @@ TEST_CASE("ComplexHypreParMatrix GetSystemMatrix",
a.AddDomainIntegrator(new VectorFEMassIntegrator(one),
new VectorFEMassIntegrator(one));
a.Assemble();
// 2. Test ParSesquilinearForm::FormSystemMatrix directly and verify that
// essential entries on the imaginary diagonal are zero.
OperatorPtr Ah;
a.FormSystemMatrix(ess_tdof_list, Ah);
ComplexHypreParMatrix *A_complex = Ah.Is<ComplexHypreParMatrix>();
REQUIRE(A_complex != nullptr);
Vector diag;
A_complex->imag().GetDiag(diag);
const Array<int> &ess_tdofs = ess_tdof_list;
const Vector &diag_h = diag;
ess_tdofs.HostRead();
diag_h.HostRead();
for (const int tdof : ess_tdofs)
{
REQUIRE(diag_h[tdof] == 0.0);
}
// 3. Test the call to ComplexHypreParMatrix::GetSystemMatrix and destroying
// the returned matrix.
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 2. Test the call to ComplexHypreParMatrix::GetSystemMatrix and destroying
// the returned matrix.
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
delete A;
}
+58
View File
@@ -494,6 +494,64 @@ TEST_CASE("Batched Linear Algebra",
}
}
#ifdef MFEM_USE_EXCEPTIONS
namespace
{
DenseTensor MakeSingularBatchedMatrices()
{
const int n = 3;
const int n_mat = 3;
DenseTensor A_batch(n, n, n_mat);
for (int i = 0; i < n_mat; ++i)
{
DenseMatrix &A = A_batch(i);
A = 0.0;
for (int j = 0; j < n; ++j)
{
A(j, j) = 2.0 + i + j;
}
}
DenseMatrix &singular = A_batch(1);
singular = 0.0;
singular(0, 0) = 1.0;
singular(2, 2) = 1.0;
return A_batch;
}
}
TEST_CASE("Batched LU factorization failure handling",
"[DenseMatrix][GPU]")
{
auto backend = GENERATE(BatchedLinAlg::NATIVE,
BatchedLinAlg::GPU_BLAS,
BatchedLinAlg::MAGMA);
if (!BatchedLinAlg::IsAvailable(backend)) { return; }
CAPTURE(backend);
SECTION("LUFactor")
{
DenseTensor A_batch = MakeSingularBatchedMatrices();
Array<int> P;
REQUIRE_THROWS_WITH(BatchedLinAlg::Get(backend).LUFactor(A_batch, P),
Catch::Matchers::Contains(
"Batch LU factorization failed"));
}
SECTION("Invert")
{
DenseTensor A_batch = MakeSingularBatchedMatrices();
REQUIRE_THROWS_WITH(BatchedLinAlg::Get(backend).Invert(A_batch),
Catch::Matchers::Contains(
"Batch LU factorization failed"));
}
}
#endif
TEST_CASE("DenseTensor copy", "[DenseMatrix][DenseTensor]")
{
DenseTensor t1(2,3,4);
+1 -1
View File
@@ -152,7 +152,7 @@ TEST_CASE("GlobalBBoxTensorGridMap Parallel",
std::map<int, std::vector<int>> pt_to_procs;
map.MapPointsToProcs(centers, 1, pt_to_procs);
REQUIRE(pt_to_procs.size() == nel + 1);
REQUIRE(pt_to_procs.size() == (unsigned)nel + 1);
for (int i = 0; i < nel; i++)
{
std::vector<int> procs = pt_to_procs[i];
+60
View File
@@ -304,6 +304,66 @@ TEST_CASE("pNCMesh PA diagonal", "[Parallel], [NCMesh]")
}
} // test case
TEST_CASE("ParNCMesh Rebalance preserves element attributes",
"[Parallel], [NCMesh]")
{
const int rank = Mpi::WorldRank();
const int nranks = Mpi::WorldSize();
if (nranks < 2) { return; }
auto mesh_fname = GENERATE("../../data/star.mesh",
"../../data/fichera.mesh");
CAPTURE(mesh_fname);
auto CheckRebalance = [rank, nranks, mesh_fname](bool refine,
bool custom_partition)
{
Mesh mesh(mesh_fname);
mesh.EnsureNCMesh();
ParMesh pmesh(MPI_COMM_WORLD, mesh);
const int attribute = 1234 + (custom_partition ? rank : 0);
for (int i = 0; i < pmesh.GetNE(); i++)
{
pmesh.SetAttribute(i, attribute);
}
pmesh.SetAttributes();
if (refine)
{
Array<int> refinements;
if (pmesh.GetNE() && (custom_partition || rank == 0))
{
refinements.Append(0);
}
pmesh.GeneralRefinement(refinements);
}
int expected_attribute = attribute;
if (custom_partition)
{
// Move every element to the next rank, as in GitHub issue #4009.
Array<int> partition(pmesh.GetNE());
partition = (rank + 1) % nranks;
pmesh.Rebalance(partition);
expected_attribute = 1234 + (rank + nranks - 1) % nranks;
}
else
{
pmesh.Rebalance();
}
for (int i = 0; i < pmesh.GetNE(); i++)
{
CHECK(pmesh.GetAttribute(i) == expected_attribute);
}
};
SECTION("Custom partition, unrefined") { CheckRebalance(false, true); }
SECTION("Custom partition, refined") { CheckRebalance(true, true); }
SECTION("Default partition, refined") { CheckRebalance(true, false); }
}
TEST_CASE("EdgeFaceConstraint", "[Parallel], [NCMesh]")
{
auto exact_soln = [](const Vector& x)