Compare commits

...
Author SHA1 Message Date
camierjs f7e0db2541 Merge branch 'nlconvpa' into FEBioMFEMiFSI-gpu 2026-07-14 14:10:20 -07:00
camierjs eab6bfdb02 Merge branch 'vecdivpa' into FEBioMFEMiFSI-gpu 2026-07-14 14:09:18 -07:00
camierjs 8d70ce79e8 Merge branch 'vecmasspa' into FEBioMFEMiFSI-gpu 2026-07-14 14:08:54 -07:00
John Camier 472da91016 Merge branch 'master' into vecmasspa 2026-07-14 14:03:32 -07:00
John Camier 1871a7122e Merge branch 'master' into vecdivpa 2026-07-14 14:03:30 -07:00
John Camier cff5d989f7 Merge branch 'master' into nlconvpa 2026-07-14 14:03:26 -07:00
Veselin Dobrev 9a87b34c47 Merge pull request #5397 from mfem/spde_small_fix
Fix in the SPDE documentation - a factor of 2
2026-07-14 12:19:15 -07:00
Veselin Dobrev 80e40ace14 Merge pull request #5372 from mfem/bugfix-project
bugfix for coefficient project to quadrature function
2026-07-14 12:18:12 -07:00
Veselin Dobrev f5d71a2798 Merge pull request #5394 from mfem/fix_find_SuiteSparse
fix for finding SuiteSparse with the latest PETSc
2026-07-14 12:16:33 -07:00
Veselin Dobrev 9981355ba2 Merge pull request #5230 from mfem/lorentz-device
Lorentz with particles on device
2026-07-14 12:14:48 -07:00
camierjs 86aebd39dc Add AddSpecialization to register all VectorConvectionNLF kernels: AddMultPA, AddMultGrad & GradDiag 2026-07-14 11:20:44 -07:00
camierjs f2fa9f1295 Use CoefficientVector constructor directly, use COMPRESSED storage 2026-07-13 16:34:27 -07:00
camierjs 572cda7deb Remove duplicate 'VectorConvectionNLF' prefix for nonlininteg registered kernels 2026-07-13 16:23:52 -07:00
camierjs 3f5dfc8bfd Avoid the std::exchange in test_pa_nlvc unit tests 2026-07-13 16:02:23 -07:00
camierjs 6086293e35 Avoid long lines in test_pa_diagonal unit tests 2026-07-13 15:44:20 -07:00
camierjs 9d729f0c11 Add integ_pa and integ_fa ownership comments in test_pa_kernels unit tests 2026-07-13 15:38:53 -07:00
camierjs 7d1f4ab1ec CHANGELOG, add diagonal test, fix use of += for AssembleDiagonalPA 2026-07-08 15:55:41 -07:00
camierjs 1066ef593f CHANGELOG, 2D mixed-order specializations, 3D MFEM_VERIFY & cleanup 2026-07-08 15:30:22 -07:00
camierjs a5ece9c0ca Add abort for ConvectiveVectorConvectionNLFIntegrator and SkewSymmetricVectorConvectionNLFIntegrator 2026-07-08 15:15:19 -07:00
camierjs 3718cb8248 CHANGELOG, nlvc bench in makefile, align instantiations 2026-07-08 15:10:26 -07:00
camierjs 321961cfc9 Add tests for user specializations 2026-07-08 15:04:30 -07:00
camierjs f0fe1796bf Fix header file kernels so users can instantiate their own specializations 2026-07-08 14:40:44 -07:00
camierjs 89460c70ca Add tests with different orders for the vector and scalar space
Fix header file kernels so users can instantiate their own specializations
2026-07-08 14:17:39 -07:00
John Camier b8671ed8a1 Merge branch 'master' into vecmasspa 2026-07-08 06:55:11 -07:00
John Camier de5ccf68ad Merge branch 'master' into vecdivpa 2026-07-08 06:55:01 -07:00
John Camier d3238fe235 Merge branch 'master' into nlconvpa 2026-07-08 06:54:48 -07:00
Ketan Mittal c9b2ed7a65 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-07-07 14:14:41 -07:00
Ketan Mittal 9eaa3cdf0a merge with master and resolve conflicts 2026-07-07 14:14:25 -07:00
Ketan MittalSeth Wattscopilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com>Tzanio Kolev
17d1afc3b7 Option to specify increase in axis-aligned bounding box size for surface meshes with FindPointsGSLIB (#5259)
* initial commit - working alg

* Add new source files to CMakeLists.txt

* fix bdr tol when min bb size is specified

* support for triangles in surface mesh capability. tested for mixed meshes as well

* empty partition fix and uninitialized values for surface mesh

* fix variable naming and getboundingboxmesh on device

* change semantics of bounding box input for surface meshes

* make style

* documentation and clean up

* fix device access

* update serial miniapp and rename some variables

* minor

* add unit tests for surface meshes

* consolidate shared machinery in a helper file

* rename bb_t

* rename some functions and clean up

* minor

* simplify includes

* remove unused parameter and improve documentation

* restore whitespace

* Make gslib local helpers static

* reviewer comments

* fix edge initialization

* add the new kernel helper in CMakeLists.txt

* manage life of crystal router object in FindPointsGSLIB

* get rid of unnecessary MFEM_DEVICE_SYNC

---------

Co-authored-by: Seth Watts <watts24@llnl.gov>
Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com>
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-07-07 13:35:30 -07:00
Tzanio Kolev 0eaa3b521a Merge branch 'master' into spde_small_fix 2026-07-07 13:31:32 -07:00
Tzanio Kolev 46d05563d9 Merge pull request #5379 from mfem/speed-up-gitignore-job
Optimize gitignore job
2026-07-07 13:05:55 -07:00
Tzanio Kolev 27c1ea66a4 Merge pull request #5380 from mfem/add-workflow-path-filters
Add path filters to non-required workflows
2026-07-07 13:04:03 -07:00
Tzanio Kolev 7b69944246 Merge branch 'master' into lorentz-device 2026-07-07 11:56:03 -07:00
John Camier 0e4d208f06 Merge branch 'master' into vecmasspa 2026-07-07 17:05:58 +02:00
John Camier eb04f3c1ea Merge branch 'master' into vecdivpa 2026-07-07 17:05:56 +02:00
John Camier 7c50e9f807 Merge branch 'master' into nlconvpa 2026-07-07 17:05:48 +02:00
blaz fa13c848c9 fix in the documentation - a factor of 2 2026-07-07 00:13:18 -07:00
Veselin Dobrev 8d09be7080 Merge pull request #5330 from mfem/hcurl_domain_lf
Add device assembly support for H(curl) VectorFEDomainLFIntegrator
2026-07-06 14:53:56 -07:00
Veselin Dobrev 7f71dacae6 Merge pull request #5388 from mfem/nbeams/fix-gko-hypre-type
Fix gko_hypre_* types creation
2026-07-06 14:52:14 -07:00
camierjs da31afdc55 Merge branch 'vecdivpa' into FEBioMFEMiFSI-gpu 2026-07-06 13:36:01 -07:00
camierjs 3a9ada1de0 Merge branch 'vecmasspa' into FEBioMFEMiFSI-gpu 2026-07-06 13:35:13 -07:00
bslazarov a7b74155f7 fix for finding SuiteSparse with the latest PETSc 2026-07-04 18:40:59 -07:00
John Camier 0c20eef8fe Merge branch 'master' into lorentz-device 2026-07-03 19:04:50 +02:00
John Camier 842514bde2 Merge branch 'master' into bugfix-project 2026-07-03 19:03:33 +02:00
John Camier ee498a19f7 Merge branch 'master' into add-workflow-path-filters 2026-07-03 19:03:11 +02:00
camierjs d93bca38aa Cleanup 2026-07-02 11:16:39 -07:00
camierjs 37c20ff70e Rework ElasticityAssembleDiagonalPA to avoid using scratch memory 2026-07-02 10:48:35 -07:00
camierjs ad804074f9 Fix ElasticityIntegrator AssembleDiagonalPA/AddMultPA QVec size 2026-07-02 10:48:35 -07:00
camierjs 2197dd8b06 Adjust SmemPAVectorMassAssembleDiagonal3D 2026-07-02 10:48:35 -07:00
camierjs b40de0e4e4 2D/3D VectorMassAssembleDiagonalPA specialized on T_Q1D 2026-07-02 10:48:35 -07:00
camierjs 9f4c3f8cbf Use max D1D/Q1D instead of T1D 2026-07-02 10:01:36 -07:00
camierjs 28eb5906f2 Use DofQuadLimits instead of local T_MDQ 2026-07-02 09:49:47 -07:00
camierjs 8c9987e63a With style 2026-07-02 09:34:49 -07:00
camierjs 3495617be6 Cleanup nlvc tests and add libCEED verifications 2026-07-02 09:17:02 -07:00
John Camier bd5f9d80b4 Merge branch 'master' into speed-up-gitignore-job 2026-07-02 18:06:03 +02:00
camierjs 07ebe7889e Cleanup & use MFEM_GENERATE_RANGES instead of overload functions 2026-07-02 07:53:46 -07:00
camierjs 18668ddca6 Direct threads and cmath for win32 2026-07-01 15:55:18 -07:00
camierjs 5e4b69f3d8 Cleanup bilininteg_vecdiv_pa 2026-07-01 14:32:50 -07:00
camierjs bfe77c97f2 Simplify tests unit test_pa_vecdiv 2026-07-01 14:12:11 -07:00
camierjs 7091d4ceb1 Improved PA VectorDivergenceIntegrator
Add shared-memory PA kernels with kernel registration, transpose support,
and unit tests.
2026-07-01 13:34:44 -07:00
camierjs e0ecd9b8ff Bring stacked changes from vecdivpa 2026-07-01 13:32:29 -07:00
John Camier af0f8520d6 Merge branch 'master' into nlconvpa 2026-07-01 21:46:46 +02:00
Veselin Dobrev 6ee3bbde89 Merge pull request #5346 from nmnobre/hypremat
Preemptively delete rownnz if ownership flags set to -1
2026-07-01 12:18:03 -07:00
Veselin Dobrev 92f4fe3bd0 Merge pull request #5383 from mfem/raja-stream-fix
Raja stream fix
2026-07-01 10:02:00 -07:00
nbeams 0171b4b02d Only set gko_hypre_* types when building with MPI 2026-06-30 21:20:25 +00:00
John Camier 68e3a929c2 Merge branch 'master' into lorentz-device 2026-06-30 17:51:21 +02:00
John Camier 06d18956f8 Merge branch 'master' into bugfix-project 2026-06-27 20:14:54 +02:00
John Camier f0d9a81fd4 Merge branch 'master' into nlconvpa 2026-06-27 20:13:25 +02:00
John Camier 60c2ac77d1 Merge branch 'master' into raja-stream-fix 2026-06-27 20:04:17 +02:00
John Camier ee23534091 Merge branch 'master' into speed-up-gitignore-job 2026-06-27 20:03:58 +02:00
John Camier c0da3d6aa9 Merge branch 'master' into add-workflow-path-filters 2026-06-27 20:01:48 +02:00
Andrew Ho fef38a9fd2 RAJA resources appear to be relatively lightweight, just construct it when needed 2026-06-26 14:45:43 -07:00
Tzanio Kolev 01aa047be1 Merge pull request #5373 from mfem/fix-project-bdr-types
[BUG] Fixed type narrowing in ProjectBdrCoefficientNormal unit test
2026-06-26 12:11:27 -07:00
Tzanio Kolev 8c68e8402f Merge branch 'master' into speed-up-gitignore-job 2026-06-26 11:44:38 -07:00
Tzanio Kolev a24e3f6dfb Merge pull request #5365 from mfem/fix-cmake-parallel
Fix CMake parallel build issue
2026-06-26 11:34:26 -07:00
Tzanio Kolev 950f406803 Merge pull request #5377 from mfem/fix-specializations
Fix AddSpecialization header includes
2026-06-26 11:33:24 -07:00
Tzanio Kolev ade7aedfc6 Merge pull request #5126 from mfem/support-shared-build-with-fetching
Support shared MFEM build when fetching TPLs
2026-06-26 11:32:12 -07:00
Andrew Ho 79bca13634 Don't insist on RAJA/CAMP always using default stream and create an internal resource with the default stream 2026-06-25 21:59:07 -07:00
camierjs e1f7df8d44 Merge branch 'master' into nlconvpa 2026-06-25 10:48:31 +02:00
John Camier 5048b1a219 Merge branch 'master' into lorentz-device 2026-06-25 07:47:47 +02:00
John Camier b84988d5f1 Merge branch 'master' into bugfix-project 2026-06-25 07:44:37 +02:00
John Camier 55aa12823c Merge branch 'master' into fix-specializations 2026-06-25 07:44:01 +02:00
Andrew Ho 80555fa132 added documentation 2026-06-24 20:45:13 -07:00
Tara Drwenski 9a98c2be01 Revert "Test: Comment out something from gitignore to test gitignore job"
This reverts commit 16f9cb63a1.
2026-06-24 14:12:34 -07:00
Tzanio Kolev 268231dd09 Merge branch 'master' into hcurl_domain_lf 2026-06-24 12:47:14 -07:00
Tara Drwenski 10880e5ad2 Add path filters to non-required workflows 2026-06-24 11:10:07 -07:00
Tara Drwenski 3df2f14eb9 Merge branch 'master' into speed-up-gitignore-job 2026-06-24 11:06:31 -07:00
Tzanio Kolev 21f1404580 Merge branch 'master' into support-shared-build-with-fetching 2026-06-24 11:03:58 -07:00
Tzanio Kolev 63a38bed84 Merge branch 'master' into hypremat 2026-06-24 11:03:55 -07:00
Tzanio Kolev 867c0a0e7e Merge pull request #5174 from mfem/tmop-mesh-validity
Ensuring mesh validity during mesh optimization using bounds on Jacobian determinant
2026-06-24 11:02:33 -07:00
Tzanio Kolev 270c4d5175 Merge pull request #5374 from mfem/fix-macos-runner
Fix MacOS runner issue
2026-06-24 10:59:51 -07:00
Veselin Dobrev 0779fbbc72 Merge pull request #5302 from lindsayad/bel-sorting
ReorderElements: sort boundary elements by face index after reordering
2026-06-24 10:31:39 -07:00
Tara Drwenski 16f9cb63a1 Test: Comment out something from gitignore to test gitignore job 2026-06-24 10:23:58 -07:00
Tara Drwenski 58da879f06 Remove mfem-analysis from badges in the contributing guide 2026-06-24 10:23:58 -07:00
Tara Drwenski 09b9e1c775 Style fix: use YES instead of true to match current style 2026-06-24 10:23:58 -07:00
Tara Drwenski 1ac86956d1 Move gitignore check from own workflow to builds-and-tests to avoid rebuilding 2026-06-24 08:32:37 -07:00
Andrew Ho 48622d3f1e need to include simplices headers in order for AddSpecialization to work 2026-06-23 16:12:38 -07:00
Tara Drwenski 5cd7f88da0 Remove XCode version setup which breaks on runner update 2026-06-23 11:46:48 -07:00
Tzanio Kolev cc6a34b575 Merge pull request #5358 from mfem/ci-num-tasks
In CI, auto-detect the number of build tasks
2026-06-23 11:45:03 -07:00
Jan Nikl 9cf0b8cb08 Fixed type narrowing in ProjectBdrCoefficientNormal unit test. 2026-06-23 10:22:52 -07:00
Andrew Ho 4843835f98 bugfix where coefficient projection assumes quadrature function is valid on host 2026-06-23 09:59:19 -07:00
Ketan Mittal 161630bf12 fix edge initialization for surface kernels 2026-06-22 16:11:56 -07:00
Veselin Dobrev fa9fdc6576 Address reviewer comments 2026-06-22 10:24:28 -07:00
Nuno NobreandJan Nikl 03c99b8dfe Apply minor rephrasing suggestions
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-06-22 18:08:23 +01:00
Ketan Mittal b65e7ee791 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-06-22 09:29:16 -07:00
Ketan Mittal 473084c8c5 nvcc fix for host/device lambdas 2026-06-22 09:28:55 -07:00
Tzanio Kolev e25f778f7b Reorganized CHANGELOG 2026-06-22 08:56:58 -07:00
Ketan Mittal 2f0bbc6fc4 Merge branch 'master' into lorentz-device 2026-06-22 08:55:49 -07:00
Tzanio Kolev b5baebd0b3 Merge branch 'master' into tmop-mesh-validity 2026-06-22 08:40:47 -07:00
Tzanio Kolev c9ecc5b466 Merge pull request #5205 from mfem/trace-p-ref
PRefinementTransferOperator for trace spaces
2026-06-22 08:40:35 -07:00
Tzanio Kolev f38de5059e Merge branch 'master' into tmop-mesh-validity 2026-06-22 08:39:13 -07:00
Tzanio Kolev 0ff0e1c76b Merge pull request #5223 from mfem/jdongg/pa-simplices
Partial assembly algorithms for Bernstein basis on simplicial meshes [pa-simplices-dev]
2026-06-22 08:37:01 -07:00
Ketan Mittal c7cba857af fix tag access in Get/SetParticle 2026-06-19 11:00:15 -07:00
Ketan Mittal 85a16c2e43 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-06-19 09:51:48 -07:00
Ketan Mittal 3c4d103982 do compact transfer between host-device during redistribute 2026-06-19 09:51:24 -07:00
Veselin Dobrev fcca0eaa71 Replace the 'mfem/github-actions' branch name 'update-num-tasks' with the
tag 'v2.7'.
2026-06-18 14:58:00 -07:00
Ketan Mittal 0bfa06f87d merge with master and resolve conflicts 2026-06-18 12:40:44 -07:00
Ketan Mittal 3aed584bf7 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-18 08:59:48 -07:00
Ketan Mittal c4d944cfad restore tolerance for barrier 2026-06-18 08:59:33 -07:00
Socratis Petrides c05114555f Merge branch 'trace-p-ref' of github.com:mfem/mfem into trace-p-ref 2026-06-17 18:41:27 -07:00
Socratis Petrides dfc15d77e5 Tzanio/codex review 2026-06-17 18:40:16 -07:00
Tzanio Kolev 25e5ac7db0 Merge branch 'master' into trace-p-ref 2026-06-17 09:14:57 -07:00
John Camier d0533903f3 Merge branch 'master' into bel-sorting 2026-06-17 07:25:22 -07:00
John Camier 38ed1e049b Merge branch 'master' into lorentz-device 2026-06-17 05:58:47 -07:00
Socratis Petrides 48e258a60c add ownership documentation 2026-06-16 17:37:39 -07:00
Socratis Petrides e1b9c355e6 fix local prolonation issue in case of vdim > 1 2026-06-16 17:36:49 -07:00
Socratis Petrides c3db6ed873 fix headers 2026-06-16 17:35:50 -07:00
Socratis PetridesandTzanio Kolev 56b3e1ecc6 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:58:22 -07:00
Socratis PetridesandTzanio Kolev 5385e090f7 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:54 -07:00
Socratis PetridesandTzanio Kolev ecb7767b70 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:21 -07:00
Socratis PetridesandTzanio Kolev 42bf39cd8c Update miniapps/dpg/makefile
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:01 -07:00
Tzanio Kolev 716f92d636 Merge pull request #5309 from mfem/fix-rocm7-hipsparse
Fix hipsparse in ROCm 7
2026-06-16 11:28:45 -07:00
Tzanio Kolev efdc17ce07 Merge pull request #5339 from mfem/tmop-AL-multiZ
TMOP - allow multiple fields in adaptive limiting.
2026-06-16 11:24:22 -07:00
John Camier ffab9d82fa Merge branch 'master' into bel-sorting 2026-06-16 07:23:55 -07:00
camierjs 891eaf2829 Fix missing B/Bt for OCCA backend 2026-06-16 07:07:34 -07:00
Ketan Mittal 053804f759 Merge branch 'master' into tmop-mesh-validity 2026-06-15 13:03:44 -07:00
Andrew Ho a8af93c4ec Merge branch 'master' into trace-p-ref 2026-06-15 12:19:38 -07:00
Andrew Ho ac7fa02927 Merge branch 'master' into support-shared-build-with-fetching 2026-06-15 11:59:04 -07:00
Ketan Mittal eeca0b4cd5 reviewer comments 2026-06-14 19:52:40 -07:00
Victor A. P. Magri ad18d18be1 Fix CMake parallel build issue with auto-fetched hypre 2026-06-13 00:10:33 -04:00
jdongg 00a1f5ed64 Add explicit return outside of if statement for mass and diffusion kernels 2026-06-12 08:05:30 -07:00
Veselin Dobrev a76b0dbc0e In GitHub CI, in the sanitizer builds of hypre and metis, do not set the
environment variables CXXFLAGS and LDFLAGS -- there are not needed and were
causing the hypre build to fail.
2026-06-12 02:43:18 -07:00
Veselin Dobrev 59ce107b33 In GitHub CI, fix caching key 2026-06-12 02:21:59 -07:00
Veselin Dobrev ce1036ccc8 Another try to define ACTIONS_VERSION in
.github/actions/sanitize/config/action.yml
in a way that it will be picked up by other sanitizer actions.
2026-06-12 01:06:06 -07:00
Veselin Dobrev b6861ed26f In GitHub CI, define ACTIONS_VERSION in
.github/actions/sanitize/config/action.yml
instead of
   .github/workflows/sanitizers.yml
where it was not picked up by some called actions.
2026-06-12 00:53:39 -07:00
Veselin Dobrev 8522cceb89 In CI, variables like ${{env.VAR}} cannot be used in "uses:" fields. 2026-06-11 23:46:43 -07:00
Veselin Dobrev dec9509a5d In GitHub CI, use and environment variable to set the version (branch or tag)
of the 'mfem/github-actions' to use.
2026-06-11 18:56:46 -07:00
Vladimir Z Tomov 5f02c6f64b fixed nvcc error for protected functions 2026-06-11 14:50:52 -07:00
jdongg 8a69b52f48 Add explicit return in diffusion and mass kernels in else branch. 2026-06-10 20:52:49 -07:00
John Camier f51b8c2047 Merge branch 'master' into nlconvpa 2026-06-10 17:01:51 -07:00
Tzanio Kolev 51a0f8accb Merge branch 'master' into jdongg/pa-simplices 2026-06-10 15:54:26 -07:00
Ketan Mittal eb4fa33a7b Merge branch 'master' of https://github.com/mfem/mfem into lorentz-device 2026-06-10 15:50:13 -07:00
Andrew Ho 42d7d20e43 Merge branch 'master' into hcurl_domain_lf 2026-06-10 15:27:57 -07:00
Nuno Nobre 75f8ca8cd4 Switch to GetHypreMemoryLocation() 2026-06-10 18:33:14 +01:00
Nuno Nobre 36e915f226 Guard hypre_CSRMatrixMemoryLocation w/ hypre version check 2026-06-10 18:11:40 +01:00
Nuno Nobre f236d70a19 Avoid calling HypreParMatrix::Write() again and clearing ptrs 2026-06-10 13:58:58 +01:00
Nuno Nobre 85e33c6645 Use hypre_CSRMatrix{I,J,Data,OwnsData} and update explainer comment 2026-06-10 13:22:46 +01:00
Veselin Dobrev 7b62f035a5 Add a fix for the issue -- alternative to the solution in PR #5346. 2026-06-10 13:14:28 +01:00
Nuno Nobre 81fb02389f Revert "Preemptively delete rownnz if ownership flags set to -1"
This reverts commit 7a6313d725.
2026-06-10 11:35:16 +01:00
Nuno Nobre 23666fd4d8 Fix missing #ifdef MFEM_USE_MPI in new unit test 2026-06-10 11:25:58 +01:00
Nuno Nobre f3a60e2f08 Merge branch 'master' into hypremat 2026-06-10 11:25:00 +01:00
Veselin Dobrev 8777bc0810 In CI, set the number of build tasks for the 'build-mfem' action automatically 2026-06-09 18:20:58 -07:00
Tzanio Kolev 596b75282d Merge pull request #5356 from mfem/update-codecov-action
Update the versions of the actions from `mfem/github-actions`
2026-06-09 17:55:16 -07:00
Ketan Mittal de97775bd0 Merge branch 'master' into tmop-mesh-validity 2026-06-09 16:11:59 -07:00
Tzanio Kolev 2ed23cec01 Merge pull request #5349 from mfem/raja-launchbounds
Raja launchbounds
2026-06-09 12:53:57 -07:00
Veselin Dobrev 11fbd1f65c In CI, update the actions from mfem/github-actions to v2.6 2026-06-09 12:14:02 -07:00
Will Pazner d3556dc2ac Merge pull request #4412 from mfem/elast-fix-4404
Verify that ordering is byVDIM when using the AMG elasticity solver
2026-06-09 11:24:30 -07:00
John Camier ad453255a5 Merge branch 'master' into bel-sorting 2026-06-09 06:46:23 -07:00
John Camier 2143ed5ca8 Merge branch 'master' into fix-rocm7-hipsparse 2026-06-09 06:46:06 -07:00
John Camier 2f21794999 Merge branch 'master' into hypremat 2026-06-09 06:45:29 -07:00
John Camier 2aa8372cdc Merge branch 'master' into jdongg/pa-simplices 2026-06-09 06:43:22 -07:00
John Camier 8d512c82f4 Merge branch 'master' into nlconvpa 2026-06-09 06:42:23 -07:00
Veselin Dobrev 80c7331a29 In CI, fix the name for the 'upload-coverage' action, to replicate the
complete job name.
2026-06-08 21:36:52 -07:00
Veselin Dobrev 9e31083745 In CI, pass the Codecov token as the environment variable CODECOV_TOKEN 2026-06-08 20:15:57 -07:00
Veselin Dobrev fc8272216b In CI, pass the Codecov token as a secret to the 'upload-coverage' action.
Also, use the CI job name as the name for the 'upload-coverage' action.
2026-06-08 20:01:09 -07:00
Ketan Mittal 0176c7664c merge and resolve conflicts 2026-06-08 19:27:48 -07:00
Ketan Mittal e1c0705785 reviewer comments 2026-06-08 19:24:21 -07:00
Veselin Dobrev 8efa320470 Update codecov action: testing 2026-06-08 14:04:39 -07:00
Andrew Ho 50b8f67bd1 Merge branch 'master' into hcurl_domain_lf 2026-06-08 12:41:22 -07:00
Vladimir Z Tomov a4d114ad6e renamed a function, improved some comments 2026-06-08 10:04:43 -07:00
Ketan Mittal 1e5609f6c9 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-08 09:16:47 -07:00
Ketan Mittal c21d90589b Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-08 09:16:28 -07:00
Ketan Mittal bef7183ce7 documentation and CHANGELOG 2026-06-08 09:16:23 -07:00
Veselin Dobrev db7da59b03 Add a test that reproduces the issue described in PR #5200 2026-06-07 19:30:38 +01:00
Tzanio Kolev c6ea42a681 Merge pull request #5045 from mfem/stefanozampini/petsc-requires-hypre
PETSc: Remove ad-hoc code to support PETSc without hypre
2026-06-06 16:52:11 -07:00
John Camier 28194a6736 Merge branch 'master' into lorentz-device 2026-06-06 16:37:35 -07:00
John Camier 8feb690d6d Merge branch 'master' into nlconvpa 2026-06-06 06:27:04 -07:00
Tzanio Kolev f6eda1fce6 Merge branch 'master' into tmop-mesh-validity 2026-06-05 16:40:10 -07:00
Tzanio Kolev a1bcd7443d Merge branch 'master' into jdongg/pa-simplices 2026-06-05 16:39:29 -07:00
Tzanio Kolev 51955e88ce Merge branch 'master' into raja-launchbounds 2026-06-05 16:32:55 -07:00
Tzanio Kolev 61bb7755a3 Merge pull request #5324 from mfem/multigrid-bcs-fix
Bugfix for GeometricMultigrid with no essential BCs
2026-06-05 16:17:51 -07:00
Tzanio Kolev a7fa61464c Merge pull request #5267 from mfem/print-interfaces-dev
Optional output of material interfaces in parallel
2026-06-05 16:13:18 -07:00
Tzanio Kolev 93329a7ba0 Merge pull request #5124 from yuyangdai/cudss-dev
Add support for parallel NVIDIA's GPU-accelerated direct sparse solver cuDSS solver[cudss-dev]
2026-06-05 16:10:03 -07:00
Ketan Mittal 9c4d742db8 update gridfunction after h-adaptivity 2026-06-05 09:41:42 -07:00
Ketan Mittal 16ac6c40a3 refactor to have TMOPNewtonSolver manage bounding detJ 2026-06-05 09:33:21 -07:00
Andrew Ho 9a69d92756 changelog 2026-06-05 08:20:01 -07:00
yuyangdai fb93e21a0c Update INSTALL file to correct cuDSS library options 2026-06-05 09:16:00 +08:00
jdongg ee2484383c Remove own_rules from intrules.cpp 2026-06-04 15:43:36 -07:00
jdongg 1863dafeca Remove own rules flag and function. 2026-06-04 15:18:35 -07:00
Vladimir Z Tomov a4c7828ac2 pr review comments 2026-06-04 15:07:12 -07:00
Ketan Mittal 0a02f5737a address reviewer comments 2026-06-04 14:00:41 -07:00
Ketan Mittal 154a1840d0 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-04 12:41:56 -07:00
Ketan Mittal 13e83cbd14 minor 2026-06-04 12:41:53 -07:00
Stefano Zampini 19d60f95ec PETSc: Remove ad-hoc code to support PETSc without hypre 2026-06-04 11:27:36 +01:00
Vladimir Z Tomov f64405ebb6 minor 2026-06-03 17:56:37 -07:00
Ketan Mittal 92f2057650 Merge branch 'master' into tmop-mesh-validity 2026-06-03 17:48:17 -07:00
Will Pazner 861d318162 Fix Doxygen comment in GeometricMultigrid 2026-06-03 15:48:39 -07:00
Ketan Mittal f3558cb78c Merge branch 'master' into tmop-AL-multiZ 2026-06-03 10:15:35 -07:00
Andrew Ho 3c5acae825 missing endif 2026-06-02 22:13:12 -07:00
Andrew Ho 9a749f8936 Merge remote-tracking branch 'base/raja-launchbounds' into raja-launchbounds 2026-06-02 22:11:42 -07:00
Andrew Ho c095d055a1 put the check/define for RAJA default stream in backends 2026-06-02 22:11:11 -07:00
Tzanio Kolev d29e914146 Merge branch 'master' into multigrid-bcs-fix 2026-06-02 18:20:52 -07:00
Andrew HoandTom Stitt a2491a0e43 Apply suggestion from @tomstitt
Co-authored-by: Tom Stitt <stitt4@llnl.gov>
2026-06-02 11:08:12 -07:00
Tzanio Kolev 4475a8c6d4 minor
Don't disable reorder_space if static condensation is enabled.
2026-06-02 10:18:15 -07:00
John Camier d5ddb83c9d Merge branch 'master' into jdongg/pa-simplices 2026-06-02 06:08:39 -07:00
John Camier fd55dc64d0 Merge branch 'master' into nlconvpa 2026-06-02 06:08:29 -07:00
yuyangdai b5e99d8d83 Fix CMake script to correctly locate cuDSS library and allocate device memory for csr values when set the cudssMatrix. 2026-06-02 13:31:25 +08:00
daiyuyangandAndrew Ho 91f58c293d Update linalg/cudss.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-06-02 13:17:09 +08:00
Tzanio Kolev 4836e7cb53 Merge branch 'master' into elast-fix-4404 2026-06-01 17:00:44 -07:00
Tzanio Kolev 6451e64637 Merge branch 'master' into cudss-dev 2026-06-01 15:55:15 -07:00
Tzanio Kolev b7ac1963bb Merge pull request #5340 from mfem/update-contact-miniapp-readme
Update Tribol shared libraries
2026-06-01 15:53:50 -07:00
camierjs 869b0e48a6 Add simplices tests with D1D > Q1D 2026-06-01 09:31:24 -07:00
Veselin Dobrev baefb786fb Correction in CHANGELOG 2026-06-01 08:58:54 -07:00
Veselin Dobrev 99523b3a97 Address reviewer feedback: add/improve Doxygen documentation. 2026-06-01 08:43:12 -07:00
Andrew HoandJohn Camier f67c1e5beb Update general/forall.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-06-01 08:31:43 -07:00
Andrew HoandJohn Camier 50af02454c Update general/forall.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-06-01 08:31:33 -07:00
Andrew HoandJohn Camier aa94eb11b7 Update general/forall.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-06-01 08:31:23 -07:00
Andrew HoandJohn Camier b73cc75635 Update general/forall.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-06-01 08:31:09 -07:00
John Camier 75180606f8 Merge branch 'master' into bel-sorting 2026-06-01 08:07:45 -07:00
John Camier 3cac4326e4 Merge branch 'master' into fix-rocm7-hipsparse 2026-06-01 08:07:06 -07:00
Ketan Mittal da5e8b5844 Merge branch 'master' into tmop-mesh-validity 2026-05-31 22:52:14 -07:00
Vladimir Z Tomov 081a90d1fa style 2026-05-31 14:45:56 -07:00
Andrew Ho 7b5f050bdc format to shorter line lengths 2026-05-31 09:47:27 -07:00
Vladimir Z Tomov d06a3eec7f use Vectors instead of allocating hypre vectors in AdvectorCG 2026-05-29 11:14:50 -07:00
Andrew Ho 637fa7cc35 Merge remote-tracking branch 'base/raja-launchbounds' into raja-launchbounds 2026-05-29 11:06:13 -07:00
Andrew Ho 3f332d50b5 some platforms require this to actually be set to 1? 2026-05-29 11:05:20 -07:00
Andrew Ho 782e468325 duplicate code 2026-05-28 17:14:12 -07:00
Andrew Ho 0c59b93ecf flag only needs to be defined, not set to a value
also updated makefile defaults
2026-05-28 16:56:11 -07:00
Andrew Ho 5a0e89cf31 ensure that RAJA/CAMP uses the platform default streams 2026-05-28 16:35:39 -07:00
Andrew Ho 4c3bfce583 Added launch bounds for Raja kernels 2026-05-28 15:38:40 -07:00
Tzanio Kolev 59d6820ff3 Merge branch 'master' into cudss-dev 2026-05-28 14:21:30 -07:00
Tzanio Kolev 9ec379f176 Merge branch 'master' into update-contact-miniapp-readme 2026-05-28 14:19:45 -07:00
jdongg 2766e56f2d Remove GaussJacobi out-of-range error check on alpha and beta before hard-coded cases. Add error check for unsupported geometries in StroudIntegrationRules class. 2026-05-28 12:58:02 -07:00
Tzanio Kolev 72cc503eb9 Updated CHANGELOG 2026-05-28 12:53:25 -07:00
jdongg 0c781120a2 Remove references to InverseDuffyTrans in comments. Add description of on-the-fly inverse Duffy transform in GetRaggedTensorDofToQuad. Fix typos in GaussJacobi MFEM_ABORT. Remove unnecessary if statement in GaussJacobi routine when floating point type is undefined. 2026-05-28 11:51:50 -07:00
camierjs 449c9f5903 Avoid duplicate meshes in test_pa_simplices 2026-05-28 06:27:50 -07:00
Veselin Dobrev 5aff935c98 Minor future-proofing suggested by Copilot.
MFEM_ASSERT does not need to explicitly print the name of the function
because that is already done automatically.
2026-05-27 10:18:49 -07:00
Tzanio Kolev 6f793edfb7 Merge branch 'master' into tmop-AL-multiZ 2026-05-27 09:34:31 -07:00
John Camier 6455a0c1fc Merge branch 'master' into jdongg/pa-simplices 2026-05-27 06:31:47 -07:00
John Camier 16af7365a2 Merge branch 'master' into nlconvpa 2026-05-27 06:30:56 -07:00
Vladimir Z Tomov b19c3c7e01 avoid Makeref 2026-05-26 22:23:20 -07:00
Vladimir Z Tomov fb7c8af59e more Newton iterations in the tmop unit test 2026-05-26 20:32:55 -07:00
Tzanio KolevandCopilot Autofix powered by AI d56298ba17 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 19:02:02 -07:00
Tzanio KolevandCopilot Autofix powered by AI b9d67dec34 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 19:00:55 -07:00
Tzanio Kolev 5b664e393d Merge branch 'master' into print-interfaces-dev 2026-05-26 18:43:14 -07:00
Veselin Dobrev c582282084 Merge pull request #5345 from mfem/copilot-dev
Updates in developer docs + instructions for the GitHub Copilot reviews
2026-05-26 18:27:49 -07:00
Tzanio Kolev 944ece4090 TODO item for ConduitDataCollection::Save() 2026-05-26 17:35:13 -07:00
Veselin DobrevandTzanio Kolev f225d0c3ef Apply suggestions from code review
Remove FIXME comments -- no actions needed.

Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-05-26 16:58:36 -07:00
Tzanio Kolev a9c98c2e3a Merge branch 'master' into print-interfaces-dev 2026-05-26 10:35:15 -07:00
Tzanio KolevandCopilot Autofix powered by AI ac8c2948e7 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 10:02:50 -07:00
Tzanio KolevandCopilot Autofix powered by AI f75b4c10d2 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 10:02:10 -07:00
Mittal, Ketan f6ea1ea0da minor 2026-05-26 10:01:57 -07:00
Nuno Nobre 7a6313d725 Preemptively delete rownnz if ownership flags set to -1 2026-05-26 17:24:58 +01:00
Mittal, Ketan 84bbead832 move lambdas to static functions for nvcc 2026-05-26 09:09:20 -07:00
Veselin Dobrev ecf6b0b44f Address some reviewer suggestions and comments 2026-05-26 08:41:10 -07:00
Vladimir TomovandCopilot Autofix powered by AI ef6510b42e typo
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-25 19:09:36 -07:00
Vladimir TomovandCopilot Autofix powered by AI 27ce32de07 typo
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-25 19:08:24 -07:00
Vladimir Z Tomov 341f6ec683 minor 2026-05-25 19:06:57 -07:00
Vladimir Z Tomov 8c88e1e26a style 2026-05-25 19:01:17 -07:00
Vladimir Z Tomov cabdb42bb4 Remap the adaptive limiting fields as a multi-component vector. 2026-05-25 18:52:04 -07:00
Vladimir Z Tomov e4a69c6245 vis in the mesh-optimizer, shorter 3d runs 2026-05-25 17:38:27 -07:00
Tzanio Kolev 5a52c676e1 Small changes in Copilot instructions 2026-05-25 15:49:19 -07:00
Tzanio Kolev 5be9693235 Initial Copilot instructions 2026-05-25 15:34:45 -07:00
Mittal, Ketan dbb751d7fb Merge branch 'master' of https://github.com/mfem/mfem into lorentz-device 2026-05-25 15:33:29 -07:00
Tzanio Kolev 5421abc4c4 Fixes and updates in CONTRIBUTING.md 2026-05-25 12:58:57 -07:00
Tzanio Kolev f2d3fb45dc Fixes and updates in INSTALL 2026-05-25 12:04:26 -07:00
John Camier e447f090cc Merge branch 'master' into jdongg/pa-simplices 2026-05-24 20:14:10 -07:00
John Camier 239c83a742 Merge branch 'master' into cudss-dev 2026-05-24 20:14:04 -07:00
John Camier b67b1af8f8 Merge branch 'master' into nlconvpa 2026-05-24 20:13:45 -07:00
maxpaik16 c315028a09 Merge branch 'master' into update-contact-miniapp-readme 2026-05-24 19:03:34 -07:00
yuyangdai 99b66f4f45 Update CHANGELOG 2026-05-25 09:20:19 +08:00
Vladimir Z Tomov c3096d08bf Better example with 2 fields in the miniapps. 2026-05-24 18:07:53 -07:00
Vladimir Z Tomov 67b906ea3b Restored the optimization when the coefficients are ConstantCoefficients. 2026-05-24 17:46:37 -07:00
Tzanio Kolev e7be50eb91 Merge pull request #5333 from nmnobre/gslib
FindPointsGSLIB: use parallel-aware ProjectDiscCoefficient
2026-05-24 10:58:16 -07:00
Tzanio Kolev 2dd3915ad5 Merge branch 'master' into gslib 2026-05-23 10:15:21 -07:00
Tzanio Kolev 29b572819a Merge branch 'master' into cudss-dev 2026-05-23 10:14:37 -07:00
Tzanio Kolev 0a0acfda66 Merge pull request #5329 from nmnobre/host
Fix GetEssentialTrueDofsVar memory allocation
2026-05-22 20:15:51 -07:00
Vladimir Z Tomov 9167fc0c58 style 2026-05-22 18:03:28 -07:00
Vladimir Z Tomov e48ffe3579 cleanup 2026-05-22 18:00:32 -07:00
Vladimir Z Tomov a7697c1db1 cleanup 2026-05-22 17:34:05 -07:00
Vladimir Z Tomov 36c1cce132 delta_max per specified function. 2026-05-22 17:09:04 -07:00
maxpaik16 7dbce44472 Update tribol libraries 2026-05-22 16:36:51 -07:00
maxpaik16 c67bd21219 Update tribol shared libraries 2026-05-22 16:35:29 -07:00
Vladimir Z Tomov 38f257cf87 Multiple fields for adaptive limiting - initial pass. 2026-05-22 16:32:34 -07:00
camierjs a0022e0330 Fix number of threads in simplices kernels to allow D1D > Q1D, revert Stroud int rules in ex1[p], fix thread race in SmemPADiffusionApplyTetrahedron 2026-05-22 06:45:02 -07:00
Tzanio Kolev 38f5b93520 Merge branch 'master' into print-interfaces-dev 2026-05-21 09:45:23 -07:00
Tzanio Kolev a078dfd59e Reviewer comments 2026-05-21 09:45:02 -07:00
camierjs 3a2f286e2d Select Stroud integration rule ine ex1[p] when needed, add verification in simplicial kernels that D1D <= Q1D 2026-05-21 08:09:33 -07:00
Tzanio Kolev 66818f5525 Merge branch 'master' into gslib 2026-05-21 07:41:18 -07:00
Tzanio Kolev b810a5e540 Merge branch 'master' into host 2026-05-21 07:41:14 -07:00
Tzanio Kolev e63e421343 Merge branch 'master' into cudss-dev 2026-05-21 07:40:59 -07:00
John Camier 0d3b658dc4 Merge branch 'master' into nlconvpa 2026-05-21 06:05:13 -07:00
Mittal, Ketan 58e6c4fc6d merge with master and resolve conflicts 2026-05-20 11:21:29 -07:00
jdongg 7031f7cb80 Remove pullback flag from simplex unit tests 2026-05-19 22:24:45 -07:00
jdongg afaced7cd6 Rebase InverseDuffyTrans removal onto master. 2026-05-19 21:58:22 -07:00
jdongg d671ca4712 Remove InverseDuffyTrans and perform inverse mapping on the fly in GetRaggedTensorDofToQuad. Remove Pullback option from StroudIntRules object. Remove unused T array from RaggedDofToQuad object and mass+diffusion integrators. 2026-05-19 21:57:06 -07:00
John Camier 1095aee346 Merge branch 'master' into jdongg/pa-simplices 2026-05-19 20:28:53 -07:00
camierjs ba8764300c Revert the nvcc warning fix that will be addressed in another PR 2026-05-19 11:43:11 -07:00
camierjs bd4504d7ae Simplify MDQ for NLVC PA kernels 2026-05-19 11:12:50 -07:00
camierjs f622b53731 Simplify nlvc unit tests 2026-05-19 10:13:31 -07:00
camierjs 510477a204 Fix nvcc with gcc use of literal operator warnings 2026-05-19 09:23:17 -07:00
Nuno Nobre 95acb1f85f Remove unnecessary namespace qualification 2026-05-19 17:15:03 +01:00
camierjs 0b9ae2202c Fix nvcc identifier warnings: preceded by whitespace in a literal operator declaration 2026-05-19 08:38:55 -07:00
camierjs 60641259b2 Revert ex1[p] refinements 2026-05-19 08:27:20 -07:00
camierjs 99e2b5023d Simplify ex1[p] with IsSimplexMesh 2026-05-19 08:26:01 -07:00
Nuno Nobre 1c7261db07 Fence code using parallel objs w/ MPI-conditional directive 2026-05-18 23:53:35 +01:00
Nuno Nobre e1b678664a FindPointsGSLIB: use parallel-aware ProjectDiscCoefficient 2026-05-18 19:02:30 +01:00
John Camier 4c59f6cfe6 Merge branch 'master' into jdongg/pa-simplices 2026-05-18 06:04:46 -07:00
John Camier 60eb714229 Merge branch 'master' into nlconvpa 2026-05-18 06:04:17 -07:00
Nuno NobreandAndrew Ho 8f06539b6e Do not assume true_ess_dofs is empty nor host-allocated
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-05-16 00:32:25 +01:00
Andrew Ho 670dfb2d98 fix compiler warning about missing enum cases 2026-05-15 12:23:24 -07:00
Andrew Ho 4167f0027c combine H(div) and H(curl) linear form extension tests 2026-05-15 12:14:13 -07:00
Andrew Ho b59ccd206d specify only for 3D H(curl) 2026-05-15 11:45:54 -07:00
Ketan Mittal cee913cc04 Merge branch 'master' into tmop-mesh-validity 2026-05-15 10:34:26 -07:00
Andrew Ho 76b033ad36 accidentally duplicated file 2026-05-15 01:19:54 -07:00
Andrew Ho 0cd0a98198 Add device assembly support for H(curl) VectorFEDomainLFIntegrator. 2026-05-15 01:08:22 -07:00
Nuno Nobre 30a8eb5ccf Fix GetEssentialTrueDofsVar memory allocation 2026-05-14 23:58:41 +01:00
Alex Lindsay bd5d928084 Pare down doxygen for GetFace group 2026-05-14 14:48:09 -07:00
John Camier 5e62d62c3d Merge branch 'master' into jdongg/pa-simplices 2026-05-14 13:21:04 -07:00
John Camier 078e59a33b Merge branch 'master' into cudss-dev 2026-05-14 13:20:48 -07:00
John Camier 1d3a723af9 Merge branch 'master' into nlconvpa 2026-05-14 13:20:27 -07:00
Ketan Mittal 5faba43544 Merge branch 'master' into bel-sorting 2026-05-14 10:34:41 -07:00
Alex Lindsay 20805d87b4 Prevent comment lines from going past column 80 2026-05-13 11:33:48 -07:00
Alex Lindsay 0119d25dfc Factor out common testing code 2026-05-13 11:32:12 -07:00
Tzanio KolevandVeselin Dobrev da4e1b5137 Update makefile
Co-authored-by: Veselin Dobrev <v-dobrev@users.noreply.github.com>
2026-05-12 13:54:55 -07:00
John Camier 22c15267f9 Merge branch 'master' into bel-sorting 2026-05-12 06:58:24 -07:00
John Camier 6c83fec2da Merge branch 'master' into fix-rocm7-hipsparse 2026-05-12 06:58:03 -07:00
Ketan Mittal 411a361ebf Merge branch 'master' into lorentz-device 2026-05-09 20:22:29 -07:00
John Camier bc845844ea Merge branch 'master' into jdongg/pa-simplices 2026-05-09 11:31:25 -07:00
John Camier 2e8e4a5377 Merge branch 'master' into nlconvpa 2026-05-09 11:31:10 -07:00
Tzanio Kolev b5a1660c2d Merge branch 'master' into cudss-dev 2026-05-09 10:41:10 -07:00
Will Pazner 5ffc2ef502 Bugfix for GeometricMultigrid with no essential BCs 2026-05-07 18:04:46 -07:00
camierjs 35ceed5193 Cleanup 2026-05-07 11:10:58 -07:00
camierjs ce52e1f51f Use MFEM_FOREACH_THREAD_DIRECT for simplices kernels 2026-05-07 11:01:25 -07:00
camierjs 16a21b366e Fix missing sync thread in smem mass simplices 2026-05-07 10:41:00 -07:00
John Camier d540fa5a12 Merge branch 'master' into bel-sorting 2026-05-07 08:06:44 -07:00
Tzanio Kolev 1221dea58e Merge branch 'master' into fix-rocm7-hipsparse 2026-05-07 07:29:53 -07:00
camierjs a5b9a7948f Add ex1p device simplices sample runs 2026-05-06 17:22:44 -07:00
camierjs 386e7e8c6a Add ex1 device simplices sample runs 2026-05-06 17:15:27 -07:00
camierjs 26077ab9aa Remove the ex_pa_simplices miniapps 2026-05-06 14:22:44 -07:00
Ketan Mittal 0ee57ac12d Merge branch 'master' into lorentz-device 2026-05-05 22:53:21 -07:00
yuyangdai 59cef5f9e3 Add a link for communication layer library in cuDSS 2026-05-06 10:07:31 +08:00
John Camier 05be944a86 Merge branch 'master' into cudss-dev 2026-05-05 15:33:09 -07:00
John Camier 139c3ddaa6 Merge branch 'master' into nlconvpa 2026-05-05 15:32:54 -07:00
camierjs 87bdb90bcc Merge branch 'master' into jdongg/pa-simplices 2026-05-05 11:20:27 -07:00
camierjs 848b54d13a Add pa-simplices to the selected files for formatting 2026-05-05 11:20:15 -07:00
John Camier c103cfa84a Merge branch 'master' into cudss-dev 2026-05-05 06:34:53 -07:00
camierjs 62214f61ae Merge branch 'master' into jdongg/pa-simplices 2026-05-05 06:29:17 -07:00
John Camier dee971c308 Merge branch 'master' into lorentz-device 2026-05-05 06:21:51 -07:00
John Camier 9b93f1c1e1 Merge branch 'master' into bel-sorting 2026-05-05 06:19:44 -07:00
John Camier 24f1022f7d Merge branch 'master' into nlconvpa 2026-05-05 06:19:01 -07:00
John Camier be03c9703f Merge branch 'master' into fix-rocm7-hipsparse 2026-05-05 06:14:52 -07:00
Tom Stitt b885fdc50f fix 2026-05-04 17:18:39 -07:00
Tom Stitt e4c0069ad9 better documentation 2026-05-04 14:41:49 -07:00
Tom StittandAndrew Ho 9fe53c2403 Apply suggestion from @helloworld922
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-05-04 14:14:01 -07:00
Mittal, Ketan bb12710561 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-05-04 14:04:25 -07:00
Mittal, Ketan f3276a0d5d address co-pilot comments 2026-05-04 14:04:11 -07:00
Ketan Mittal 00b9678b1d Merge branch 'master' into tmop-mesh-validity 2026-05-04 10:31:41 -07:00
Tzanio Kolev 426b5dc6cd Merge branch 'master' into lorentz-device 2026-05-02 13:00:07 -07:00
camierjs 5e15a2e29b Cleanup PA simplices tests 2026-04-30 20:56:34 -07:00
camierjs d168f5b489 Revert intrules in PA simplices unit tests 2026-04-30 20:06:24 -07:00
camierjs 7a5d13e86a Fix & simplify ceed benchmark tests 2026-04-30 18:58:43 -07:00
camierjs 2e0d1697b9 Use specific simplices diffusion specializations when Q=P
Use integrations rules from integrators for unit testing
2026-04-30 18:25:13 -07:00
camierjs 87168f51ab Meld back toward master, removing un-warnings 2026-04-30 14:43:49 -07:00
jdongg 2558e32ee0 Switch std::beta to std::tgamma for compliances with C++11 standard 2026-04-30 13:28:32 -07:00
jdongg 7521efeff9 Add unit tests for Stroud quadrature rules as well as general Gauss-Jacobi rules 2026-04-30 13:11:57 -07:00
camierjs 3b23fd4941 Add extra meshs for PA simplices tests 2026-04-30 10:54:44 -07:00
camierjs d6a084322b Use new StroudIntRules for PA simplices tests 2026-04-30 09:34:01 -07:00
camierjs d2a38dd3d4 Merge branch 'jdongg/pa-simplices' of github.com:mfem/mfem into jdongg/pa-simplices 2026-04-30 09:27:34 -07:00
camierjs 17aab43082 Fix PA simplices tests with ir, fix PADiffusionApplyTriangle accumulate 2026-04-30 09:27:11 -07:00
jdongg e5c3cf4ee3 Move Duffy transform from IntegrationRule class to free-standing functions. Create StroudIntegrationRules class for caching Stroud rules. 2026-04-30 07:10:38 -07:00
camierjs b09cc88bab Just avoid the funct_coeff 2026-04-29 19:06:33 -07:00
camierjs 620904124e Init also y in test_pa_simplices 2026-04-29 18:50:03 -07:00
camierjs 6e55cfdc39 Shuffle back bilininteg mass pa simplices 2026-04-29 18:46:16 -07:00
camierjs bef23c5770 Split bilininteg mass pa simplices 2026-04-29 18:38:24 -07:00
camierjs a6fcf162a6 Split bilininteg diffusion pa simplices
and add initial unit tests
2026-04-29 18:20:12 -07:00
John Camier 29caf08098 Merge branch 'master' into cudss-dev 2026-04-29 17:05:01 -07:00
John Camier ba9af41877 Merge branch 'master' into jdongg/pa-simplices 2026-04-29 17:04:48 -07:00
John Camier 2762b9dbfc Merge branch 'master' into bel-sorting 2026-04-29 17:04:40 -07:00
John Camier d75a153df8 Merge branch 'master' into fix-rocm7-hipsparse 2026-04-29 17:04:16 -07:00
John Camier edc4d9a187 Merge branch 'master' into nlconvpa 2026-04-29 17:04:08 -07:00
Mittal, Ketan 0dd81462c0 use forall_switch instead of MFEM_FORALL 2026-04-29 15:29:46 -07:00
Mittal, Ketan 25514d6e8e merge with master and resolve conflicts 2026-04-29 15:02:32 -07:00
Mittal, Ketan 9f01e61a57 add unit test for redistribution of particle data when it is on device 2026-04-29 14:58:30 -07:00
daiyuyang e217864f16 Merge branch 'master' into cudss-dev 2026-04-28 15:17:58 +08:00
yuyangdai de8aacddce Replace enum class with enum 2026-04-28 14:59:09 +08:00
Mittal, Ketan 90fdd7e762 Merge branch 'master' of https://github.com/mfem/mfem into lorentz-device 2026-04-27 14:00:49 -07:00
Mittal, Ketan cfa6594977 minor 2026-04-27 14:00:47 -07:00
John Camier b170c6ae54 Merge branch 'master' into jdongg/pa-simplices 2026-04-27 08:12:29 -07:00
John Camier c4b4ad3224 Merge branch 'master' into bel-sorting 2026-04-27 08:11:59 -07:00
John Camier cecd75aff6 Merge branch 'master' into fix-rocm7-hipsparse 2026-04-27 08:11:53 -07:00
camierjs 2277decd8c With style 2026-04-25 14:24:05 -07:00
camierjs 36f6ff983a VectorConvectionNLFAddMultGradPA3D fallback checks, fix copilot reviews and Win32 math defines 2026-04-25 14:23:22 -07:00
camierjs 6fc6cf9186 Fix Windows compile-time constant expressions 2026-04-25 13:39:08 -07:00
camierjs bb06604dac Avoid narrowing non-constant-expression in initializer list 2026-04-25 13:24:06 -07:00
camierjs 058c6b2dee Avoid documenting NLVC registered kernels 2026-04-25 13:13:51 -07:00
camierjs 94135f3ed2 Merge branch 'camierjs-NLConvPA' into mfem-NLConvPA 2026-04-25 12:56:25 -07:00
camierjs 47c1d6230a Simplify NLVC diagonal kernels 2026-04-25 12:52:57 -07:00
camierjs a22c2c8d72 Cleanup instantiated NLVC registered kernels 2026-04-25 12:21:08 -07:00
camierjs 65f6ade43d Remove low order 3D VectorConvection kernels 2026-04-25 11:29:10 -07:00
camierjs 64cf121310 Meld back MFEM header 2026-04-25 11:07:13 -07:00
camierjs 92e1eace88 Revert test_nl_convection_nd 2026-04-25 11:06:24 -07:00
camierjs e6a3835983 Meld back toward master, rename nlvc unit tests 2026-04-25 11:01:45 -07:00
camierjs d97c8ec672 Cleanup debug traces 2026-04-25 10:33:20 -07:00
camierjs 6d9f34a3d7 Cleanup debug traces, nlvc benchmarks & use transposed adjugate 2026-04-25 08:50:32 -07:00
camierjs 5b73d20291 Cleanup NLF VConv diagonal 2026-04-25 07:15:07 -07:00
camierjs 4febbb7721 Merge branch 'master' into mfem-NLConvPA 2026-04-25 06:08:49 -07:00
camierjs ab81de5bf5 Merge branch 'master' into camierjs-NLConvPA 2026-04-25 05:49:57 -07:00
camierjs 23814cc1fa nlvc diagonal tests & benchmarks 2026-04-24 20:44:59 -07:00
camierjs 6307cef7cb Merge branch 'NLConvPA' of github.com:camierjs/mfem-NLConvPA into camierjs-NLConvPA 2026-04-24 17:51:37 -07:00
camierjs 156f7f930d NVTX marks 2026-04-24 17:51:35 -07:00
camierjs 0d5f21188d Add missing low order specializations 2026-04-24 17:48:28 -07:00
camierjs a786d4f293 SmemPAConvectionNLGradDiagonal 2026-04-24 17:46:27 -07:00
camierjs 018ab7b974 wip SmemPAConvectionNLGradDiagonalPA2D 2026-04-24 14:21:49 -07:00
camierjs f66aaa46bd Merge branch 'camierjs-NLConvPA' into nlconvpa 2026-04-20 14:42:09 -07:00
jdongg 50b197e754 Remove unused variables 2026-04-20 13:05:50 -07:00
jdongg 69dc2b5142 Make style 2026-04-20 12:56:55 -07:00
jdongg fc26f0a773 Switch temporary 1d GJ rules for Stroud construction from heap to stack memory. Clean up diffusion and mass kernels so they all use transposed basis arrays in quad-to-dof operation. Enforce uniform loop index notation across all kernels. 2026-04-20 12:53:00 -07:00
jdongg d160b88706 Guard against 0-size shared memory arrays in simplex diffusion kernels 2026-04-17 19:09:46 -07:00
jdongg 007d5e3e43 Re-add specializations for D1D=1 and D1D=9 since they no longer cause issues 2026-04-17 18:55:58 -07:00
jdongg fd78d48dfb Fix memory leaks (missing virtual destructor for DofToQuad after new derived class); fix race condition in diffusion integrator 2026-04-17 14:38:53 -07:00
Veselin Dobrev f1174bfbf5 Added new methods in class ParFiniteElementSpace: HaveDofSigns and
ApplyDofSigns.

Used the new methods to fix bugs in:
* the ParGridFunction constructor that reads input from a stream
* the method ParGridFunction::SaveAsOne
2026-04-17 09:08:19 -07:00
camierjs c1d1df3d80 Merge branch 'master' into jdongg/pa-simplices 2026-04-17 07:04:10 -07:00
Tom Stitt c0d32918b0 add explicit call to hipsparseSpMV_preprocess to avoid runtime errors in rocsparse with ROCm 7
add checks to all cu/hipsparse calls, which i've confirmed would have cought this

add a static toggle for vendor libsparse usage
2026-04-16 12:36:23 -07:00
Socratis Petrides 03937d0dc2 Fix typo 2026-04-16 12:01:56 -07:00
camierjs eff6bc5abc Probe for ConstantCoefficient first 2026-04-16 10:52:14 -07:00
jdongg 63e24c2a25 Fix documentation of RAGGED_TENSOR mode in DofToQuad. Move Bernstein-specific fields to derived class RaggedDofToQuad. Guard against overwriting Stroud rules in IntegrationRules::Set. 2026-04-15 20:05:06 -07:00
camierjs baf29bff27 Add NVTX marks and fmt::fmt 2026-04-15 16:12:35 -07:00
camierjs 9969e42270 Split LOVectorConvectionNLFAddMultGradPA3DType 2026-04-15 16:02:13 -07:00
camierjs 4936834c5e Split LO/HO VectorConvectionNLF Grad kernels 2026-04-15 15:46:30 -07:00
camierjs da51f42c90 H100x runs 2026-04-15 15:33:31 -07:00
Socratis Petrides c202f9244a more fixes from review comments 2026-04-15 15:07:55 -07:00
camierjs a7dd90466e Register VectorConvectionNLFAddMultPA, benchmarks 2026-04-15 14:35:32 -07:00
camierjs d231431ca7 LOSmemPAConvectionNLGradApply3D 2026-04-15 09:50:20 -07:00
camierjs e55b49b932 Ini kernels LO 2026-04-15 06:52:21 -07:00
Alex LindsayandClaude Sonnet 4.6 76cf9a2d00 Add Doxygen for Mesh::GetFaceElements()
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-14 15:50:11 -07:00
Alex LindsayandClaude Sonnet 4.6 4f796b3708 ReorderElements: sort boundary elements by face index after reordering
After calling ReorderElements the elements[] array is reordered but
boundary[] was left in its original arbitrary order.  be_to_face is
correctly rebuilt by vertex lookup, so individual face lookups remained
correct, but the boundary element ordering was inconsistent with the new
volume element ordering.

Sort boundary[] and be_to_face[] by face index (be_to_face[i]) after
rebuilding the face tables.  Face indices are assigned by
GetElementToFaceTable in the new element order, so sorting by face index
produces the same result as GenerateBoundaryElements would on a mesh
originally stored in the reordered element order.  Because each boundary
face has exactly one boundary element the sort is a strict total order
with no ties, making the output deterministic: two meshes with the same
geometry but different initial numberings produce identical mesh files
after Hilbert reordering.

For 3D meshes bel_to_edge (boundary-element-to-edge table) is also
permuted to stay consistent with the new boundary element ordering.

Add unit tests covering 3D hex, 2D quad, and 3D tet meshes that verify
the adjacent-element indices are non-decreasing across the sorted
boundary element list.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-14 15:32:32 -07:00
camierjs e480c5f37b Register VectorConvectionNLFAddMultGradPA 2026-04-14 13:07:16 -07:00
camierjs 3911f44906 SmemPAConvectionNLGradApply3D and FunctionCoefficient 2026-04-14 11:59:05 -07:00
camierjs 6fa3bc57eb SmemPAConvectionNLGradApply2D running 2026-04-13 22:45:16 -07:00
camierjs 157a1f04f9 wip SmemPAConvectionNLGradApply2D 2nd part 2026-04-13 22:20:31 -07:00
camierjs 7bc531ba39 wip SmemPAConvectionNLGradApply2D 2026-04-13 21:47:37 -07:00
camierjs 9343b54c89 Added SmemPAConvectionNLApply2D 2026-04-13 18:28:46 -07:00
camierjs 0ec3e1d21a wip SmemPAConvectionNLApply3D new kernels 2026-04-13 17:12:57 -07:00
camierjs 4aa44a9b39 Merge remote-tracking branch 'refs/remotes/origin/NLConvPA' into NLConvPA 2026-04-13 13:20:34 -07:00
camierjs d191d332f8 Merge remote-tracking branch 'refs/remotes/origin/NLConvPA' into NLConvPA 2026-04-13 13:19:06 -07:00
camierjs 9dd104c211 fix fmt_FOUND 2026-04-13 13:18:33 -07:00
camierjs e62d26a450 wip Q_adj 2026-04-13 13:18:10 -07:00
camierjs 7ee86d6e75 wip SmemPAConvectionNLGradApply2D 2026-04-13 12:48:49 -07:00
camierjs 3d0878ded5 Setup PA NLConv tests 2026-04-12 15:24:11 -07:00
Veselin Dobrev 65acd08d38 Small tweak in ParMesh::Print 2026-04-10 19:50:10 -07:00
Veselin Dobrev 0acc85d962 In ParMesh::PrintAsOne() fix the interface attribute to make it different
from all real boundary attributes.
2026-04-10 17:41:26 -07:00
Veselin Dobrev 4e35641c28 Fix GCC warning 2026-04-10 12:57:48 -07:00
Veselin Dobrev 8594867ab6 Modify ParMesh::PrintAsOne() to use the settings of SetPrintShared() and
SetPrintInterfaces().

Added some suggestions/questions as FIXME comments.
2026-04-10 12:21:02 -07:00
John Camier 6e0fbbd3bf Merge branch 'master' into cudss-dev 2026-04-09 06:43:58 -07:00
yuyangdai a2b34ab650 Add conditional compilation for cudss_solver in ex1p 2026-04-08 15:05:50 +08:00
yuyangdai f478f687ff Update cuDSS solver integration:
- Use the full name of default communication library and threading library.
- Check the `CUDSS_COMM_LIB` and `CUDSS_THREADING_LIB` in environment first.
- Add the `cudss-solver` option in ex1p
2026-04-08 13:48:56 +08:00
daiyuyangandWill Pazner 0bf998510f Update config/defaults.mk
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2026-04-08 09:45:40 +08:00
Socratis Petrides 2399e5e294 refactored trace tests to avoid code dublication 2026-04-06 18:53:36 -07:00
Socratis Petrides 205b694162 Merge branch 'master' into trace-p-ref 2026-04-06 18:30:43 -07:00
Socratis Petrides f2b661ca05 more fixes 2026-04-06 18:30:15 -07:00
Socratis Petrides c8ef971f03 review fixes 2026-04-06 18:11:06 -07:00
Socratis Petrides 10aa1133b2 review fixes 2026-04-06 13:22:59 -07:00
John Camier 05c3a2c83f Merge branch 'master' into jdongg/pa-simplices 2026-04-04 07:11:19 -07:00
yuyangdai d916299a49 Update INSTALL documentation for CUDSS requirements and backend support 2026-04-03 14:33:59 +08:00
yuyangdai 10840ac6b4 Refactor CuDSSSolver initialization. 2026-04-03 13:56:20 +08:00
jdongg d194a26542 Fix memory leak 2026-04-02 06:12:50 -07:00
Justin Dong a3a368dfd9 Make style 2026-04-02 05:26:08 -07:00
jdongg e8f39487b9 Fix doxygen formatting 2026-04-02 05:08:15 -07:00
jdongg ec6e2a7c74 Address PR reviews, pt. 1: removed unused lines, improve documentation, remove individual IntegrationRule objects for Stroud components, remove IsBernsteinSimplexSpace function from fespace and combine with UsesRaggedTensorBasis, ensuring checks for entire -mesh geoemtries are done, introduce MAX_Q1D_SIMPLEX and
MAX_D1D_SIMPLEX.
2026-04-02 04:51:31 -07:00
yuyangdai b437016f6e Refactor cuDSS library path resolution in defaults.mk for improved handling of missing libraries 2026-04-02 15:01:00 +08:00
yuyangdai 5170bbe010 Fix CUDSS library checks and improve status output in CMake and makefile 2026-04-01 16:16:01 +08:00
yuyangdai f8fcdf6a97 Update cuDSS configuration and library paths in CMake and makefiles; refactor cuDSS solver integration.
- Add `MFEM_CUDSS_COMM_LIB` for OpenMPI communication library
- Add 'MFEM_CUDSS_THREADING_LIB' for threading library
- Add SetMatrixCuDSS() to set matrix values for cudss
- Remove SetMatrixSortRow()
- Remove unused options and simplify conditions in ex1.cpp and ex1p.cpp.
2026-04-01 13:49:02 +08:00
John Camier d10c908b38 Merge branch 'master' into cudss-dev 2026-03-29 17:51:58 -07:00
John Camier cb1f54ed2b Merge branch 'master' into jdongg/pa-simplices 2026-03-29 17:31:29 -07:00
John Camier 89f85cee21 Merge branch 'master' into cudss-dev 2026-03-26 13:49:44 -07:00
camierjs 00da00b93a Add Copyright headers 2026-03-24 08:42:34 -07:00
camierjs 4eca673111 Cosmetic trailing spaces 2026-03-24 08:33:06 -07:00
John Camier 4507a02249 Merge branch 'master' into cudss-dev 2026-03-24 08:01:00 -07:00
John Camier 9a18da5aaa Merge branch 'master' into jdongg/pa-simplices 2026-03-24 07:58:41 -07:00
camierjs 4e31111827 Add BP'7' for simplices 2026-03-20 17:50:05 -07:00
camierjs 0cbf5b5562 CEED bench GLL simplices BP5 fix 2026-03-20 16:39:12 -07:00
camierjs d81bbc1214 BP5 hex/tet 2026-03-20 16:34:25 -07:00
camierjs 41255f308e Fix Benchmark internal namespace 2026-03-20 14:44:22 -07:00
camierjs 96be0cafdf Add benchmark tests with simplices 2026-03-20 08:30:23 -07:00
camierjs 55fc2f806c Merge branch 'master' into jdongg/pa-simplices 2026-03-20 05:59:14 -07:00
Ketan Mittal 08bf7f991b Merge branch 'master' into lorentz-device 2026-03-17 19:01:33 -07:00
Tzanio Kolev 61c7187a86 Review comments 2026-03-12 11:17:14 -07:00
Tzanio KolevandCopilot 4fbefc6987 Update mesh/pmesh.cpp
Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
2026-03-11 12:20:13 -07:00
Tzanio Kolev 8dd75d2548 typo 2026-03-11 11:32:06 -07:00
Tzanio Kolev b0c2ec505f Added optional output of material interfaces in parallel 2026-03-11 11:20:59 -07:00
Ketan Mittal 0503cbd41c Merge branch 'master' into lorentz-device 2026-03-11 10:57:58 -07:00
Ketan Mittal 74d9671ec8 Merge branch 'master' into lorentz-device 2026-03-07 16:14:45 -08:00
Ketan Mittal 9917fa8306 Merge branch 'master' into tmop-mesh-validity 2026-03-06 10:12:51 -08:00
Mittal, Ketan 74f64934e8 Merge branch 'master' of https://github.com/mfem/mfem into lorentz-device 2026-03-05 11:10:05 -08:00
Mittal, Ketan 8ff5affe45 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-03-05 11:09:59 -08:00
Mittal, Ketan 2e7a6d745c cosmetic 2026-03-05 11:09:50 -08:00
camierjs eb95b46fad Missing Stroud flags 2026-03-05 09:42:14 -08:00
camierjs a0778c759e StroudFlag to bool 2026-03-05 09:04:05 -08:00
camierjs a922c1f2c3 Merge remote-tracking branch 'origin/catch-tests' into jdongg/pa-simplices 2026-03-05 08:31:35 -08:00
camierjs 17600f59f2 make style 2026-03-05 07:02:26 -08:00
camierjs 6cfec88f47 Merge branch 'master' into jdongg/pa-simplices 2026-03-05 07:01:30 -08:00
jdongg f11009d674 Switch 'or' to || for windows tests 2026-03-05 02:02:00 -08:00
jdongg 5df384f6ef Remove unused variables 2026-03-05 01:33:54 -08:00
jdongg daa555b68d Change pullback of Stroud rule to reference cube in AssemblePA routines to avoid assigning pointer to integration rule to a const integration rule 2026-03-05 00:59:36 -08:00
jdongg 76cfafe700 Switch 2D mass and diffusion kernels on triangles to stack memory 2026-03-04 18:51:54 -08:00
Socratis Petrides 58976c6f41 add relaxation in the smoother to fix non-PD issue of the preconditioner 2026-03-03 22:18:50 -08:00
Chris Vogl aa4f1bc8e4 Removed conditional and combined setting of initial GSLIB flags (from @nmnobre) 2026-02-27 12:03:06 -08:00
Chris VoglandNuno Nobre fcc4b2dade Switch to using OPTFLAGS for METIS fetching (from @nmnobre)
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-26 15:31:16 -08:00
Chris VoglandNuno Nobre 3c137f36cb Improved consistency across TPL fetching (from @nmnobre)
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-26 15:27:10 -08:00
Chris VoglandNuno Nobre 9211b97eb0 Fixed typo in GSLIB_FLAGS name (from @nmnobre)
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-26 15:21:57 -08:00
Ketan Mittal 9cb3c1e1a6 Merge branch 'master' into lorentz-device 2026-02-26 13:02:01 -08:00
Mittal, Ketan 0f449eb906 Merge branch 'lorentz-device' of https://github.com/mfem/mfem into lorentz-device 2026-02-23 15:53:08 -08:00
Mittal, Ketan cdca79060b print when data is on device 2026-02-23 15:52:56 -08:00
camierjs 0114739e5e Fix more warnings 2026-02-21 18:14:23 -08:00
camierjs 708b642be2 Merge branch 'master' into jdongg/pa-simplices 2026-02-21 13:00:45 -08:00
camierjs d78357494d Fix some other warnings 2026-02-21 11:59:33 -08:00
camierjs 9398b1a6e0 Fix warnings 2026-02-21 11:52:23 -08:00
camierjs b23a087f42 Test for fec in IsBernsteinSimplexSpace 2026-02-21 11:27:58 -08:00
Ketan Mittal 3505a5a354 Merge branch 'master' into lorentz-device 2026-02-20 09:31:38 -08:00
Ketan Mittal 72203ebb5c Merge branch 'master' into tmop-mesh-validity 2026-02-17 14:24:27 -08:00
Mittal, Ketan d62ffe149a style 2026-02-17 08:49:04 -08:00
Tzanio Kolev 86b2ad3a61 Merge branch 'master' into jdongg/pa-simplices 2026-02-17 08:29:10 -08:00
Socratis Petrides 35ba876ecf CI fix 2026-02-14 15:15:31 -08:00
Socratis Petrides 2886c4b211 pass ess_bdr_marker in pmg 2026-02-13 17:11:10 -08:00
Socratis Petrides 9455ed83b2 add ess_tdof treatement in pmg 2026-02-13 17:10:44 -08:00
Mittal, Ketan cc46bfdfcc Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-02-13 13:03:04 -08:00
Mittal, Ketan 010e96b6f0 remove shadow variable 2026-02-13 13:02:52 -08:00
Ketan Mittal edf551c40a Merge branch 'master' into tmop-mesh-validity 2026-02-13 09:47:32 -08:00
Mittal, Ketan fd36e1b177 Merge branch 'findpts-device-data-movement' of https://github.com/mfem/mfem into lorentz-device 2026-02-11 09:41:08 -08:00
Mittal, Ketan d390f9a7d1 remove timers 2026-02-11 09:41:02 -08:00
Mittal, Ketan 128f650dd3 documentation 2026-02-11 08:52:31 -08:00
Mittal, Ketan 2bbb317831 minor 2026-02-10 16:35:22 -08:00
Mittal, Ketan 36835e62e0 remove unused variable 2026-02-10 16:25:01 -08:00
Mittal, Ketan 70867ab87b minor 2026-02-10 16:04:18 -08:00
Mittal, Ketan 4b59c90b08 merge and resolve conflicts 2026-02-10 15:56:43 -08:00
Ketan Mittal b538e9d34c Merge branch 'master' into tmop-mesh-validity 2026-02-10 15:48:42 -08:00
Mittal, Ketan 79733572d7 initial commit 2026-02-10 09:23:02 -08:00
Mittal, Ketan 6b5d1e55de Merge branch 'findpts-device-data-movement' of https://github.com/mfem/mfem into lorentz-device 2026-02-10 09:19:07 -08:00
Mittal, Ketan 6b5eb2f92f minor 2026-02-10 09:17:06 -08:00
Mittal, Ketan 49c346233e minor clean up 2026-02-10 09:13:57 -08:00
Socratis Petrides 505661eb5f minor 2026-02-09 19:25:36 -08:00
Mittal, Ketan e81352715e Merge branch 'findpts-device-data-movement' of https://github.com/mfem/mfem 2026-02-09 13:32:17 -08:00
Ketan Mittal 159f1873d6 Merge branch 'master' into particle-device 2026-02-09 13:26:35 -08:00
Mittal, Ketan fa07a503dd minor 2026-02-09 13:26:08 -08:00
Mittal, Ketan d65fcc5d8c minor comment and split line 2026-02-09 09:56:14 -08:00
Mittal, Ketan 1696197f54 merge with master 2026-02-09 09:41:52 -08:00
Mittal, Ketan 19089ac132 merge on host instead of device 2026-02-09 09:37:31 -08:00
yuyangdai a3e229ef82 Make SetMatrix methods private and remove Init method
- SetMatrix methods are called inside SetOperator
- Init method has been removed as it is unnecessary
2026-02-09 16:40:39 +08:00
camierjs 5e9d32a0f7 Fix CMake PA simplices test, add visualization 2026-02-07 14:11:09 -08:00
camierjs 724866141a make style, remove unused-variable & shadows 2026-02-07 13:54:12 -08:00
John Camier f17c9eef12 Merge branch 'master' into jdongg/pa-simplices 2026-02-07 13:29:43 -08:00
John Camier 00b6dcdd37 Merge branch 'master' into cudss-dev 2026-02-07 13:21:22 -08:00
Socratis Petrides 69c1d0d15b changelog 2026-02-07 12:52:47 -08:00
Socratis Petrides 5f7d138aef style 2026-02-06 18:18:19 -08:00
Tzanio Kolev 4659602e63 Merge branch 'master' into jdongg/pa-simplices 2026-02-06 07:11:45 -08:00
jdongg 1771a26426 Updated new function documentation and CHANGELOG 2026-02-05 15:04:39 -08:00
jdongg b42a383201 Modified Stroud rules to output nodes in the simplex, with InverseDuffyTrans routine added to pull back to reference cube when necessary 2026-02-05 13:23:18 -08:00
Tzanio Kolev 19f444489f Merge branch 'master' into cudss-dev 2026-02-05 11:03:38 -08:00
Socratis Petrides deae7e4997 minor 2026-02-05 10:33:10 -08:00
psocratis 6142015168 refactoring 2026-02-04 22:46:27 -08:00
psocratis 305a7f1e02 added pmg and cpmg options to dpg miniapps 2026-02-04 18:40:51 -08:00
psocratis 0bff07026c additional pmg and complex pmg options added including coarse solve option 2026-02-04 18:40:17 -08:00
psocratis 32c3e5dfd8 add complexoperator-->complexhypre 2026-02-04 18:39:05 -08:00
psocratis ee78679277 add blockoperator-->monolythic 2026-02-04 18:38:08 -08:00
jdongg 185f8b099d Small bug fixes to 2D mass and diffusion simplex kernels 2026-02-04 16:48:07 -08:00
psocratis 89a0f18b30 small leak 2026-02-04 11:14:18 -08:00
Socratis Petrides 4cbc345ba6 fix makefile and cmakelists.txt 2026-02-03 21:58:34 -08:00
Socratis Petrides abe3843712 typo 2026-02-03 21:47:04 -08:00
Socratis Petrides 3c2d1e4814 shadow var fix 2026-02-03 21:11:13 -08:00
Socratis Petrides a8c4ed3c79 expand p-MG to the complex case 2026-02-03 18:51:04 -08:00
Socratis Petrides 0aa73fa285 simplify precond construction in pmax 2026-02-03 15:12:01 -08:00
Socratis Petrides f1652b5ba0 add convenient functions for solver construction 2026-02-03 12:35:58 -08:00
Socratis Petrides fed0baf6b6 fix leak in complex case 2026-02-03 12:34:23 -08:00
Socratis Petrides 49a8ee5b54 fix possible leak in case of static_cond and add method to retrun trace_fes 2026-02-03 12:32:40 -08:00
Socratis Petrides c9569a0629 shadow var fix 2026-02-02 23:50:36 -08:00
Socratis Petrides a4e83b5836 p-mg cleanup 2026-02-02 22:30:02 -08:00
Socratis Petrides 8b285047bd some helper function for constructor order of fecols + Clone impl 2026-02-02 22:29:34 -08:00
psocratis f999be0372 valgrind fixes for p-mg 2026-02-01 18:12:32 -08:00
psocratis e639cfd75b style 2026-02-01 15:49:29 -08:00
psocratis 2753692303 Merge branch 'master' into trace-p-ref 2026-02-01 15:42:07 -08:00
psocratis 2722e979fc fix mg-transpose issue 2026-02-01 15:41:26 -08:00
psocratis 144c0ce106 CI 2026-02-01 14:49:58 -08:00
psocratis b64172e12e remove no longer needed code 2026-02-01 13:27:06 -08:00
psocratis c0072fcb84 fix windows CI issue 2026-02-01 13:02:18 -08:00
psocratis 37cd34a0e1 fix shadow var 2026-01-30 22:50:44 -08:00
psocratis 005be84af4 valgrind fixes 2026-01-30 22:44:48 -08:00
Socratis Petrides 0ef682c156 fixing dim for AMSvsADS in smoother 2026-01-30 17:03:26 -08:00
Socratis Petrides 3faa838774 expanding unit tests to serial/parallel true transfer 2026-01-30 17:02:53 -08:00
Socratis Petrides f10c2793c9 adding NC case in serial 2026-01-30 17:02:18 -08:00
Socratis Petrides 9f4b83362a p-ref MG for dpg diffusion 2026-01-28 00:02:13 -08:00
Socratis Petrides 8ef423ca65 traces p-ref example for p-diffusion 2026-01-28 00:01:00 -08:00
Socratis Petrides aff5173656 fix CI 2026-01-26 22:45:47 -08:00
Socratis Petrides 2164f07f01 starte p-mg for dpg-diffusion 2026-01-26 22:08:38 -08:00
Mittal, Ketan 2b3657c76b Merge branch 'master' of https://github.com/mfem/mfem into particle-device 2026-01-26 09:32:18 -08:00
Mittal, Ketan 63d45eb194 initial commit 2026-01-26 09:32:05 -08:00
Socratis Petrides fc0037da31 minor 2026-01-25 22:06:19 -08:00
Socratis Petrides 795a1a29a3 minor print fix 2026-01-25 00:11:44 -08:00
Socratis Petrides 476d9c5111 fix serial case 2026-01-25 00:06:18 -08:00
Socratis Petrides 64625f7333 tests for parallel trace/assemble prefoperator 2026-01-24 18:12:49 -08:00
Socratis Petrides 4ae8c1bc2b started on p-multigrid in serial 2026-01-24 18:12:11 -08:00
Socratis Petrides 4ae97f1834 hpp file changes for pref assembly 2026-01-24 18:11:37 -08:00
Socratis Petrides 816e8e4bea add option to get the preftransfer operator in parallel as a hypreparmatrix 2026-01-24 18:11:12 -08:00
Socratis Petrides c14d630f14 tet ND case fixed for assembled Transfer P 2026-01-24 00:03:09 -08:00
Socratis Petrides 899433f79f added PrerinementTransfer SparseMatrix. Still need to address the ND tet case wrt to doftrans 2026-01-23 18:54:42 -08:00
Ketan Mittal ff9294e5b0 Merge branch 'master' into tmop-mesh-validity 2026-01-23 15:41:31 -08:00
Socratis Petrides 16db883427 style 2026-01-22 22:11:46 -08:00
Socratis Petrides 1834268091 complete pref-tests 2026-01-22 22:11:25 -08:00
Socratis Petrides 947e25f769 clean up of ProjectTrace methods 2026-01-22 22:10:59 -08:00
Socratis Petrides 9fbe90527a fix trace pref prolongation 2026-01-22 22:09:35 -08:00
daiyuyang 10c637837c Add dependency check for CUDSS in MFEMConfig.cmake.in 2026-01-22 16:56:17 +08:00
jdongg 076f1450e5 Cleanup junk files 2026-01-22 00:46:41 -08:00
Socratis Petrides bfa672c644 error computation 2026-01-22 00:24:33 -08:00
Justin Dong 30ac5d8d38 Fixed style issues 2026-01-21 20:35:36 -08:00
jdongg 5a2b05cc35 Remove unused variables, cleanup after rebasing. 2026-01-21 20:00:49 -08:00
Socratis Petrides afe388a573 style 2026-01-21 19:35:22 -08:00
Socratis Petrides 50a9cce9f4 first tests on project 2026-01-21 19:35:03 -08:00
Socratis Petrides 8ee2fdbfd9 Project coeff for trace/skeleton 2026-01-21 19:34:40 -08:00
Socratis Petrides d6b4594150 Pref mat-free for trace space 2026-01-21 19:34:06 -08:00
jdongg 771e263db8 Rebase onto main 2026-01-21 18:10:45 -08:00
jdongg ee13fec158 Added shared memory implementation of diffusion and mass integrators with partial assembly 2026-01-21 17:43:41 -08:00
daiyuyangandAndrew Ho 665ba30f65 Update linalg/cudss.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:38:05 +08:00
Justin Dong 0162a6727d Optimized ragged tensor nested for loops by collapsing to 1d loops with appropriate forward and inverse maps to recover nested indices 2026-01-21 17:36:37 -08:00
daiyuyangandAndrew Ho d6de2c1a1d Update examples/ex1p.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:35:32 +08:00
daiyuyangandAndrew Ho c4b4cfdc2f Update examples/ex1p.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:35:18 +08:00
daiyuyangandAndrew Ho 3435475ae0 Update examples/ex1.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:34:32 +08:00
Justin Dong 31fae9d195 Adding support for partial assembly of tetrahedrons for mass and diffusion integrators with Bernstein basis 2026-01-21 17:34:25 -08:00
Justin Dong 6ce5fe8fac Initial implementation of partial assembly on triangles for Bernstein basis
Rebase onto master
2026-01-21 17:33:07 -08:00
daiyuyangandAndrew Ho e43b58fa02 Update config/cmake/modules/FindCUDSS.cmake
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:32:27 +08:00
jdongg 1739cebfdb Added shared memory implementation of diffusion and mass integrators with partial assembly 2026-01-21 17:03:17 -08:00
John Camier 27c8412439 Merge branch 'master' into cudss-dev 2026-01-21 08:23:31 -08:00
daiyuyang 5aaef22cc0 Add conditional support for cuDSS solver in ex1.cpp 2026-01-19 15:16:31 +08:00
daiyuyang f9df36a6de Update parameter names in documentation for clarity in cudss.hpp 2026-01-19 14:21:19 +08:00
daiyuyang b3a75a9295 Add cuDSS solver option in ex1.cpp 2026-01-19 13:42:00 +08:00
daiyuyang b8c90d24f1 Make MPI optional and add OpenMP support in CuDSSSolver
- Make the MPI optional: the CuDSSSolver now does not need MPI necessary
- Add the OpenMP supports by cudssSetThreadingLayer() API
2026-01-19 13:40:56 +08:00
Mittal, Ketan 9e5b93a532 initial commit with working prototype 2026-01-15 18:30:23 -08:00
Will Pazner 5cdc7499ca Merge remote-tracking branch 'origin/master' into elast-fix-4404 2025-12-29 13:00:52 -08:00
Mittal, Ketan a7ca72c59f update get jacobian function 2025-12-22 12:24:15 -08:00
Mittal, Ketan 8267c7ea64 Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-22 12:18:20 -08:00
Mittal, Ketan d2cf8fef73 remove unneeded function 2025-12-22 12:18:17 -08:00
Mittal, Ketan 2df3fc8ddb remove unneeded function 2025-12-22 12:16:05 -08:00
daiyuyang a8eba594d2 Update linalg/cudss.hpp and linalg/cudss.cpp; Revert general/device.cpp
- Move ` CUDA_REAL_T` to the cpp file.
- Remove `CuDSSHandle` singleton and replace it with a static cudssHandle_t variable.
- Delete `CuDSSHandle::Init()` from ex1p
2025-12-17 09:52:36 +08:00
yuyangdai 52291cadbf Revert code docs; add constructor comment; update the .gitignore 2025-12-15 13:09:45 +08:00
yuyangdai 1865b430b0 Revert makefile and example/CMakeLists.txt; delete examples/cudss 2025-12-15 11:01:42 +08:00
yuyangdai d8734b4b18 Update the examples/ex1p.cpp
- Add the global cudss handle before using the cudss solver.
2025-12-15 10:27:47 +08:00
yuyangdai eae1fa217b Add CuDSSHandle singleton and update CuDSSSolver
- Remove unused variable `myid`.
- Move `MFEM_CUDSS_CHECK` and `mfem_cudss_error` into linalg/cudss.cpp.
- Disable move copy constructor and move assignment.
- Introduce `CuDSSHandle` singleton to manage cuDSS handle lifetime.
- Rename `InitHandle` to `Init`
2025-12-15 10:27:22 +08:00
Mittal, Ketan 59ee0d93c6 incorporate bounds from other branch 2025-12-14 16:57:34 -08:00
Mittal, Ketan d8164fd158 wip: check if function positive 2025-12-14 16:40:44 -08:00
Mittal, Ketan 23ce5f08e7 Merge branch 'plbound-extremum' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-13 17:37:42 -08:00
Mittal, Ketan e059253549 Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-13 17:35:13 -08:00
gengyan.zgy 7dc8b0d9fe Fix CuDSSSolver::ArrayMult() for single RHS 2025-12-11 14:32:20 +08:00
yuyangdai 3f236b406f Update examples/ex1p.cpp
- Add cudss solver into the ex1p.cpp
2025-12-11 13:55:16 +08:00
yuyangdai 47dcdd8159 Update config/cmake/modules/FindCUDSS.cmake and examples/cudss/CMakeLists.txt
- Add a newline at end of file
2025-12-11 13:53:49 +08:00
yuyangdai d93d36d6d3 Update linalg/cudss.cpp and linalg/cudss.hpp
- Define a cuDSS error check macro, MFEM_CUDSS_CHECK(x);
- Rename the method from InitRhsSol to SetNumRHS;
- Utilize CuMemAlloc/ CuMemcpyDtoD instead of cudaMalloc/cudaMemcpy;
- Update MPI_Comm usage.
2025-12-11 13:52:57 +08:00
kmittal2 c89807da1e use in TMOP solver 2025-12-10 20:20:17 -08:00
kmittal2 53de3bde2c Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-06 18:08:33 -08:00
kmittal2 56414350e2 initial commit - function to extract Jacobian determinant gridfunction in mesh/pmesh 2025-12-06 18:07:57 -08:00
yuyangdai 6bcad07a5c Update the n_global variable name for global number of rows 2025-12-05 13:50:36 +08:00
yuyangdai d56493c2c5 Refactored enum classes MatType and MatViewType, updated variable names, and update some comments. 2025-12-05 11:01:29 +08:00
daiyuyang ec67fe536e Merge branch 'master' into cudss-dev 2025-12-02 14:04:17 +08:00
daiyuyang eff793d6fa Merge branch 'master' into cudss-dev 2025-11-28 13:53:59 +08:00
Chris Vogl e5111df7c6 rename fetched GSLIB option variable to be consistent with fetched METIS 2025-11-24 16:20:53 -08:00
Chris Vogl 4fe80816d2 added fPIC flag option for fetched METIS, also includes optimization flags as done with fetched GSLIB 2025-11-24 16:20:16 -08:00
Chris Vogl 0839f915b4 added position independent code option for fetched hypre 2025-11-24 16:19:10 -08:00
Chris Vogl 23b4a5a5b8 style cleanup of GSLIB fetching to match other fetching code 2025-11-24 16:17:42 -08:00
yuyangdai e5e7280a17 Ignore the solution files of cudss sample run 2025-11-24 15:18:53 +08:00
yuyangdai 773dc57712 Using the make style to format the class CuDSSSolver and example/cudss/ex1p codes 2025-11-24 15:16:48 +08:00
yuyangdai 98a043b195 Update the doc files for CuDSSSolver 2025-11-24 15:15:22 +08:00
yuyangdai c2a3c83099 Update the makefiles for CuDSSSolver 2025-11-24 15:14:51 +08:00
daiyuyang 3ea90aff27 Add cudss solver 2025-11-24 15:10:45 +08:00
gengyan.zgy 23a362d6ad Update cmake files for cuDSS 2025-11-24 15:10:30 +08:00
Justin Dong 41d15a3a0a Optimized ragged tensor nested for loops by collapsing to 1d loops with appropriate forward and inverse maps to recover nested indices 2025-07-22 12:13:15 -07:00
Justin Dong daca70bf80 Adding support for partial assembly of tetrahedrons for mass and diffusion integrators with Bernstein basis 2025-07-02 11:28:39 -07:00
Justin Dong 0e02aa947a Initial implementation of partial assembly on triangles for Bernstein basis 2025-06-30 21:22:25 -07:00
Tzanio Kolev 44ca1fbf8f minor 2024-07-18 09:36:14 -07:00
Tzanio Kolev 27b84d79b8 Verify that ordering is byVDIM when using the AMG elasticity solver 2024-07-18 09:30:52 -07:00
187 changed files with 15413 additions and 5692 deletions
@@ -94,6 +94,16 @@ inputs:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
# Unfortunately, "uses:" fields cannot have references to variables like
# ${{env.MFEM_ACTIONS_VERSION}}, so the branch/tag name has to be hard coded.
# Therefore, in the future, when updating the version of the
# mfem/github-actions to use, we'll have to replace:
# - all definitions of MFEM_ACTIONS_VERSION and
# - all "uses:" fields that refer to mfem/github-actions.
MFEM_ACTIONS_VERSION:
description: Version (branch or tag) of the mfem/github-actions to use.
default: v2.7
runs:
using: 'composite'
steps:
@@ -118,6 +128,7 @@ runs:
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
echo MFEM_ACTIONS_VERSION=${{inputs.MFEM_ACTIONS_VERSION}} >> $GITHUB_ENV
shell: bash
- name: Env (dir)
+1 -1
View File
@@ -53,7 +53,7 @@ runs:
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- uses: mfem/github-actions/build-mfem@v2.5
- uses: mfem/github-actions/build-mfem@v2.7
if: ${{steps.debug.outputs.cache-hit != 'true'}}
env:
CXXFLAGS: ${{env.CXXFLAGS}}
+6
View File
@@ -12,6 +12,11 @@
name: 'Install MPI'
description: 'Installs MPI and set up its environment variables'
inputs:
NO_FLAGS:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
runs:
using: 'composite'
steps:
@@ -27,6 +32,7 @@ runs:
shell: bash
- name: Env (bis)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
+2 -2
View File
@@ -37,14 +37,14 @@ runs:
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{env.MFEM_ACTIONS_VERSION}}
- uses: actions/cache/restore@v5 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
- name: Hypre/Metis links
if: ${{inputs.par == 'true'}}
+42
View File
@@ -0,0 +1,42 @@
# MFEM Pull Request Review Agent Guide
## Purpose and scope
Review MFEM PRs for correctness, maintainability, performance, portability, test coverage, and MFEM consistency. Use the diff and PR context; reference source files, tests, and CI results when available. Follow `CONTRIBUTING.md`, especially Developer Guidelines, PR rules, checklist, and testing.
## Critical review pillars
- Correctness and numerical behavior
- API and user-facing impact
- Performance implications
- Maintainability and portability
## Review workflow
1. Read the PR description, linked issues, and intended behavior.
2. Inspect the diff before commenting.
3. Identify affected MFEM components, examples, tests, build or docs changes, and downstream APIs.
4. Analyze the code against the critical review pillars.
5. Compare the change against nearby code and MFEM patterns; flag unmotivated deviations.
6. Check whether tests and documentation were updated appropriately.
7. Review CI results and suggest actions.
8. Produce a structured review with prioritized findings.
9. Always limit conclusions to available evidence.
## MFEM-specific review checklist
- Component-aware scope: identify the touched subsystem (FEM, solvers, preconditioners, linear algebra, mesh, examples, miniapps, build, or docs) and assess its impact against the review pillars.
- Numerical and algorithmic behavior: assess issues in convergence, stability, tolerances, precision, iteration limits, and failure handling. If clear opportunities exist to improve the algorithmic approach, call them out with expected impact.
- API and user-facing impact: assess backward compatibility, user-visible behavior and default changes, migration impact, deprecations, and whether documentation clearly explains user-facing API changes.
- Data structure and memory semantics: assess ownership, lifetime, aliasing, container behavior, and device-host synchronization.
- Parallel and serial behavior: assess whether the change preserves equivalent semantics in serial and parallel modes where applicable; if logic is currently mode-specific, check whether extension to the other mode is straightforward (clear abstractions, no hard-wired assumptions), document constraints, and call out expected behavior differences explicitly.
- Backend and portability impact: assess likely cross-backend risks in CPU, CUDA, HIP, OCCA, RAJA, partial assembly, fallback paths, compiler compatibility, and platform assumptions.
- Build, dependency, and configuration impact: assess CMake or make changes, optional dependency behavior, and feature-flag interactions.
- Tests and docs alignment: check available regression or unit coverage evidence for changed behavior, and ensure docs are updated for new flags, APIs, options, or behavior changes.
- MFEM developer-guideline fit: keep code lean, simple, general, logically separated, and portable; suggest C++17 improvements when they clearly improve safety, clarity, or maintainability.
- New source files, examples, or miniapps: if a PR adds source/header files, verify they are properly wired into the relevant `makefile` and `CMakeLists.txt`, referenced in docs where applicable (including `doc/CodeDocumentation.dox`), and added to top-level `.gitignore` only when generated artifacts require it.
- Changelog: verify `CHANGELOG` is updated if the PR introduces significant new features or user-facing changes.
- MFEM conventions: use `real_t`; use `mfem::out`/`mfem::err` instead of `std::cout`/`std::cerr` in library code; flag large/binary files; if AI assistance is apparent but undisclosed, suggest following `CONTRIBUTING.md`.
- Edge cases: if the PR touches complex or error-prone areas, suggest additional tests for edge cases, failure modes, and parallel behavior.
## Commenting guidelines
- Keep comments concise, actionable, and grounded in the diff.
- Focus on correctness, behavior changes, and user impact over style nits.
- Be professional, concise, collaborative, technically precise, and avoid unsupported assumptions.
+3 -7
View File
@@ -13,7 +13,7 @@ Note that some of these scripts use the shared MFEM GitHub Actions from the exte
<https://github.com/mfem/github-actions>
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch (or tag) in the above from which the action is taken.
The current CI workflows are:
@@ -29,16 +29,12 @@ Runs a number of static repository-level sanity checks.
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
## `mfem-analysis.yml` (`build-analysis`)
Checks if the code builds and satisfies minimal requirements.
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
## `builds-and-tests.yml`
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
One matrix job runs `tests/scripts/gitignore` after `make test-noclean` to check generated artifacts against `.gitignore`.
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
- `mfem/github-actions/build-hypre`
+25 -23
View File
@@ -40,6 +40,7 @@ env:
METIS_ARCHIVE_MAC: metis-4.0.3-mac.tgz
METIS_TOP_DIR: metis-4.0.3
MFEM_TOP_DIR: mfem
MFEM_ACTIONS_VERSION: v2.7
# Note for future improvements:
#
@@ -110,6 +111,7 @@ jobs:
build-system: make
hypre-target: int64
precision: fp64
gitignore-check: YES
- os: ubuntu-latest
target: opt
codecov: NO
@@ -170,20 +172,6 @@ jobs:
env
shell: bash
# For info on Xcode see:
# - https://github.com/actions/runner-images/issues/12541
# - https://github.com/actions/runner-images/blob/releases/macos-15-arm64/20250811/images/macos/macos-15-arm64-Readme.md#xcode
- name: Xcode version setup (MacOS)
if: matrix.os == 'macos-latest'
run: |
XCODE_PATH="/Applications/Xcode_16.4.app"
echo "> sudo xcode-select -s ${XCODE_PATH}"
sudo xcode-select -s ${XCODE_PATH}
echo "> g++ -v"
g++ -v
echo "> clang++ -v"
clang++ -v
# Only get MPI if defined for the job.
# TODO: It would be nice to have only one step, e.g. with a dedicated
# action, but I (@adrienbernede) don't see how at the moment.
@@ -228,11 +216,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-${{ env.MFEM_ACTIONS_VERSION }}
- name: get hypre
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.5
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -242,7 +230,7 @@ jobs:
- name: get hypre (Windows)
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.5
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -258,11 +246,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
- name: install metis
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
uses: mfem/github-actions/build-metis@v2.7
with:
archive: ${{ matrix.os != 'macos-latest' && env.METIS_ARCHIVE || env.METIS_ARCHIVE_MAC }}
dir: ${{ env.METIS_TOP_DIR }}
@@ -304,7 +292,7 @@ jobs:
# MFEM build and test
- name: build
uses: mfem/github-actions/build-mfem@v2.5
uses: mfem/github-actions/build-mfem@v2.7
env:
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
with:
@@ -330,7 +318,13 @@ jobs:
- name: tests
if: matrix.build-system == 'make' && (matrix.target == 'opt' || matrix.os == 'ubuntu-latest')
run: |
cd ${{ env.MFEM_TOP_DIR }} && make test
cd ${{ env.MFEM_TOP_DIR }}
if [[ "${{ matrix.gitignore-check }}" == "YES" ]]; then
make test-noclean
else
make test
fi
shell: bash
- name: cmake checks
if: matrix.build-system == 'cmake' && matrix.target == 'dbg'
@@ -375,8 +369,16 @@ jobs:
# Code coverage (process and upload reports)
- name: codecov
if: matrix.codecov == 'YES'
uses: mfem/github-actions/upload-coverage@v2.5
uses: mfem/github-actions/upload-coverage@v2.7
with:
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
project_dir: ${{ env.MFEM_TOP_DIR }}
directories: "fem general linalg mesh"
env:
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
- name: gitignore
if: matrix.gitignore-check == 'YES'
run: |
cd ${{ env.MFEM_TOP_DIR }}/tests/scripts
./runtest gitignore
+10
View File
@@ -14,9 +14,19 @@ name: "Static Analysis"
on:
push:
branches: ["master", "next"]
paths-ignore: &docs-only-paths
- "**/*.md"
- "doc/**"
- ".binder/**"
- "CITATION.cff"
- "LICENSE"
- "NOTICE"
- "CHANGELOG"
- "INSTALL"
pull_request:
# The branches below must be a subset of the branches above
branches: ["master"]
paths-ignore: *docs-only-paths
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
-100
View File
@@ -1,100 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
name: "Build Analysis"
permissions:
actions: write
on:
push:
branches:
- master
- next
pull_request:
workflow_dispatch:
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
env:
HYPRE_ARCHIVE: v2.19.0.tar.gz
HYPRE_TOP_DIR: hypre-2.19.0
METIS_ARCHIVE: metis-4.0.3.tar.gz
METIS_TOP_DIR: metis-4.0.3
COVERAGE_ENV: mfem-coverage
jobs:
gitignore:
runs-on: ubuntu-latest
steps:
- name: checkout MFEM
uses: actions/checkout@v6
with:
path: mfem
- name: Get MPI (Linux)
run: |
sudo apt-get install openmpi-bin libopenmpi-dev
export OMPI_MCA_rmaps_base_oversubscribe=1
- name: Cache Hypre Install
id: hypre-cache
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
- name: Get Hypre
if: steps.hypre-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
target: int32
- name: Cache Metis Install
id: metis-cache
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
- name: Install Metis
if: steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{ env.METIS_ARCHIVE }}
dir: ${{ env.METIS_TOP_DIR }}
# MFEM build and test
- name: build-mfem
uses: mfem/github-actions/build-mfem@v2.5
with:
os: ${{ runner.os }}
target: opt
codecov: NO
mpi: par
build-system: make
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
metis-dir: ${{ env.METIS_TOP_DIR }}
mfem-dir: mfem
- name: test (no clean)
run: |
cd mfem && make test-noclean
- name: gitignore
run: |
cd mfem/tests/scripts
./runtest gitignore
+6 -2
View File
@@ -19,18 +19,22 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.5
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{env.HYPRE_TGZ}}
dir: ${{env.HYPRE_DIR}}
+6 -2
View File
@@ -19,18 +19,22 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
uses: mfem/github-actions/build-metis@v2.7
with:
archive: ${{env.METIS_TGZ}}
dir: ${{env.METIS_DIR}}
+10
View File
@@ -17,7 +17,17 @@ permissions:
on:
push:
branches: ["master", "next"]
paths-ignore: &docs-only-paths
- "**/*.md"
- "doc/**"
- ".binder/**"
- "CITATION.cff"
- "LICENSE"
- "NOTICE"
- "CHANGELOG"
- "INSTALL"
pull_request:
paths-ignore: *docs-only-paths
workflow_dispatch:
concurrency:
+73 -16
View File
@@ -11,38 +11,95 @@
Version 4.9.1 (development)
===========================
- Policy for AI-assisted contribution added to CONTRIBUTING.md
- Added policy for AI-assisted contribution to CONTRIBUTING.md.
Discretization improvements
---------------------------
- Extend FindPointsGSLIB to support surface meshes.
- Improved FindPointsGSLIB surface mesh capability with support for simplices
and an option to specify axis-aligned bounding box padding for near-surface
point queries.
- Replaced legacy simplex quadrature rules with symmetric positive-weight
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
rules guarantee all-positive weights and interior quadrature points,
improving numerical stability. Higher orders fall back to Grundmann-Moller.
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
2015.
Tet rules (d=1-13): Witherden & Vincent (ibid).
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
2022.
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
integrators.
- Improved the gridfunction projection routines. Projections work for Scalar,
- Replaced legacy simplex quadrature rules with symmetric positive weight rules
for triangles (orders 0-25) and tetrahedra (orders 0-20). These rules
guarantee all-positive weights and interior quadrature points, improving
numerical stability. Higher orders fall back to Grundmann-Moller.
* Triangle rules: Witherden and Vincent, DOI: 10.1016/j.camwa.2015.03.017
* Tet rules (d=1-13): Witherden and Vincent (same as above)
* Tet rules (d=14-20): Chuluunbaatar et al., DOI: 10.1016/j.camwa.2022.08.016
- Added support for general 1D Gauss-Jacobi quadrature rules and Stroud conical
quadrature rules on triangles and tetrahedra.
- Improved the GridFunction projection routines. Projections work for Scalar,
Vector and VectorFE, also NURBS versions. Optionally different types of
projections can be selected, default behaviour has not changed.
projections can be selected, default behavior has not changed.
- Added methods to estimate function extremum using piecewise linear bounds +
- Added GridFunction projection methods for trace spaces, i.e., project
coefficients on the mesh skeleton.
- Added methods to estimate function extremum using piecewise linear bounds plus
recursive subdivision.
- Extend FindPointsGSLIB to support surface meshes.
Meshing improvements
--------------------
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
bounds on the determinant of the mesh transformation Jacobian.
- Added PA support for TMOP's adaptive limiting functionality. Multiple
GridFunctions and Coefficients can be combined to form a composite term.
- Improved support for 1D NURBS meshes with variable order, including using
the patches construct for 1D NURBS meshes.
- Added the option to include material interfaces (faces separating elements
with different element attributes) as additional boundary elements, for
parallel visualization, e.g. with GLVis. This is supported by both the Print
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
Linear and nonlinear solvers
----------------------------
- Added support for trace spaces in PRefinementTransferOperator. This is used in
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
DPG miniapps).
GPU computing
-------------
- Added PA gradient and diagonal support for VectorConvectionNLFIntegrator
(AssembleGradPA, AddMultGradPA, AssembleGradDiagonalPA).
- Improved partial assembly for VectorDivergenceIntegrator with shared-memory
kernels, kernel registration, and transpose support.
- Improved partial-assembly diagonal kernels for VectorMassIntegrator (shared-
memory specializations) and ElasticityIntegrator (no scratch Q-vector).
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
New and updated examples and miniapps
-------------------------------------
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
capability.
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
leverage the ParticleSet capability.
- Added (Complex)PRefinementMultigrid solver option in the DPG miniapps.
Miscellaneous
-------------
- Fixed signed DOF handling in ParGridFunction reading (read constructor) and
saving via SaveAsOne(). Simplified the process of applying the DOF signs by
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
method will return immediately if no sign flips are needed.
Version 4.9, released on Dec 11, 2025
+10 -1
View File
@@ -433,6 +433,15 @@ if (MFEM_USE_STRUMPACK)
endif()
endif()
# cuDSS can only be enabled in CUDA
if (MFEM_USE_CUDSS)
if (MFEM_USE_CUDA)
find_package(CUDSS REQUIRED)
else()
message(FATAL_ERROR " *** cuDSS requires that CUDA be enabled.")
endif()
endif()
# GnuTLS
if (MFEM_USE_GNUTLS)
find_package(_GnuTLS REQUIRED)
@@ -631,7 +640,7 @@ find_package(Threads REQUIRED)
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CUDSS CALIPER CODIPACK
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
ALGOIM ENZYME CUDA::cudart)
+64 -65
View File
@@ -3,12 +3,12 @@
</p>
<p align="center">
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
</p>
@@ -84,7 +84,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
follow the [MFEM PR Rules](#mfem-pr-rules).
- When your contribution is fully working and ready to be reviewed, add
the `ready-for-review` label.
- PRs are treated similarly to journal submission with an "editor" assigning two
- PRs are treated similarly to journal submission, with an "editor" assigning two
reviewers to evaluate the changes.
- The reviewers have 3 weeks to evaluate the PR and work with the author to
fix issues and implement improvements.
@@ -125,7 +125,7 @@ The MFEM source code has the following structure:
│ ├── petsc
│ ├── pumi
│ ├── sundials
| └── superlu
└── superlu
├── fem
│ ├── ceed
│ ├── dfem
@@ -137,10 +137,6 @@ The MFEM source code has the following structure:
│ ├── moonolith
│ ├── qinterp
│ └── tmop
│ | ├── assemble
│ | ├── metrics
│ | ├── mult
│ | └── tools
├── general
├── linalg
│ ├── batched
@@ -153,11 +149,10 @@ The MFEM source code has the following structure:
│ ├── common
│ ├── contact
│ ├── dfem
│ ├── diag-smoothers
│ ├── dpg
│ ├── electromagnetics
│ ├── fluids
│ │ ├── navier
│ │ └── schrodinger-flow
│ ├── gslib
│ ├── hdiv-linear-solver
│ ├── hooke
@@ -167,6 +162,7 @@ The MFEM source code has the following structure:
│ ├── nurbs
│ ├── parelag
│ ├── performance
│ ├── plasma
│ ├── shifted
│ ├── solvers
│ ├── spde
@@ -197,15 +193,15 @@ respectively.
- The main finite element classes are:
+ [`FiniteElement`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElementCollection.html)
+ [`FiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1FiniteElementSpace.html)
+ [`GridFunction`](https://docs.mfem.org/html/classmfem_1_1GridFunction.html)
+ [`BilinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
- The main linear algebra classes and sources are
+ [`Operator`](https://docs.mfem.org/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1Vector.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`DenseMatrix`](https://docs.mfem.org/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://docs.mfem.org/html/classmfem_1_1SparseMatrix.html)
+ Sparse [smoothers](https://docs.mfem.org/html/sparsesmoothers_8hpp.html) and linear [solvers](https://docs.mfem.org/html/solvers_8hpp.html)
@@ -217,8 +213,8 @@ shared geometric entities between different tasks. The parallel source files
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
- The main parallel classes are
+ [`ParMesh`](https://docs.mfem.org/html/solvers_8hpp.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParNCMesh.html)
+ [`ParFiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1ParFiniteElementSpace.html)
+ [`ParGridFunction`](https://docs.mfem.org/html/classmfem_1_1ParGridFunction.html)
+ [`ParBilinearForm`](https://docs.mfem.org/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://docs.mfem.org/html/classmfem_1_1ParLinearForm.html)
@@ -228,14 +224,14 @@ have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
#### GPU and general device support
GPU and multi-core CPU support is based on device kernels supporting different
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
backends (CUDA, HIP, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
device/host memory manager.
- The main device-relevant classes and sources are:
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html), [`hip.hpp`](https://docs.mfem.org/html/hip_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
#### Utilities, building and documentation
- The `general/` directory contains C++ classes that serve as utilities for
@@ -249,8 +245,8 @@ device/host memory manager.
- `examples` and `miniapps` respectively gather simple and more fully-featured
demonstrations of the usage on MFEM. They both rely on `data/` for the
collection of meshes.
- The `tests/` directory contains a unit test suite and will later contain more
tests that run example codes.
- The `tests/` directory contains a unit test suite, additional tests, and
benchmarks.
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
website.
@@ -284,8 +280,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
the top of https://github.com/mfem.
- Consider making your membership public by going to https://github.com/orgs/mfem/people
and clicking on the organization visibility drop box next to your name.
- Project discussions and announcements will be posted at
https://github.com/orgs/mfem/teams/everyone.
- Project discussions and announcements will be posted at https://github.com/orgs/mfem/discussions,
tagging the `@mfem/everyone` team when appropriate.
#### Structure
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
@@ -345,11 +341,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Well-designed simple code is frequently more general and powerful.
- Lean code base is easier to understand by new collaborators.
- New features should be added only if they are necessary or generally useful.
- Introduction of language constructions not currently used in MFEM should be
- Introduction of language constructs not currently used in MFEM should be
justified and generally avoided (to maintain portability to various systems
and compilers, including early access hardware).
- We prefer basic C++ and the C++03 standard, to keep the code readable by
a large audience and to make sure it compiles anywhere.
- We prefer basic C++. Use C++17 features judiciously, prioritizing readability,
consistency with existing MFEM code, and portability to different systems,
compilers and device backends.
- *Keep the code general and reasonably efficient*
- The main goal is fast prototyping for research and application development.
@@ -392,7 +389,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- When your branch is ready for other developers to review / comment on
the code, create a pull request towards `mfem:master`.
- Pull request typically have titles like:
- Pull requests typically have titles like:
`Description [new-feature-dev]`
@@ -413,12 +410,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
team will add reviewers as appropriate.
- List outstanding TODO items in the description, see PR #222 for an example.
- List outstanding TODO items in the description.
- When your contribution is fully working and ready to be reviewed, add
the `ready-for-review` label.
or request the `ready-for-review` label.
- PRs are treated similarly to journal submission with an "editor" assigning
- PRs are treated similarly to journal submission, with an "editor" assigning
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
the PR and work with the author to implement improvements and fix issues.
@@ -444,7 +441,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
checks in GitHub Actions enforce MFEM-specific rules which are explained in
the error messages and the `tests/scripts` directory.
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
- Also note that the tests `branch-history` and `repo-check` found in GitHub
Actions can be triggered automatically before each push using git hooks. See
the [git hooks README](config/githooks/README.md) for a detailed explanation.
@@ -501,15 +498,15 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
3. To ensure the quality of the PR by making sure that the code adheres to the [Developer Guidelines](#developer-guidelines), e.g. all methods, data members, and functions have documentation, including data ownership and lifetime, new examples/miniapps have a corresponding PR in mfem/web, major features have `CHANGELOG` entries, etc.
3. To seek help from the editors in case of difficulties.
4. To seek help from the editors in case of difficulties.
4. To complete the review in a timely manner: 3 weeks from assignment.
5. To complete the review in a timely manner: 3 weeks from assignment.
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
6. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
7. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
8. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
#### Responsibilities of Authors
@@ -535,30 +532,30 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Code builds.
- [ ] Code passes `make style`.
- [ ] Update `CHANGELOG`:
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Update `INSTALL`:
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Has a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] Update `.gitignore`:
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] New examples:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
- [ ] Add any files generated by it to the `clean` target.
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new examples directory (e.g. `examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
@@ -575,13 +572,13 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] New capability:
- [ ] All new public, protected, and private classes, methods, data members, and functions have full Doxygen-style documentation in source comments. Documentation should include descriptions of member data, function arguments and return values, template parameters, and prerequisites for calling new functions.
- [ ] Pointer arguments and return values must specify whether ownership is being transferred or lent with the call.
@@ -683,7 +680,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
- [ ] Update URL shortlinks:
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
- [ ] Add the new shortlinks to the MFEM package in `spack`.
- [ ] Update website in `mfem/web` repo:
- Update version and shortlinks in `src/index.md` and `src/download.md`.
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
@@ -735,22 +732,24 @@ commit or push, see the [README](config/githooks/README.md) in the `config/githo
directory.
### Linux and Mac smoke tests
### GitHub Actions smoke tests
We use GitHub Actions to drive the default tests on the `master` and `next`
branches. See the `.github/workflows` files and the logs at
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
Testing using GitHub Actions should be kept lightweight, as there is a time
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
GitHub Actions testing should be kept lightweight, as there is a time
constraint on jobs. The current workflows cover Linux, macOS, and Windows
configurations.
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
- Tests on the `next` branch are currently scheduled to run each night.
### Additional Windows smoke test
### Windows smoke test
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor` file and the
build logs at
We also use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor.yml` file
and the build logs at
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
CMake is used to generate the MSVC Project files and drive the build. A release
+31 -16
View File
@@ -38,14 +38,13 @@ the option MFEM_USE_METIS.
MFEM also includes support for devices such as GPUs, and programming models such
as CUDA, HIP, OCCA, OpenMP and RAJA.
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
a newer compiler, e.g. GCC version 4.9 or higher.
- Starting with version 4.9, MFEM requires a C++17 compiler.
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
https://developer.nvidia.com/cuda-toolkit
- HIP support requires an AMD GPU and an installation of the ROCm software stack
https://rocmdocs.amd.com
https://rocm.docs.amd.com
- OCCA support requires the OCCA library
https://libocca.org
@@ -83,9 +82,9 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
make parallel -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
CUDA build:
make cuda -j 4
@@ -115,14 +114,14 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
make -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
Parallel build with fetching of hypre and METIS:
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
make -j 4
@@ -134,7 +133,8 @@ CUDA build:
HIP build:
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 \
-DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
make -j 4
Example codes (serial/parallel, depending on the build):
@@ -269,6 +269,7 @@ Compilers:
CXX - C++ compiler, serial build
MPICXX - MPI C++ compiler, parallel build
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
HIP_CXX - The HIP compiler, e.g. 'hipcc'
Compiler options:
OPTIM_FLAGS - Options for optimized build
@@ -395,6 +396,11 @@ MFEM_USE_STRUMPACK = YES/NO
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_CUDSS = YES/NO
Enable MFEM functionality based on the cuDSS library. When using cuDSS, CUDA
support must be also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
When enabled, this option uses the CUDSS_* library options, see below.
MFEM_USE_GINKGO = YES/NO
Enable MFEM functionality based on the Ginkgo library, which provides
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
@@ -554,13 +560,13 @@ MFEM_USE_RAJA = YES/NO
MFEM_USE_OCCA = YES/NO
Enables support for the OCCA library in MFEM. OCCA is an open-source library
which aims to make it easy to program different types of devices (e.g. CPU,
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
GPU, FPGA) by providing a unified API for interacting with JIT-compiled
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_GSLIB = YES/NO
Enables MFEM functionality based on the GSLIB library, and specifically its
FindPoints component, which provides a robust algorithms to evaluate finite
FindPoints component, which provides robust algorithms to evaluate finite
element functions in a collection of points in physical space. When enabled,
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
@@ -719,9 +725,18 @@ The specific libraries and their options are:
Options: STRUMPACK_OPT, STRUMPACK_LIB.
Versions: STRUMPACK >= 3.0.0.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
C++ compiler that supports the C++-17 standard. For additional requirements
and dependencies of specific modules, see the Ginkgo webpage below.
- CUDSS (optional), used when MFEM_USE_CUDSS = YES. Note that CUDSS requires
CUDA 12.x toolkit and the cuDSS libraries. The supported communication backend
is OpenMPI 4.x (default), and OpenMPI 4.x or a later version must be pre-built.
The source files in the cuDSS tarball provide guidance for developing custom
MPI implementations.
URL: https://developer.nvidia.com/cudss
https://docs.nvidia.com/cuda/cudss/advanced_features.html#communication-layer-library-in-cudss
Options: CUDSS_OPT, CUDSS_LIB.
Versions: cuDSS >= 0.6.0.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Ginkgo may have additional
requirements and module-specific dependencies; see the webpage below.
URL: https://ginkgo-project.github.io
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or
Debug).
@@ -793,7 +808,7 @@ The specific libraries and their options are:
Options: CONDUIT_OPT, CONDUIT_LIB.
Versions: Conduit >= 0.3.1.
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
- ADIOS2 (optional), used when MFEM_USE_ADIOS2 = YES.
URL: https://adios2.readthedocs.io/
Versions: ADIOS >= 2.5.0.
@@ -869,7 +884,7 @@ The specific libraries and their options are:
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
Versions: RAJA >= 2022.10.3.
- Moonolith (optional), use when MFEM_USE_MOONOLITH = YES.
- Moonolith (optional), used when MFEM_USE_MOONOLITH = YES.
URL: https://bitbucket.org/zulianp/par_moonolith
Options: MOONOLITH_DIR
Versions: MOONOLITH >= 1.1.0.
@@ -957,7 +972,7 @@ CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
To use a specific generator use the "-G <generator>" option of cmake:
cmake <mfem-source-dir> -G "Xcode"
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
cmake <mfem-source-dir> -G "Visual Studio 17 2022"
cmake <mfem-source-dir> -G "MinGW Makefiles"
With CMake it is possible to build MFEM as a shared library using the standard
@@ -1202,7 +1217,7 @@ larger problems, there are two options:
Specific options for HIP
========================
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
ROCm install, as well as having `$ROCM_PATH/bin` in `PATH`.
Specific options for RAJA+HIP+MPI
=================================
+5
View File
@@ -35,6 +35,7 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_CUDSS @MFEM_USE_CUDSS@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
@@ -109,6 +110,10 @@ if (MFEM_USE_RAJA)
find_dependency(RAJA)
endif()
if (MFEM_USE_CUDSS)
find_dependency(cudss)
endif (MFEM_USE_CUDSS)
if (MFEM_USE_UMPIRE)
find_dependency(umpire)
endif()
+9
View File
@@ -108,6 +108,15 @@
// Enable MFEM functionality based on the STRUMPACK library.
#cmakedefine MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
#cmakedefine MFEM_USE_CUDSS
// CUDSS communication layer library path
#cmakedefine MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
#cmakedefine MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable functionality based on the Ginkgo library.
#cmakedefine MFEM_USE_GINKGO
+68
View File
@@ -0,0 +1,68 @@
if (NOT cudss_DIR AND CUDSS_DIR)
set(cudss_DIR ${CUDSS_DIR}/lib/cmake/cudss)
endif()
message(STATUS "Looking for CUDSS ...")
message(STATUS " in CUDSS_DIR = ${CUDSS_DIR}")
message(STATUS " cudss_DIR = ${cudss_DIR}")
find_package(cudss)
set(CUDSS_FOUND ${cudss_FOUND})
set(CUDSS_LIBRARIES "cudss")
if (CUDSS_FOUND)
message(STATUS
"Found CUDSS target: ${CUDSS_LIBRARIES} (version: ${cudss_VERSION})")
else()
set(msg STATUS)
if (CUDSS_FIND_REQUIRED)
set(msg FATAL_ERROR)
endif()
message(${msg}
"CUDSS not found. Please set CUDSS_DIR to the install prefix.")
endif()
if(CUDSS_FOUND AND TARGET cudss)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION)
if(NOT CUDSS_LIBRARY_LOCATION)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION_RELEASE)
endif()
if(CUDSS_LIBRARY_LOCATION)
get_filename_component(CUDSS_LIBRARY_DIR "${CUDSS_LIBRARY_LOCATION}" DIRECTORY)
else()
message(WARNING "Could not determine the location of the cuDSS library.")
endif()
else()
message(WARNING "cuDSS target not available; cannot determine library directory.")
endif()
# Set the full name of the cuDSS threading library if OpenMP is enabled.
# The threading layer library (libcudss_mtlayer_gomp.so) is located under the
# cuDSS library directory by default.
if (MFEM_USE_OPENMP)
find_file(
CUDSS_THREADING_LIB
NAMES libcudss_mtlayer_gomp.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_THREADING_LIB AND CUDSS_THREADING_LIB)
set(MFEM_CUDSS_THREADING_LIB "${CUDSS_THREADING_LIB}")
endif()
message(STATUS "CUDSS threading layer library: ${MFEM_CUDSS_THREADING_LIB}")
endif()
# Set the full name of the cuDSS communication library if MFEM use OpenMPI.
# The communication layer library (libcudss_commlayer_mpi.so) is located under the
# cuDSS library directory by default.
# The communication layer library is used pre-built communication layers for OpenMPI
# by default.
if (MFEM_USE_MPI)
find_file(
CUDSS_COMM_LIB
NAMES libcudss_commlayer_openmpi.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_COMM_LIB AND CUDSS_COMM_LIB)
set(MFEM_CUDSS_COMM_LIB "${CUDSS_COMM_LIB}")
endif()
message(STATUS "CUDSS communication layer library: ${MFEM_CUDSS_COMM_LIB}")
endif()
+8 -10
View File
@@ -18,19 +18,17 @@
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
enable_language(C)
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(GSLIB_FETCH_VERSION 1.0.9)
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
if (CMAKE_C_FLAGS)
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
endif()
if (BUILD_SHARED_LIBS)
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
endif()
add_library(GSLIB STATIC IMPORTED)
# set options (technically flags because GSLIB does not use cmake)
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(GSLIB_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
if (BUILD_SHARED_LIBS)
set(GSLIB_FLAGS "${GSLIB_FLAGS} -fPIC")
endif()
# define external project and create future include directory so it is present
# to pass CMake checks at end of MFEM configuration step
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_FLAGS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
include(ExternalProject)
ExternalProject_Add(gslib
@@ -40,7 +38,7 @@ if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND ""
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS=${GSLIB_FLAGS}"
INSTALL_COMMAND "")
file(MAKE_DIRECTORY ${PREFIX}/include)
# set imported library target properties
+3 -1
View File
@@ -44,6 +44,9 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
# set options and associated dependencies
set(HYPRE_CMAKE_OPTIONS "")
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
if (BUILD_SHARED_LIBS)
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_POSITION_INDEPENDENT_CODE:BOOL=ON)
endif()
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
get_cmake_property(all_vars VARIABLES)
foreach(var ${all_vars})
@@ -95,7 +98,6 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
SOURCE_SUBDIR src
PREFIX ${HYPRE_INSTALL}
BUILD_COMMAND ${CMAKE_COMMAND} --build . -- -j${CMAKE_BUILD_PARALLEL_LEVEL}
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${HYPRE_INSTALL} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${HYPRE_CMAKE_OPTIONS})
file(MAKE_DIRECTORY ${HYPRE_INSTALL}/include)
# set imported library target properties
+10 -2
View File
@@ -19,10 +19,18 @@
# - METIS_VERSION_5 (cache variable)
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
enable_language(C)
set(METIS_FETCH_VERSION 4.0.3)
add_library(METIS STATIC IMPORTED)
# set options (technically flags because METIS does not use cmake)
set(METIS_FLAGS "-Wno-implicit-int -Wno-incompatible-pointer-types")
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(METIS_FLAGS "${METIS_FLAGS} ${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
if (BUILD_SHARED_LIBS)
set(METIS_FLAGS "${METIS_FLAGS} -fPIC")
endif()
# define external project
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with default options")
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with ${METIS_FLAGS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/metis)
include(ExternalProject)
ExternalProject_Add(metis
@@ -32,7 +40,7 @@ if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND tar -xzf ../metis/metis-${METIS_FETCH_VERSION}-mac.tgz --strip=1
BUILD_COMMAND $(MAKE) COPTIONS=-Wno-incompatible-pointer-types
BUILD_COMMAND $(MAKE) clean && $(MAKE) "OPTFLAGS=${METIS_FLAGS}"
INSTALL_COMMAND mkdir -p ${PREFIX}/lib && cp libmetis.a ${PREFIX}/lib/)
# set imported library target properties
add_dependencies(METIS metis)
+9 -9
View File
@@ -22,15 +22,15 @@ include(MfemCmakeUtilities)
mfem_find_package(SuiteSparse SuiteSparse SuiteSparse_DIR "" "" "" ""
"Paths to headers required by SuiteSparse."
"Libraries required by SuiteSparse."
ADD_COMPONENT "UMFPACK" "include;suitesparse" umfpack.h "lib" umfpack
ADD_COMPONENT "KLU" "include;suitesparse" klu.h "lib" klu
ADD_COMPONENT "AMD" "include;suitesparse" amd.h "lib" amd
ADD_COMPONENT "BTF" "include;suitesparse" btf.h "lib" btf
ADD_COMPONENT "CHOLMOD" "include;suitesparse" cholmod.h "lib" cholmod
ADD_COMPONENT "COLAMD" "include;suitesparse" colamd.h "lib" colamd
ADD_COMPONENT "CAMD" "include;suitesparse" camd.h "lib" camd
ADD_COMPONENT "CCOLAMD" "include;suitesparse" ccolamd.h "lib" ccolamd
ADD_COMPONENT "config" "include;suitesparse" SuiteSparse_config.h "lib"
ADD_COMPONENT "UMFPACK" "include;include/suitesparse;suitesparse" umfpack.h "lib" umfpack
ADD_COMPONENT "KLU" "include;include/suitesparse;suitesparse" klu.h "lib" klu
ADD_COMPONENT "AMD" "include;include/suitesparse;suitesparse" amd.h "lib" amd
ADD_COMPONENT "BTF" "include;include/suitesparse;suitesparse" btf.h "lib" btf
ADD_COMPONENT "CHOLMOD" "include;include/suitesparse;suitesparse" cholmod.h "lib" cholmod
ADD_COMPONENT "COLAMD" "include;include/suitesparse;suitesparse" colamd.h "lib" colamd
ADD_COMPONENT "CAMD" "include;include/suitesparse;suitesparse" camd.h "lib" camd
ADD_COMPONENT "CCOLAMD" "include;include/suitesparse;suitesparse" ccolamd.h "lib" ccolamd
ADD_COMPONENT "config" "include;include/suitesparse;suitesparse" SuiteSparse_config.h "lib"
suitesparseconfig)
if (SuiteSparse_FOUND AND METIS_VERSION_5)
+6
View File
@@ -157,4 +157,10 @@ constexpr real_t operator""_r(unsigned long long v)
#endif
#endif // MFEM_USE_MPI not defined
#ifndef MFEM_USE_CUDA
#ifdef MFEM_USE_CUDSS
#error Building with cuDSS (MFEM_USE_CUDSS=YES) requires CUDA (MFEM_USE_CUDA=YES)
#endif
#endif // MFEM_USE_CUDSS not defined
#endif // MFEM_CONFIG_HPP
+9
View File
@@ -108,6 +108,15 @@
// Enable MFEM functionality based on the STRUMPACK library.
// #define MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
// #define MFEM_USE_CUDSS
// CUDSS communication layer library path
// #define MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
// #define MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable MFEM features based on the Ginkgo library.
// #define MFEM_USE_GINKGO
+3
View File
@@ -36,6 +36,9 @@ MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_SUPERLU5 = @MFEM_USE_SUPERLU5@
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_CUDSS = @MFEM_USE_CUDSS@
MFEM_CUDSS_COMM_LIB = @MFEM_CUDSS_COMM_LIB@
MFEM_CUDSS_THREADING_LIB = @MFEM_CUDSS_THREADING_LIB@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_AMGX = @MFEM_USE_AMGX@
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
+1
View File
@@ -38,6 +38,7 @@ option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_SUPERLU5 "Use the old SuperLU_DIST 5.1 version" OFF)
option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_CUDSS "Enable cuDSS usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
+15 -1
View File
@@ -153,6 +153,7 @@ MFEM_USE_SUPERLU = NO
MFEM_USE_SUPERLU5 = NO
MFEM_USE_MUMPS = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_CUDSS = NO
MFEM_USE_GINKGO = NO
MFEM_USE_AMGX = NO
MFEM_USE_MAGMA = NO
@@ -368,6 +369,19 @@ STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
$(SCOTCH_LIB) $(SCALAPACK_LIB)
# CUDSS library configuration
CUDSS_DIR = @MFEM_DIR@/../cudss
CUDSS_INCLUDE_DIR = $(CUDSS_DIR)/include
CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
CUDSS_LIB = \
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
# The cuDSS communication and threading libraries.
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR),$(CUDSS_LIBRARY_DIR)/libcudss_mtlayer_gomp.so))))
# Ginkgo library configuration
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
GINKGO_SEARCH_DIR = $(subst @MFEM_DIR@,$(MFEM_DIR),$(GINKGO_DIR))
@@ -621,7 +635,7 @@ PARELAG_LIB = -L$(PARELAG_DIR)/build/src -lParELAG
AXOM_DIR = @MFEM_DIR@/../axom
TRIBOL_DIR = @MFEM_DIR@/../tribol
TRIBOL_OPT = -I$(TRIBOL_DIR)/include -I$(AXOM_DIR)/include
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -ltribol_shared -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
-laxom_slam -laxom_slic -laxom_core
# Enzyme configuration
+36 -23
View File
@@ -50,6 +50,10 @@
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cpu
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
//
// Device simplices sample runs:
// ex1 -pa -d gpu -m ../data/inline-tet.mesh
// ex1 -pa -d gpu -m ../data/inline-tri.mesh
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
@@ -138,25 +142,25 @@ int main(int argc, char *argv[])
}
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
// Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
FiniteElementCollection *fec;
bool delete_fec;
auto basis_type = (pa && mesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
fec = new H1_FECollection(order, dim, basis_type);
}
else if (mesh.GetNodes())
{
fec = mesh.GetNodes()->OwnFEC();
delete_fec = false;
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
fec = new H1_FECollection(order = 1, dim, basis_type);
}
FiniteElementSpace fespace(&mesh, fec);
cout << "Number of finite element unknowns: "
@@ -224,17 +228,29 @@ int main(int argc, char *argv[])
// 11. Solve the linear system A X = B.
if (!pa)
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#ifdef MFEM_USE_CUDSS
if (Device::Allows(Backend::CUDA_MASK))
{
// Use cuDSS to solve the system.
CuDSSSolver cudss_solver;
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
}
}
else
{
@@ -273,17 +289,14 @@ int main(int argc, char *argv[])
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << x << flush;
}
// 15. Free the used memory.
if (delete_fec)
{
delete fec;
}
if (order > 0) { delete fec; }
return 0;
}
+61 -35
View File
@@ -42,7 +42,11 @@
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/square-mixed.mesh
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/fichera-mixed.mesh
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
// mpirun -np 4 ex1p -pa -d ceed-cpu -m ../data/beam-tet.mesh
//
// Device simplices sample runs:
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tet.mesh
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tri.mesh
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
@@ -83,6 +87,9 @@ int main(int argc, char *argv[])
const char *device_config = "cpu";
bool visualization = true;
bool algebraic_ceed = false;
#ifdef MFEM_USE_CUDSS
bool cudss_solver = false;
#endif
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -102,6 +109,10 @@ int main(int argc, char *argv[])
args.AddOption(&algebraic_ceed, "-a", "--algebraic",
"-no-a", "--no-algebraic",
"Use algebraic Ceed solver");
#endif
#ifdef MFEM_USE_CUDSS
args.AddOption(&cudss_solver, "-cudss", "--cudss-solver", "-no-cudss",
"--no-cudss-solver", "Use the cuDSS Solver.");
#endif
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
@@ -158,19 +169,20 @@ int main(int argc, char *argv[])
}
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
// use continuous Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
FiniteElementCollection *fec;
bool delete_fec;
auto basis_type = (pa && pmesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
fec = new H1_FECollection(order, dim, basis_type);
}
else if (pmesh.GetNodes())
{
fec = pmesh.GetNodes()->OwnFEC();
delete_fec = false;
if (myid == 0)
{
cout << "Using isoparametric FEs: " << fec->Name() << endl;
@@ -178,8 +190,7 @@ int main(int argc, char *argv[])
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
fec = new H1_FECollection(order = 1, dim, basis_type);
}
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_BigInt size = fespace.GlobalTrueVSize();
@@ -248,33 +259,51 @@ int main(int argc, char *argv[])
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use Jacobi smoothing, for now.
Solver *prec = NULL;
if (pa)
#ifdef MFEM_USE_CUDSS
if (!pa && (Device::Allows(Backend::CUDA_MASK) && cudss_solver))
{
if (UsesTensorBasis(fespace))
{
if (algebraic_ceed)
{
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
// Solve using a direct solver with cuDSS
CuDSSSolver cudss_solver(MPI_COMM_WORLD);
cudss_solver.SetMatrixSymType(
CuDSSSolver::SYMMETRIC_POSITIVE_DEFINITE);
cudss_solver.SetMatrixViewType(CuDSSSolver::UPPER);
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
prec = new HypreBoomerAMG;
Solver *prec = NULL;
if (pa)
{
if (UsesTensorBasis(fespace))
{
if (algebraic_ceed)
{
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
}
else
{
prec = new HypreBoomerAMG;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec)
{
cg.SetPreconditioner(*prec);
}
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec) { cg.SetPreconditioner(*prec); }
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
// 14. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
@@ -308,10 +337,7 @@ int main(int argc, char *argv[])
}
// 17. Free the used memory.
if (delete_fec)
{
delete fec;
}
if (order > 0) { delete fec; }
return 0;
}
+9
View File
@@ -95,6 +95,15 @@ int main(int argc, char *argv[])
args.PrintOptions(cout);
}
if (amg_elast && !static_cond && reorder_space)
{
if (myid == 0)
cerr << "\nThe AMG elasticity solver requires ordering byVDIM! "
<< "Ignoring the specified option -nodes/--by-nodes.\n"
<< endl;
reorder_space = false;
}
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
+10 -1
View File
@@ -57,6 +57,8 @@ set(SRCS
integ/lininteg_domain_grad.cpp
integ/lininteg_domain_vectorfe.cpp
integ/nonlininteg_vecconvection_pa.cpp
integ/nonlininteg_vecconvection_pa_diag.cpp
integ/nonlininteg_vecconvection_pa_grad.cpp
integ/nonlininteg_vecconvection_mf.cpp
coefficient.cpp
complex_fem.cpp
@@ -133,7 +135,7 @@ set(SRCS
tmop/assemble/diag2.cpp
tmop/assemble/grad2_limit.cpp
tmop/assemble/grad2.cpp
tmop/assemble/diag3_limit.cpp
tmop/assemble/diag3_limit.cpp
tmop/assemble/diag3.cpp
tmop/assemble/grad3_limit.cpp
tmop/assemble/grad3.cpp
@@ -195,14 +197,20 @@ set(HDRS
integ/bilininteg_dgtrace_kernels.hpp
integ/bilininteg_vecdiffusion_kernels.hpp
integ/bilininteg_convection_kernels.hpp
integ/bilininteg_diffusion_pa_simplices.hpp
integ/bilininteg_diffusion_kernels.hpp
integ/bilininteg_elasticity_kernels.hpp
integ/bilininteg_hcurl_kernels.hpp
integ/bilininteg_hdiv_kernels.hpp
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
integ/bilininteg_mass_pa_simplices.hpp
integ/bilininteg_vecdiffusion_pa.hpp
integ/bilininteg_vecdiv_pa.hpp
integ/bilininteg_vecmass_pa.hpp
integ/nonlininteg_vecconvection_pa.hpp
integ/nonlininteg_vecconvection_pa_diag.hpp
integ/nonlininteg_vecconvection_pa_grad.hpp
coefficient.hpp
complex_fem.hpp
convergence.hpp
@@ -309,6 +317,7 @@ set(HDRS
tmop_tools.hpp
tmop_amr.hpp
gslib.hpp
gslib/gslib_kernel_helpers.hpp
transfer.hpp
hyperbolic.hpp
integrator.hpp
+22 -4
View File
@@ -1345,7 +1345,8 @@ real_t DiffusionIntegrator::ComputeFluxEnergy
}
const IntegrationRule &DiffusionIntegrator::GetRule(
const FiniteElement &trial_fe, const FiniteElement &test_fe)
const FiniteElement &trial_fe, const FiniteElement &test_fe,
const bool stroud)
{
int order;
if (trial_fe.Space() == FunctionSpace::Pk)
@@ -1362,7 +1363,15 @@ const IntegrationRule &DiffusionIntegrator::GetRule(
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
}
MassIntegrator::MassIntegrator(const IntegrationRule *ir)
@@ -1449,7 +1458,8 @@ void MassIntegrator::AssembleElementMatrix2(
const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans)
const ElementTransformation &Trans,
const bool stroud)
{
// int order = trial_fe.GetOrder() + test_fe.GetOrder();
const int order = trial_fe.GetOrder() + test_fe.GetOrder() + Trans.OrderW();
@@ -1458,7 +1468,15 @@ const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
}
+67 -3
View File
@@ -2184,11 +2184,22 @@ public:
const Vector&, const Vector&,
Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const bool, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&,
Vector&, const int, const int);
using DiagonalKernelType = void(*)(const int, const bool, const Array<real_t>&,
const Array<real_t>&, const Vector&, Vector&,
const int, const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
struct Kernels { Kernels(); };
@@ -2341,7 +2352,8 @@ public:
void AddMultPatchPA(const int patch, const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
const FiniteElement &test_fe,
const bool stroud = false);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2352,6 +2364,13 @@ public:
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
const IntegrationRule* GetDefaultIntegrationRule(
@@ -2388,11 +2407,22 @@ public:
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&, Vector&,
const int, const int);
using DiagonalKernelType = void(*)(const int, const Array<real_t>&,
const Vector&, Vector&, const int,
const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
struct Kernels { Kernels(); };
@@ -2441,7 +2471,8 @@ public:
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans);
const ElementTransformation &Trans,
const bool stroud = false);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2452,6 +2483,13 @@ public:
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
@@ -2651,14 +2689,22 @@ public:
void AddMultMF(const Vector &x, Vector &y) const override;
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
// PA AddMultPA kernels
using VectorMassAddMultPAType =
void(*)(const int, const int,
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
MFEM_REGISTER_KERNELS(VectorMassAddMultPA,
VectorMassAddMultPAType,
(int, int, int));
// PA DiagonalPA kernels
using VectorMassAssembleDiagonalPAType =
void(*)(const int, const int, const int,
const real_t*, const real_t*, real_t*);
MFEM_REGISTER_KERNELS(VectorMassAssembleDiagonalPA,
VectorMassAssembleDiagonalPAType,
(int /*dim*/, int /*q1d*/));
};
@@ -3060,6 +3106,24 @@ public:
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultTransposePA(const Vector &x, Vector &y) const override;
using VectorDivergenceAddMultPAType =
void (*)(const int ne,
const Array<real_t> &b, const Array<real_t> &g, const Array<real_t> &bt,
const Vector &op, const Vector &x, Vector &y,
const int tr_d1d, const int te_d1d, const int q1d);
MFEM_REGISTER_KERNELS(VectorDivergenceAddMultPA,
VectorDivergenceAddMultPAType,
(int, int, int, int));
using VectorDivergenceAddMultTransposePAType =
void (*)(const int ne,
const Array<real_t> &bt, const Array<real_t> &gt, const Array<real_t> &b,
const Vector &q, const Vector &x, Vector &y,
const int tr_d1d, const int te_d1d, const int q1d);
MFEM_REGISTER_KERNELS(VectorDivergenceAddMultTransposePA,
VectorDivergenceAddMultTransposePAType,
(int, int, int, int));
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans);
+6
View File
@@ -54,6 +54,8 @@ void Coefficient::Project(QuadratureFunction &qf)
QuadratureSpaceBase &qspace = *qf.GetSpace();
const int ne = qspace.GetNE();
Vector values;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
@@ -327,6 +329,8 @@ void VectorCoefficient::Project(QuadratureFunction &qf)
const int ne = qspace.GetNE();
DenseMatrix values;
Vector col;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
@@ -695,6 +699,8 @@ void MatrixCoefficient::Project(QuadratureFunction &qf, bool transpose)
QuadratureSpaceBase &qspace = *qf.GetSpace();
const int ne = qspace.GetNE();
DenseMatrix values, matrix;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
+13 -26
View File
@@ -830,15 +830,9 @@ ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
int vsize = pfes->GetVSize();
Vector::Load(input, 2*vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
real_t *h_data = HostReadWrite();
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
@@ -1051,15 +1045,14 @@ void ParComplexGridFunction::Save(std::ostream &os) const
os << '\n';
int vsize = pfes->GetVSize();
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t *h_data = const_cast<real_t*>(HostRead());
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
if (pfes->GetOrdering() == Ordering::byNODES)
{
@@ -1070,14 +1063,8 @@ void ParComplexGridFunction::Save(std::ostream &os) const
Vector::Print(os, pfes->GetVDim());
}
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
os.flush();
}
+4
View File
@@ -114,6 +114,10 @@ void ConduitDataCollection::Save()
n_mesh["fields"][name]);
}
// TODO: in parallel, we need to call ParFiniteElementSpace::ApplyDofSigns
// for all ParGridFunction objects before and after saving, see
// ParGridFunction::Save.
// save mesh data
SaveMeshAndFields(myid,
n_mesh,
+42 -1
View File
@@ -167,7 +167,15 @@ public:
/** @brief Full multidimensional representation which does not use tensor
product structure. The ordering of the degrees of freedom is the
same as TENSOR, but the sizes of B and G are the same as FULL.*/
LEXICOGRAPHIC_FULL
LEXICOGRAPHIC_FULL,
/** @brief Ragged tensor product representation using 1D matrices/tensors
with dimensions using 1D number of quadrature points and ragged tensor degrees of
freedom. */
/** Used only for partial assembly of the H1 positive basis. The
size of B is d1d x qnpt x dim. Since different Gauss-Jacobi quadrature rules
are employed in each dimension, we need to store dim arrays. */
RAGGED_TENSOR
};
/// Describes the contents of the #B, #Bt, #G, and #Gt arrays, see #Mode.
@@ -228,6 +236,39 @@ public:
const Array<DofToQuad*> &dof2quad_array,
const IntegrationRule &ir,
DofToQuad::Mode mode);
virtual ~DofToQuad() = default;
};
/** @brief Structure representing the matrices/tensors needed to evaluate (in
reference space) the values, gradients, divergences, or curls of a positive
FiniteElement on simplices at the quadrature points of Stroud conical quadrature. */
class RaggedDofToQuad : public DofToQuad
{
public:
/** @brief Special basis function structures for positive (Bernstein) basis with
partial assembly. The storage layout of Ba1 is ndof x nqpt for scalar elements.
The storage layout of Ba2 is ndof x ndof x nqpt. In particular, we have
Ba2(iqpt, a1, a2) = B^{p-a1}_{a2}(x_{iqpt}). */
Array<real_t> Ba1, Ba2, Ba3;
Array<real_t> Ba1t, Ba2t, Ba3t;
/** @brief Special structures for gradients of positive basis with partial assembly.
The gradient arrays exploit properties of the Bernstein basis which allow grad(B^p_alpha)
to be expressed as the sum of products of B^{p-1}_alpha and the barycentric coordinates.
Thus, Ga1 and Ga2 simply contain the ragged tensor product components of B^{p-1}_alpha */
Array<real_t> Ga1, Ga2, Ga3;
Array<real_t> Ga1t, Ga2t, Ga3t;
/** @brief Mapping from the Bernstein multi-index (a_1, ..., a_d) to the lexicographic
dof index. */
Array<int> lex_map;
Array<int> forward_map2d_diff, forward_map3d_diff;
Array<int> inverse_map2d_diff, inverse_map3d_diff;
Array<int> forward_map2d_mass, forward_map3d_mass;
Array<int> inverse_map2d_mass, inverse_map3d_mass;
};
/// Describes the function space on each element
+302
View File
@@ -557,6 +557,101 @@ H1Pos_TriangleElement::H1Pos_TriangleElement(const int p)
}
}
const DofToQuad &H1Pos_TriangleElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt*ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize((int)nqpt*ndof*ndof);
rd2q->Ba1t.SetSize(nqpt*ndof);
rd2q->Ba2t.SetSize((int)nqpt*ndof*ndof);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt*(ndof -1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->Ga1t.SetSize(nqpt*(ndof -1));
rd2q->Ga2t.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->lex_map.SetSize(ndof * ndof);
Vector shape_a1(ndof), shape_a2(ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2((ndof-1) * (ndof-1));
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). Additionally, the Bernstein PA algorithms expect evaluation of the
// component 1D bases at the Stroud nodes pulled back to the unit square, so perform the pullback
// on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
rd2q->Ba2t[i + nqpt*(j + ndof*k)] = rd2q->Ba2[k + ndof*(j + ndof*i)] = shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
rd2q->Ga2t[i + nqpt*(j + (ndof-1)*k)] = rd2q->Ga2[k + (ndof-1)*(j +
(ndof-1)*i)] = shape_Ga2(k);
}
}
}
}
// stores the mapping from 2D Bernstein multi-index (i,j,p-i-j) to the
// lexicographic DOF ordering
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
int idx = ((2 * (ndof-1) + 3) - j) * j / 2 + i;
rd2q->lex_map[j + ndof*i] = idx;
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TriangleElement::CalcShape(
const int p, const real_t l1, const real_t l2, real_t *shape)
@@ -749,6 +844,213 @@ H1Pos_TetrahedronElement::H1Pos_TetrahedronElement(const int p)
}
}
const DofToQuad &H1Pos_TetrahedronElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
const int basis_dim2d = ndof*(ndof+1) / 2;
const int basis_dim3d = ndof*(ndof+1)*(ndof+2) / 6;
const int basis_dim2d_diff = (ndof-1)*(ndof) / 2;
const int basis_dim3d_diff = (ndof-1)*(ndof)*(ndof+1) / 6;
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt * ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize(nqpt * basis_dim2d);
// third component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba3.SetSize(nqpt * basis_dim3d);
rd2q->Ba1t.SetSize(nqpt * ndof);
rd2q->Ba2t.SetSize(nqpt * basis_dim2d);
rd2q->Ba3t.SetSize(nqpt * basis_dim3d);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt * (ndof-1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt * basis_dim2d_diff);
// stores third component of ragged tensor basis with order p-1
rd2q->Ga3.SetSize(nqpt * basis_dim3d_diff);
rd2q->Ga1t.SetSize(nqpt * (ndof-1));
rd2q->Ga2t.SetSize(nqpt * basis_dim2d_diff);
rd2q->Ga3t.SetSize(nqpt * basis_dim3d_diff);
rd2q->lex_map.SetSize(ndof * ndof * ndof);
rd2q->forward_map2d_diff.SetSize((ndof-1) * (ndof-1));
rd2q->forward_map3d_diff.SetSize((ndof-1) * (ndof-1) * (ndof-1));
rd2q->inverse_map2d_diff.SetSize(2 * basis_dim2d_diff);
rd2q->inverse_map3d_diff.SetSize(3 * basis_dim3d_diff);
rd2q->forward_map2d_mass.SetSize(ndof * ndof);
rd2q->forward_map3d_mass.SetSize(ndof * ndof * ndof);
rd2q->inverse_map2d_mass.SetSize(2 * basis_dim2d);
rd2q->inverse_map3d_mass.SetSize(2 * basis_dim3d);
// forward and inverse maps for multi-index to collpased 1d index for diffusion, can combine
// these four loops, but need four idx's and clause for shorter diff loops
int idx = 0;
for (int i = 0; i < ndof-1; i++)
{
for (int j = 0; j < ndof-i-1; j++)
{
rd2q->forward_map2d_diff[j + (ndof-1)*i] = idx;
rd2q->inverse_map2d_diff[2*idx] = i;
rd2q->inverse_map2d_diff[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof-1; k++)
{
for (int j = 0; j < ndof-k-1; j++)
{
for (int i = 0; i < ndof-k-j-1; i++)
{
rd2q->forward_map3d_diff[k + (ndof-1)*(j + (ndof-1)*i)] = idx;
rd2q->inverse_map3d_diff[3*idx] = i;
rd2q->inverse_map3d_diff[1 + 3*idx] = j;
rd2q->inverse_map3d_diff[2 + 3*idx] = k;
idx++;
}
}
}
// forward and inverse maps for multi-index to collpased 1d index for mass
idx = 0;
for (int j = 0; j < ndof; j++)
{
for (int i = 0; i < ndof-j; i++)
{
rd2q->forward_map2d_mass[j + ndof*i] = idx;
rd2q->inverse_map2d_mass[2*idx] = i;
rd2q->inverse_map2d_mass[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof; k++)
{
for (int j = 0; j < ndof-k; j++)
{
for (int i = 0; i < ndof-k-j; i++)
{
rd2q->forward_map3d_mass[k + ndof*(j + ndof*i)] = idx;
rd2q->inverse_map3d_mass[2*idx] = i;
rd2q->inverse_map3d_mass[1 + 2*idx] = j;
// d2q->inverse_map3d_mass[2 + 3*idx] = k;
idx++;
}
}
}
Vector shape_a1(ndof), shape_a2(ndof * ndof), shape_a3(ndof * ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2(ndof-1), shape_Ga3(ndof-1);
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). The first 'nqpt' points in the third dimension have the same z-coordinates
// as those of the 1D rule for the third dimension (i.e. Gauss-Legendre rule). Additionally,
// the Bernstein PA algorithms expect evaluation of the component 1D bases at the Stroud nodes
// pulled back to the unit cube, so perform the pullback on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
const real_t z = ir.IntPoint(nqpt*nqpt*i).z / (1.0 - ir.IntPoint(
nqpt*nqpt*i).x - ir.IntPoint(nqpt*nqpt*i).y);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
const int a_2d_mass = rd2q->forward_map2d_mass[k + ndof*j];
rd2q->Ba2t[i + nqpt*a_2d_mass] = rd2q->Ba2[a_2d_mass + basis_dim2d*i] =
shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
const int a_2d_diff = rd2q->forward_map2d_diff[k + (ndof-1)*j];
rd2q->Ga2t[i + nqpt*a_2d_diff] = rd2q->Ga2[a_2d_diff + basis_dim2d_diff*i] =
shape_Ga2(k);
Poly_1D::CalcBernstein(ndof-2-j-k, z, shape_Ga3);
}
Poly_1D::CalcBernstein(ndof-1-j-k, z, shape_a3);
for (int m = 0; m < ndof-j-k; m++)
{
const int a_3d_mass = rd2q->forward_map3d_mass[m + ndof*(k + ndof*j)];
rd2q->Ba3t[i + nqpt*a_3d_mass] = rd2q->Ba3[a_3d_mass + basis_dim3d*i] =
shape_a3(
m);
if (j < ndof-1 && k < ndof-j-1 && m < ndof-j-k-1)
{
// // collapsed 1D access
// d2q->Ga3[i + nqpt*(m + d2q->offset3d[k + (ndof-1)*j])] = shape_Ga3(m);
// collapsed 1D access with forward mapping
const int a_3d_diff = rd2q->forward_map3d_diff[m + (ndof-1)*(k + (ndof-1)*j)];
rd2q->Ga3t[i + nqpt*a_3d_diff] = rd2q->Ga3[a_3d_diff + basis_dim3d_diff*i] =
shape_Ga3(m);
}
}
}
}
}
// stores the mapping from 3D Bernstein multi-index (i,j,k,p-i-j-k) to the
// lexicographic DOF ordering
int p = ndof - 1;
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
for (int k = 0; k < ndof-i-j; k++)
{
int dof = (p+1)*(p+2)*(p+3) / 6;
int tet = (p-k)*(p-k+1)*(p-k+2) / 6;
int tri = (p+1-k-j)*(p+2-k-j)/2;
int multi_idx = dof - tet - tri + i;
rd2q->lex_map[k + ndof*(j + ndof*i)] = multi_idx;
}
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TetrahedronElement::CalcShape(
const int p, const real_t l1, const real_t l2, const real_t l3,
+30
View File
@@ -191,6 +191,21 @@ public:
/// Construct the H1Pos_TriangleElement of order @a p
H1Pos_TriangleElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)/2 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
real_t *shape);
@@ -220,6 +235,21 @@ public:
/// Construct the H1Pos_TetrahedronElement of order @a p
H1Pos_TetrahedronElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)(p+3)/6 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
const real_t z, real_t *shape);
+27
View File
@@ -250,6 +250,14 @@ public:
its GetOrder() method. */
virtual FiniteElementCollection *Clone(int p) const;
/** @brief Return the order parameter used to construct this collection.
* This differs from GetOrder() depending on the collection type. */
virtual int GetConstructorOrder() const
{
MFEM_ABORT("Collection " << Name() << " does not support GetConstructorOrder");
return -1;
}
protected:
const int base_p; ///< Order as returned by GetOrder().
@@ -314,6 +322,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new H1_FECollection(p, dim, b_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~H1_FECollection();
};
@@ -343,6 +354,10 @@ class H1_Trace_FECollection : public H1_FECollection
public:
H1_Trace_FECollection(const int p, const int dim,
const int btype = BasisType::GaussLobatto);
FiniteElementCollection *Clone(int p) const override
{ return new H1_Trace_FECollection(p, dim+1, b_type); }
};
/// Arbitrary order "L2-conforming" discontinuous finite elements.
@@ -396,6 +411,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new L2_FECollection(p, dim, b_type, m_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~L2_FECollection();
};
@@ -456,6 +474,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new RT_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return base_p-1; }
virtual ~RT_FECollection();
};
@@ -536,6 +557,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new ND_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return dim>1 ? base_p : base_p+1; }
virtual ~ND_FECollection();
};
@@ -548,6 +572,9 @@ public:
ND_Trace_FECollection(const int p, const int dim,
const int cb_type = BasisType::GaussLobatto,
const int ob_type = BasisType::GaussLegendre);
FiniteElementCollection *Clone(int p) const override
{ return new ND_Trace_FECollection(p, dim+1, cb_type, ob_type); }
};
/// Arbitrary order 3D H(curl)-conforming Nedelec finite elements in 1D.
+1 -2
View File
@@ -4631,9 +4631,8 @@ FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
ElementDofOrdering GetEVectorOrdering(const FiniteElementSpace& fes)
{
return UsesTensorBasis(fes)?
return (UsesTensorBasis(fes) || fes.UsesRaggedTensorBasis()) ?
ElementDofOrdering::LEXICOGRAPHIC:
ElementDofOrdering::NATIVE;
}
} // namespace mfem
+12
View File
@@ -1514,6 +1514,18 @@ public:
return dynamic_cast<const L2_FECollection*>(fec) != NULL;
}
/// @brief Return true if the mesh contains only one topology, the elements are
/// all triangles or tetrahedrons, and the elements are ragged tensor elements
/// i.e. Bernstein/positive basis.
bool UsesRaggedTensorBasis() const
{
bool simplex = this->GetMesh()->IsSimplexMesh();
bool positive =
dynamic_cast<const mfem::H1Pos_TriangleElement *>(this->GetTypicalFE()) ||
dynamic_cast<const mfem::H1Pos_TetrahedronElement *>(this->GetTypicalFE());
return simplex && positive;
}
/** In variable-order spaces on nonconforming (NC) meshes, this function
controls whether strict conformity is enforced in cases where coarse
edges/faces have higher polynomial order than their fine NC neighbors.
+167
View File
@@ -2256,6 +2256,104 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
}
}
void GridFunction::AccumulateAndCountTraceValues(
Coefficient *coeff[], VectorCoefficient *vcoeff,
Array<int> &values_counter)
{
if (vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
"vcoeff vdim != fes VDim");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::VALUE &&
fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR,
"Can only call ProjectTraceCoefficient on scalar value-type "
"trace elements. "
"Use ProjectTraceCoefficientNormal for RT and "
"ProjectTraceCoefficientTangent for ND finite elements.");
}
Array<int> vdofs;
Vector vc;
values_counter.SetSize(Size());
values_counter = 0;
const int vdim = fes->GetVDim();
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
const FiniteElement *fe = fes->GetFaceElement(i);
const int fdof = fe->GetDof();
ElementTransformation *transf = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
fes->GetFaceVDofs(i, vdofs);
for (int j = 0; j < fdof; j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
transf->SetIntPoint(&ip);
if (vcoeff) { vcoeff->Eval(vc, *transf, ip); }
for (int d = 0; d < vdim; d++)
{
if (!vcoeff && !coeff[d]) { continue; }
real_t val = vcoeff ? vc(d) : coeff[d]->Eval(*transf, ip);
int ind = vdofs[fdof*d+j];
if ( ind < 0 )
{
val = -val, ind = -1-ind;
}
if (++values_counter[ind] == 1)
{
(*this)(ind) = val;
}
else
{
(*this)(ind) += val;
}
}
}
}
}
void GridFunction::AccumulateAndCountTraceTangentValues(
VectorCoefficient &vcoeff, Array<int> &values_counter)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()
->GetRangeType() == FiniteElement::VECTOR &&
fes->GetTypicalTraceElement()
->GetMapType() == FiniteElement::H_CURL,
"Not an ND FE space!");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetPhysRangeDim(
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
"vcoeff vdim != PhysRangeDim");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
Vector lvec;
values_counter.SetSize(Size());
values_counter = 0;
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
fes->GetFaceVDofs(i, dofs);
lvec.SetSize(fe->GetDof());
fe->Project(vcoeff, *T, lvec);
accumulate_dofs(dofs, lvec, *this, values_counter);
}
}
void GridFunction::ComputeMeans(AvgType type, Array<int> &zones_per_vdof)
{
switch (type)
@@ -2698,6 +2796,74 @@ void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
}
}
void GridFunction::ProjectTraceCoefficient(Coefficient *coeff[])
{
Array<int> values_counter;
AccumulateAndCountTraceValues(coeff, NULL, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficient(Coefficient &coeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == 1, "ProjectTraceCoefficient(Coefficient&)"
"is only valid for scalar GridFunction");
Coefficient *coeff_p = &coeff;
ProjectTraceCoefficient(&coeff_p);
}
void GridFunction::ProjectTraceCoefficient(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == vcoeff.GetVDim(),
"Incompatible vcoeff vdim and fes vdim");
Array<int> values_counter;
AccumulateAndCountTraceValues(NULL, &vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR &&
fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::INTEGRAL, "Not an RT FE space!");
MFEM_VERIFY(vcoeff.GetVDim() == fes->GetMesh()->SpaceDimension(),
"vcoeff vdim (" << vcoeff.GetVDim()
<< ") != SpaceDimension ("
<< fes->GetMesh()->SpaceDimension() << ")");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
int dim = vcoeff.GetVDim();
Vector vc(dim), nor(dim), lvec;
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
lvec.SetSize(fe->GetDof());
for (int j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
vcoeff.Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
lvec(j) = (vc * nor);
}
fes->GetFaceVDofs(i, dofs);
SetSubVector(dofs, lvec);
}
}
void GridFunction::ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff)
{
Array<int> values_counter;
AccumulateAndCountTraceTangentValues(vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectCoefficientGlobalL2(VectorCoefficient &vcoeff,
real_t rtol, int iter)
{
@@ -5286,6 +5452,7 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
Vector lel, uel;
GetElementBounds(plb, lel, uel, vdim);
+24
View File
@@ -578,6 +578,13 @@ protected:
const Array<int> &bdr_attr,
Array<int> &values_counter);
void AccumulateAndCountTraceValues(Coefficient *coeff[],
VectorCoefficient *vcoeff,
Array<int> &values_counter);
void AccumulateAndCountTraceTangentValues(VectorCoefficient &vcoeff,
Array<int> &values_counter);
// Complete the computation of averages; called e.g. after
// AccumulateAndCountZones().
void ComputeMeans(AvgType type, Array<int> &zones_per_vdof);
@@ -663,6 +670,23 @@ public:
ProjectBdrCoefficient(&coeff_p, attr);
}
/// Project a Coefficient on a GridFunction defined on H1 trace space
void ProjectTraceCoefficient(Coefficient *coeff[]);
void ProjectTraceCoefficient(Coefficient &coeff);
/** @brief Project a VectorCoefficient @a vcoeff on a GridFunction
defined on a Vector H1 trace space. Note that this also works
for a scalar H1 trace space, where only the first component of
@a vcoeff is used. */
void ProjectTraceCoefficient(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an RT trace space */
void ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an ND trace space */
void ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on the GridFunction, modifying only
DOFs on the boundary associated with the boundary attributes marked in
the @a attr array. */
+1257 -734
View File
File diff suppressed because it is too large Load Diff
+161 -46
View File
@@ -12,6 +12,9 @@
#ifndef MFEM_GSLIB
#define MFEM_GSLIB
#include <map>
#include <vector>
#include "../config/config.hpp"
#ifdef MFEM_USE_MPI
#include "pgridfunc.hpp"
@@ -119,6 +122,11 @@ protected:
// IntegrationRules for simplex->Quad/Hex and to project to p_max in-case of
// p-refinement.
Array<IntegrationRule *> ir_split;
/// Integration rules built at the field polynomial order (only for surface
/// meshes when mesh order is not the same as gridfunction order).
Array<IntegrationRule *> ir_split_sol;
/// Order at which #ir_split_sol was built; -1 means not built.
int ir_split_sol_order = -1;
Array<FiniteElementSpace *> fes_rst_map; //FESpaces to map Quad/Hex->Simplex
Array<GridFunction *> gf_rst_map; // GridFunctions to map Quad/Hex->Simplex
FiniteElementCollection *fec_map_lin;
@@ -134,6 +142,8 @@ protected:
AvgType avgtype; // average type used for L2 functions
Array<int> split_element_map;
Array<int> split_element_index;
// Geometry::Type (as int) of the original element for each split quad.
Array<int> split_element_geom;
int NE_split_total; // total number of elements after mesh splitting
int mesh_points_cnt; // number of mesh nodes
// Tolerance to ignore points found beyond the mesh boundary.
@@ -141,6 +151,12 @@ protected:
double bdr_tol;
// Use CPU functions for Mesh/GridFunction on device for gslib1.0.7
bool gpu_to_cpu_fallback = false;
// Check if a point is inside the oriented bounding box of an
// element before the Newton iteration.
// Note: only used in MFEM implementation (not in gslib) which currently
// supports GPU kernels for area meshes in 2D, volume meshes in 3D,
// and surface meshes in 1D/2D/3D.
bool obb_check = true;
// Device specific data used for FindPoints
struct DEV_STRUCT
@@ -162,11 +178,16 @@ protected:
mutable double surf_dist_tol;
} DEV;
/// Use GSLIB for communication and interpolation
// Helper function to setup and free gslib's crystal router.
void SetupCrystal(); // Called inside Setup and SetupSurf_base
void FreeCrystal(); // Called inside FreeData
/// Use GSLIB for communication and interpolation. Updates field_out on
/// host.
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
/// Uses GSLIB Crystal Router for communication followed by MFEM's
/// interpolation functions
/// interpolation functions. Updates field_out on host.
virtual void InterpolateGeneral(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering);
@@ -181,12 +202,26 @@ protected:
IntegrationRule *irule,
int order);
/** @brief Build integration rules at the given @a order for each split mesh
* and store them in @a ir_out. Requires that \ref SetupSplitMeshes has
* already been called. */
virtual void SetupIntegrationRules(const int order,
Array<IntegrationRule *> &ir_out);
/** @brief Helper function that calls \ref SetupSplitMeshes and
* \ref SetupIntegrationRuleForSplitMesh. */
* \ref SetupIntegrationRules. */
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
/// Get GridFunction value at the points expected by GSLIB.
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
/** @brief Get GridFunction value at the points expected by GSLIB.
* @param[in] gf_in Grid function to evaluate.
* @param[out] node_vals Output values.
* @param[in] ir_in If non-null, use these rules instead of #ir_split.
* @param[in] by_element If true, output has element-major layout
* [nel][vdim][ndofs]; otherwise component-major
* layout [vdim][total_pts]. */
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals,
const Array<IntegrationRule *> *ir_in = nullptr,
bool by_element = false) const;
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
* simplices, find the original element number (that was split into
@@ -293,9 +328,10 @@ protected:
const unsigned n,
const uint nel,
const unsigned m,
const double bbox_tol,
const double bbox_rel_size_inc,
const uint local_hash_size,
const uint global_hash_size);
const uint global_hash_size,
const Vector *aabb_sz_inc);
/// Preprocess 3D surface mesh needed for FindPoints.
void findptssurf_setup_3(DEV_STRUCT &devs,
@@ -303,17 +339,47 @@ protected:
const unsigned n,
const uint nel,
const unsigned m,
const double bbox_tol,
const double bbox_rel_size_inc,
const uint local_hash_size,
const uint global_hash_size,
const int rD);
const int rD,
const Vector *aabb_sz_inc);
/** @brief Shared implementation for the public surface-setup methods.
*
* @details Initializes the surface-search data structures, builds the
* split-element representation expected by gslib, and constructs the
* element bounding boxes used by the MFEM surface kernels.
*
* If @a aabb_sz_inc is null, the setup stores the default oriented
* bounding boxes and uses @a bbox_rel_size_inc as their relative size
* increase factor.
*
* If @a aabb_sz_inc is non-null, the setup stores axis-aligned bounding
* boxes only, applies the requested absolute AABB expansion in each
* physical direction, and adjusts the tolerance @a bdr_tol so points
* found in the expanded region are classified as border points.
*
* @param[in] m Input surface mesh.
* @param[in] bbox_rel_size_inc Relative size increase applied when
* expanding each element bounding box during
* setup.
* @param[in] aabb_sz_inc Optional total absolute AABB expansion
* applied to the stored axis-aligned
* bounding boxes after construction.
* @param[in] newt_tol Newton tolerance for the point-search
* kernels.
*/
void SetupSurf_Base(Mesh &m,
const double bbox_rel_size_inc,
const Vector *aabb_sz_inc,
const double newt_tol);
public:
/// Serial constructor
FindPointsGSLIB();
/// Serial constructor + setup with given Mesh (see \ref Setup)
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
FindPointsGSLIB(Mesh &mesh_in, const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
@@ -322,7 +388,7 @@ public:
FindPointsGSLIB(MPI_Comm comm_);
/// Constructor + setup with given ParMesh (see \ref Setup)
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
FindPointsGSLIB(ParMesh &mesh_in, const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
#endif
@@ -338,23 +404,59 @@ public:
Note: not tested with periodic (L2).
Note: the input mesh \p m must have Nodes set.
@param[in] m Input mesh.
@param[in] bb_t (Optional) Relative size of bounding box around
each element.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for simultaneous
iteration. This alters performance and
memory footprint.
@param[in] m Input mesh.
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
when expanding each element bounding box.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for
simultaneous iteration. This alters
performance and memory footprint.
*/
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
void Setup(Mesh &m, const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
/// Preprocess the surface mesh to compute data for FindPoints.
void SetupSurf(Mesh &m,
const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12);
/** @brief Preprocess the surface mesh to compute data for FindPoints using
* absolute AABB expansion.
*
* @details This method computes only axis-aligned bounding boxes and
* increases their total length by a user-specified amount in each
* physical direction. The absolute AABB expansion is applied
* symmetrically to the lower and upper bounds.
*
* The size of @a aabb_sz_inc determines how the expansion values are
* interpreted:
* - `1`: one expansion value used in every direction for every element
* - `NElements`: one expansion value per element, reused in x/y/z
* directions
* - `SpaceDim`: one expansion value per physical direction, reused for
* every element
* - `NElements*SpaceDim`: one expansion value per element and direction,
* ordered as `(dx1,dy1,dz1, ... dxN,dyN,dzN)`
*
* This method disables the oriented bounding-box precheck because the
* stored boxes are modified only in their axis-aligned representation.
*
* @param[in] m Input surface mesh.
* @param[in] aabb_sz_inc Total absolute AABB expansion applied in
* each physical direction to the stored
* axis-aligned bounding boxes.
* @param[in] newt_tol Newton tolerance for the point-search
* kernels.
*
* @note We disable the oriented bounding box check with this setup.
* @a bdr_tol is also adjusted so that all points in the AABBs can
* be found.
*/
void SetupSurfWithAABBExpansion(Mesh &m, const Vector &aabb_sz_inc,
const double newt_tol = 1.0e-12);
/** @brief Searches positions given in physical space by \p point_pos.
@@ -401,7 +503,8 @@ public:
/// Setup FindPoints and search positions
void FindPoints(Mesh &m, const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** @brief Interpolation of field values at prescribed reference space
@@ -413,7 +516,11 @@ public:
mesh that was given to Setup().
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value.
The output ordering is determined from field_in.*/
The output ordering is determined from field_in.
@note: field_out is moved to device if field_in is on device. Otherwise,
field_out memory allocation is not changed.
*/
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
/// Interpolation of field values, with output ordering specification.
@@ -468,7 +575,12 @@ public:
* @details When using FindPoints, gslib may return points as found on the
* boundary even when they are slightly outside the domain. This tolerance
* is used to filter such points based on the distance^2 value and mark them
* as not found.*/
* as not found.
*
* @note When the SetupSurfWithAABBExpansion method is used for surface
* meshes, this tolerance is automatically computed based on the size of
* expanded AABBs. Using this method will override that computed tolerance.
* */
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
{
bdr_tol = bdr_tol_;
@@ -603,25 +715,28 @@ public:
Note: not tested with periodic meshes (L2).
Note: the input mesh \p m must have Nodes set.
@param[in] m Input mesh.
@param[in] meshid A unique # for each overlapping mesh. This id is
used to make sure that points being searched are not
looked for in the mesh that they belong to.
@param[in] gfmax (Optional) GridFunction in H1 that is used as a
discriminator when one point is located in multiple
meshes. The mesh that maximizes gfmax is chosen.
For example, using the distance field based on the
overlapping boundaries is helpful for convergence
during Schwarz iterations.
@param[in] bb_t (Optional) Relative size of bounding box around
each element.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for simultaneous
iteration. This alters performance and
memory footprint.*/
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = NULL,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
@param[in] m Input mesh.
@param[in] meshid A unique # for each overlapping mesh.
This id is used to make sure that points
being searched are not looked for in the
mesh that they belong to.
@param[in] gfmax (Optional) GridFunction in H1 that is used
as a discriminator when one point is
located in multiple meshes. The mesh that
maximizes gfmax is chosen. For example,
using the distance field based on the
overlapping boundaries is helpful for
convergence during Schwarz iterations.
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
when expanding each element bounding box.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for
simultaneous iteration. This alters
performance and memory footprint.*/
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = nullptr,
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** Searches positions given in physical space by \p point_pos. All output
@@ -677,7 +792,7 @@ class GSOPGSLIB
protected:
struct gslib::crystal *cr; // gslib's internal data
struct gslib::comm *gsl_comm; // gslib's internal data
struct gslib::gs_data *gsl_data = NULL;
struct gslib::gs_data *gsl_data = nullptr;
int num_ids;
public:
+64 -170
View File
@@ -11,7 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/kernels.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -27,8 +27,6 @@
#pragma GCC diagnostic pop
#endif
#include <climits>
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
@@ -54,127 +52,14 @@ struct findptsElementGPT_t
double x[DIM], jac[DIM * DIM], hes[4];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[DIM], A[DIM * DIM];
dbl_range_t x[DIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[DIM];
double fac[DIM];
unsigned int *offset;
int max;
};
// Eval the ith Lagrange interpolant and its first derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x - z[j]);
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN+i] = 2.0 * lCoeff[i] * u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x - z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN+i] = 2.0 * lCoeff[i] * u1;
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
}
// Axis-aligned bounding box test.
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
const double x[2])
{
double test = 1;
for (int d = 0; d < 2; ++d)
{
double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
test = test < 0 ? test : b_d;
}
return test;
}
// Axis-aligned bounding box test followed by oriented bounding-box test.
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
const double x[2])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0)
{
return bxyz;
}
else
{
double dxyz[2];
for (int d = 0; d < 2; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1;
for (int d = 0; d < 2; ++d)
{
double rst = 0;
for (int e = 0; e < 2; ++e)
{
rst += b->A[d * 2 + e] * dxyz[e];
}
double brst = (rst + 1) * (1 - rst);
test = test < 0 ? test : brst;
}
return test;
}
}
// Element index corresponding to hash mesh that the point is located in.
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[2])
{
const int n = p->hash_n;
int sum = 0;
for (int d = 2 - 1; d >= 0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
}
return sum;
}
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<DIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_first_der;
using gslib::lag_eval_second_der;
/*Solve Ax=y. A is row-major */
static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
@@ -185,12 +70,6 @@ static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
x[1] = idet*(A[0]*y[1] - A[2]*y[0]);
}
/* L2 norm squared. */
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
{
return x[0] * x[0] + x[1] * x[1];
}
/* the bit structure of flags is CSSRR
the C bit --- 1<<4 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
@@ -352,7 +231,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *res,
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2(resid);
const double dist2 = l2norm2<2>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d = 0; d < 2; ++d)
@@ -695,25 +574,25 @@ static MFEM_HOST_DEVICE double tensor_ig2_j(double *g_partials,
}
template<int T_D1D = 0>
static void FindPointsLocal2D_Kernel(const int npt,
const double tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
static void FindPointsLocal2DKernel(const int npt,
const double tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
@@ -1175,30 +1054,45 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
switch (DEV.dof1d)
{
case 2:
return FindPointsLocal2D_Kernel<2>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
FindPointsLocal2DKernel<2>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
return FindPointsLocal2D_Kernel<3>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
FindPointsLocal2DKernel<3>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
return FindPointsLocal2D_Kernel<4>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
FindPointsLocal2DKernel<4>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 5:
return FindPointsLocal2D_Kernel<5>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
FindPointsLocal2DKernel<5>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx,
plhm, plhf, plho, pcode, pelem,
pref, pdist, pgll1d, plc, DEV.dof1d);
FindPointsLocal2DKernel(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
#undef DIM2
+29 -157
View File
@@ -11,9 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/kernels.hpp"
#include <climits>
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -59,128 +57,15 @@ struct findptsElemPt
double x[DIM], jac[DIM * DIM], hes[18];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[DIM], A[DIM * DIM];
dbl_range_t x[DIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[DIM];
double fac[DIM];
unsigned int *offset;
// int max;
};
// Eval the ith Lagrange interpolant and its first derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2*(x-z[j]);
u1 = d_j*u1+u0;
u0 = d_j*u0;
}
}
p0[i] = lCoeff[i]*u0;
p0[pN+i] = 2.0*lCoeff[i]*u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2*(x-z[j]);
u2 = d_j*u2+u1;
u1 = d_j*u1+u0;
u0 = d_j*u0;
}
}
p0[i] = lCoeff[i]*u0;
p0[pN+i] = 2.0*lCoeff[i]*u1;
p0[2*pN+i] = 8.0*lCoeff[i]*u2;
}
// Axis-aligned bounding box test.
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
const double x[3])
{
double b_d;
for (int d = 0; d < 3; ++d)
{
b_d = (x[d]-b->x[d].min)*(b->x[d].max-x[d]);
if (b_d < 0) { return b_d; }
}
return b_d;
}
// Axis-aligned bounding box test followed by oriented bounding-box test.
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
const double x[3])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0)
{
return bxyz;
}
else
{
double dxyz[3];
for (int d = 0; d < 3; ++d)
{
dxyz[d] = x[d]-b->c0[d];
}
double test = 1;
for (int d = 0; d < 3; ++d)
{
double rst = 0;
for (int e = 0; e < 3; ++e)
{
rst += b->A[d*3+e]*dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test < 0 ? test : brst;
}
return test;
}
}
// Element index corresponding to hash mesh that the point is located in.
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[3])
{
const int n = p->hash_n;
int sum = 0;
for (int d = 3-1; d >= 0; --d)
{
sum *= n;
int i = (int)floor((x[d]-p->bnd[d].min)*p->fac[d]);
sum += i < 0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<DIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_first_der;
using gslib::lag_eval_second_der;
using gslib::lin_solve_sym_2;
// Solve Ax=y. A is row-major.
static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
@@ -199,22 +84,6 @@ static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
x[2] = idet*(inv6*y[0]+inv7*y[1]+inv8*y[2]);
}
// Solve Ax=y. A is a symmetric 2x2 matrix.
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
const double A[3],
const double y[2])
{
const double idet = 1 / (A[0]*A[2]-A[1]*A[1]);
x[0] = idet*(A[2]*y[0]-A[1]*y[1]);
x[1] = idet*(A[0]*y[1]-A[1]*y[0]);
}
// L2 norm.
static MFEM_HOST_DEVICE inline double l2norm2(const double x[3])
{
return x[0]*x[0]+x[1]*x[1]+x[2]*x[2];
}
/* the bit structure of flags is CTTSSRR
the C bit --- 1<<6 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
@@ -459,7 +328,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsPt *res,
const findptsPt *p,
const double tol)
{
const double dist2 = l2norm2(resid);
const double dist2 = l2norm2<3>(resid);
const double decr = p->dist2-dist2;
const double pred = p->dist2p;
for (int d = 0; d < 3; ++d)
@@ -1809,33 +1678,36 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
{
case 2:
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
break;
case 3:
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
break;
case 4:
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
break;
case 5:
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
break;
default:
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc,
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc,
DEV.dof1d);
break;
}
}
#undef pMax
+107 -176
View File
@@ -11,6 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -52,113 +53,14 @@ struct findptsElementGPT_t
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[sDIM], A[sDIM*sDIM];
dbl_range_t x[sDIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[sDIM];
double fac[sDIM];
unsigned int *offset;
};
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x-z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
p0[i] = lCoeff[i] * u0;
p1[i] = 2.0 * lCoeff[i] * u1;
p2[i] = 8.0 * lCoeff[i] * u2;
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
const double x[sDIM])
{
double b_d;
for (int d=0; d<sDIM; ++d)
{
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
if (b_d < 0) // if outside in any dimension
{
return b_d;
}
}
return b_d; // only positive if inside
}
/* positive when given point is possibly inside given obbox b */
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
const double x[sDIM])
{
const double bxyz = obbox_axis_test(b,x);
if (bxyz<0) // test if point is in AABB
{
return bxyz;
}
else // test OBB only if inside AABB
{
double dxyz[sDIM];
for (int d=0; d<sDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1;
for (int d=0; d<sDIM; ++d)
{
double rst = 0;
for (int e=0; e<sDIM; ++e)
{
rst += b->A[d*2 + e] * dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test<0 ? test : brst;
}
return test;
}
}
/* Hash index in the hash table to the elements that possibly contain the point x */
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[2])
{
const int n = p->hash_n;
int sum = 0;
for (int d=sDIM-1; d>=0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
{
return x[0] * x[0] + x[1] * x[1];
}
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<sDIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
using gslib::AABB_test;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_second_der;
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
@@ -187,29 +89,29 @@ static MFEM_HOST_DEVICE inline int point_index(const int x)
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out->dist2, out->index, out->x, out->oldr in any event,
leaving out->r, out->dr, out->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
const double resid[2],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2(resid);
const double dist2 = l2norm2<2>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
out->x[0] = p->x[0];
out->x[1] = p->x[1];
out->oldr = p->r;
out->dist2 = dist2;
out_pt->x[0] = p->x[0];
out_pt->x[1] = p->x[1];
out_pt->oldr = p->r;
out_pt->dist2 = dist2;
if (decr >= 0.01*pred)
{
if (decr >= 0.9*pred) // very good iteration
{
out->tr = p->tr*2;
out_pt->tr = p->tr*2;
}
else // somewhat good iteration
{
out->tr = p->tr;
out_pt->tr = p->tr;
}
return false;
}
@@ -220,21 +122,21 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out->tr = v0/4.0;
out->dist2 = p->dist2;
out->r = p->oldr;
out->flags = p->flags>>3;
out->dist2p = -HUGE_VAL;
out_pt->tr = v0/4.0;
out_pt->dist2 = p->dist2;
out_pt->r = p->oldr;
out_pt->flags = p->flags>>3;
out_pt->dist2p = -HUGE_VAL;
if (pred < dist2*tol)
{
out->flags |= CONVERGED_FLAG;
out_pt->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
out,
out_pt,
const double jac[2],
const double rhess,
const double resid[2],
@@ -304,9 +206,9 @@ newton_edge_fin:
{
new_flags |= CONVERGED_FLAG;
}
out->r = newr;
out->dist2p = -v;
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
out_pt->r = newr;
out_pt->dist2p = -v;
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
}
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
@@ -332,26 +234,27 @@ static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal2D_Kernel( const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0 )
static void FindPointsEdgeLocal2DKernel( const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const bool obb_check,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0 )
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
@@ -412,22 +315,34 @@ static void FindPointsEdgeLocal2D_Kernel( const int npt,
{
const unsigned int el = *elp;
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
bool pass_bb = true;
obbox_t box;
int n_box_ents = 3*sDIM + sDIM2;
for (int idx = 0; idx < sDIM; ++idx)
if (obb_check)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
pass_bb = (bbox_test(&box, x_i) >= 0);
}
else
{
for (int d = 0; d < sDIM; ++d)
{
box.x[d].min = boxinfo[n_box_ents*el + d];
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
}
pass_bb = (AABB_test(&box, x_i) >= 0);
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
if (obbox_test(&box,x_i)>=0)
if (pass_bb)
{
//------------ findpts_local ------------------
{
@@ -516,11 +431,14 @@ static void FindPointsEdgeLocal2D_Kernel( const int npt,
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
}
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
edge.x[d][j] = elx[d][j];
}
}
@@ -681,28 +599,41 @@ void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
const bool obb_chk = obb_check;
switch (DEV.dof1d)
{
case 2:
return FindPointsEdgeLocal2D_Kernel<2>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal2DKernel<2>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
return FindPointsEdgeLocal2D_Kernel<3>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal2DKernel<3>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
return FindPointsEdgeLocal2D_Kernel<4>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal2DKernel<4>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
return FindPointsEdgeLocal2D_Kernel(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
FindPointsEdgeLocal2DKernel(npt, DEV.newt_tol, dist2tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
#undef sDIM
+109 -181
View File
@@ -11,6 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -54,117 +55,14 @@ struct findptsElementGPT_t
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[sDIM], A[sDIM*sDIM];
dbl_range_t x[sDIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[sDIM];
double fac[sDIM];
unsigned int *offset;
};
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j=0; j<pN; ++j)
{
if (i!=j)
{
double d_j = 2 * (x-z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
p0[i] = lCoeff[i] * u0;
p1[i] = 2.0 * lCoeff[i] * u1;
p2[i] = 8.0 * lCoeff[i] * u2;
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
const double x[sDIM])
{
double b_d;
for (int d=0; d<sDIM; ++d)
{
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
if (b_d < 0) // if outside in any dimension
{
return b_d;
}
}
return b_d; // only positive if inside in all dimensions
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
const double x[sDIM])
{
const double bxyz = obbox_axis_test(b, x);
if (bxyz<0)
{
return bxyz;
}
else
{
double dxyz[3];
// dxyz: distance of the point from the center of the OBB
for (int d=0; d<sDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
// transform dxyz to the local coordinate system of the OBB,
// and check if the point is inside the OBB [-1,1]^sDIM
double test = 1;
for (int d=0; d<sDIM; ++d)
{
double rst = 0;
for (int e=0; e<sDIM; ++e)
{
rst += b->A[d*sDIM + e] * dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test<0 ? test : brst;
}
return test;
}
}
/* Hash index in the hash table to the elements that possibly contain the point x */
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[sDIM])
{
const int n = p->hash_n;
int sum = 0;
for (int d=sDIM-1; d>=0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
{
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2] );
}
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<sDIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
using gslib::AABB_test;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_second_der;
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
@@ -175,47 +73,46 @@ static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
#define CONVERGED_FLAG (1u<<2)
#define FLAG_MASK 0x07u
/* returns the number of constrained reference coordinates, max 2
/* returns the number of constrained reference coordinates, max 1
*/
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
{
const int y = (flags | flags>>1);
return (y & 1u) + (y>>2 & 1u);
return ((flags | flags>>1) & 1u);
}
static MFEM_HOST_DEVICE inline int point_index(const int x)
{
return ((x>>1)&1u) | ((x>>2)&2u);
return ((x>>1)&1u);
}
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out->dist2, out->index, out->x, out->oldr in any event,
leaving out->r, out->dr, out->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
const double resid[3],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = norm2(resid);
const double dist2 = l2norm2<sDIM>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d=0; d<sDIM; ++d)
{
out->x[d] = p->x[d];
out_pt->x[d] = p->x[d];
}
out->oldr = p->r;
out->dist2 = dist2;
out_pt->oldr = p->r;
out_pt->dist2 = dist2;
if (decr>=0.01*pred)
{
if (decr>=0.9*pred) // very good iteration
{
out->tr = 2*p->tr;
out_pt->tr = 2*p->tr;
}
else // good iteration
{
out->tr = p->tr;
out_pt->tr = p->tr;
}
return false;
}
@@ -226,21 +123,21 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out->tr = v0/4.0;
out->dist2 = p->dist2;
out->r = p->oldr;
out->flags = p->flags>>3;
out->dist2p = -HUGE_VAL;
out_pt->tr = v0/4.0;
out_pt->dist2 = p->dist2;
out_pt->r = p->oldr;
out_pt->flags = p->flags>>3;
out_pt->dist2p = -HUGE_VAL;
if (pred<dist2*tol)
{
out->flags |= CONVERGED_FLAG;
out_pt->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
out,
out_pt,
const double jac[sDIM*rDIM],
const double rhes,
const double resid[sDIM],
@@ -314,9 +211,9 @@ newton_edge_fin:
{
new_flags |= CONVERGED_FLAG;
}
out->r = nr;
out->dist2p = -v;
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
out_pt->r = nr;
out_pt->dist2p = -v;
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
#undef EVAL
}
@@ -338,31 +235,32 @@ static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
{
dx[d] = x[d] - elx[d][ir];
}
dist2[ir] = norm2(dx);;
dist2[ir] = l2norm2(dx);
r[ir] = z[ir];
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal3D_Kernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
static void FindPointsEdgeLocal3DKernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const bool obb_check,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
@@ -419,21 +317,35 @@ static void FindPointsEdgeLocal3D_Kernel(const int npt,
for (; elp!=ele; ++elp)
{
const unsigned int el = *elp;
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
bool pass_bb = true;
obbox_t box;
int n_box_ents = 3*sDIM + sDIM2;
for (int idx = 0; idx < sDIM; ++idx)
if (obb_check)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
pass_bb = (bbox_test(&box, x_i) >= 0);
}
for (int idx = 0; idx < sDIM2; ++idx)
else
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
for (int d = 0; d < sDIM; ++d)
{
box.x[d].min = boxinfo[n_box_ents*el + d];
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
}
pass_bb = (AABB_test(&box, x_i) >= 0);
}
if (obbox_test(&box, x_i)>=0)
if (pass_bb)
{
//// findpts_local ////
{
@@ -521,11 +433,14 @@ static void FindPointsEdgeLocal3D_Kernel(const int npt,
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
}
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
edge.x[d][j] = elx[d][j];
}
}
@@ -688,28 +603,41 @@ void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
const bool obb_chk = obb_check;
switch (DEV.dof1d)
{
case 2:
return FindPointsEdgeLocal3D_Kernel<2>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
return FindPointsEdgeLocal3D_Kernel<3>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
return FindPointsEdgeLocal3D_Kernel<4>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsEdgeLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
return FindPointsEdgeLocal3D_Kernel(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
FindPointsEdgeLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
#undef rDIM2
+131 -206
View File
@@ -11,6 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
@@ -51,124 +52,15 @@ struct findptsElementGPT_t
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*(rDIM+1)];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[sDIM], A[sDIM*sDIM];
dbl_range_t x[sDIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[sDIM];
double fac[sDIM];
unsigned int *offset;
};
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x - z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN+i] = 2.0 * lCoeff[i] * u1;
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
const double x[sDIM])
{
double b_d;
for (int d=0; d<sDIM; ++d)
{
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
if (b_d < 0) // if outside in any dimension
{
return b_d;
}
}
return b_d; // only positive if inside in all dimensions
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
const double x[sDIM])
{
const double bxyz = AABB_test(b, x);
if (bxyz<0)
{
return bxyz;
}
else
{
double dxyz[3];
// dxyz: distance of the point from the center of the OBB
for (int d=0; d<sDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
// tranform dxyz to the local coordinate system of the OBB,
// and check if the point is inside the OBB [-1,1]^sDIM
double test = 1;
for (int d=0; d<sDIM; ++d)
{
double rst = 0;
for (int e=0; e<sDIM; ++e)
{
rst += b->A[d*sDIM + e] * dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test<0 ? test : brst;
}
return test;
}
}
/* Hash index in the hash table to the elements that possibly contain the point x */
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[sDIM])
{
const int n = p->hash_n;
int sum = 0;
for (int d=sDIM-1; d>=0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
const double A[3],
const double y[2])
{
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
}
static MFEM_HOST_DEVICE inline double l2norm2(const double x[sDIM])
{
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2]);
}
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<sDIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
using gslib::AABB_test;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_second_der;
using gslib::lin_solve_sym_2;
/* the bit structure of flags is CSSRR
the C bit --- 1<<4 --- is set when the point is converged
@@ -219,18 +111,10 @@ static MFEM_HOST_DEVICE inline int point_index(const int x)
return ((x>>1)&1u) | ((x>>2)&2u);
}
static MFEM_HOST_DEVICE inline findptsElementGEdge_t
static MFEM_HOST_DEVICE inline void
get_edge(const double *elx[3], const double *wtend, int ei,
double *workspace, int &side_init, int jidx, int pN)
int &side_init, int jidx, int pN, findptsElementGEdge_t &edge)
{
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = workspace + d*pN;
edge.dxdn[d] = workspace + sDIM*pN + d*pN;
edge.d2xdn[d] = workspace + 2*sDIM*pN + d*pN;
}
// given edge index, compute normal and tangential directions
const int dn = ei>>1, //0 for rmin/rmax, 1 for smin/smax
de = plus_1_mod_2(dn); // 1 for rmin/rmax, 0 for smin/smax
@@ -256,7 +140,6 @@ get_edge(const double *elx[3], const double *wtend, int ei,
edge.d2xdn[dd][jj] = sums_k[1];
#undef ELX
}
return edge;
}
static MFEM_HOST_DEVICE inline findptsElementGPT_t get_pt(const double *elx[3],
@@ -312,34 +195,34 @@ static MFEM_HOST_DEVICE inline findptsElementGPT_t get_pt(const double *elx[3],
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out->dist2, out->index, out->x, out->oldr in any event,
leaving out->r, out->dr, out->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
const double resid[3],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2(resid);
const double dist2 = l2norm2<sDIM>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d=0; d<sDIM; ++d)
{
out->x[d] = p->x[d];
out_pt->x[d] = p->x[d];
}
for (int d=0; d<rDIM; ++d)
{
out->oldr[d] = p->r[d];
out_pt->oldr[d] = p->r[d];
}
out->dist2 = dist2;
out_pt->dist2 = dist2;
if (decr>=0.01*pred)
{
if (decr>=0.9*pred) // very good iteration
{
out->tr = 2*p->tr;
out_pt->tr = 2*p->tr;
}
else // good iteration
{
out->tr = p->tr;
out_pt->tr = p->tr;
}
return false;
}
@@ -351,17 +234,17 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
which is why we divide by 4 below */
double v0 = fabs(p->r[0] - p->oldr[0]),
v1 = fabs(p->r[1] - p->oldr[1]);
out->tr = ( v0>v1 ? v0 : v1 )/4;
out->dist2 = p->dist2;
out->flags = p->flags >> 5;
out->dist2p = -HUGE_VAL;
out_pt->tr = ( v0>v1 ? v0 : v1 )/4;
out_pt->dist2 = p->dist2;
out_pt->flags = p->flags >> 5;
out_pt->dist2p = -HUGE_VAL;
for (int d=0; d<rDIM; ++d)
{
out->r[d] = p->oldr[d];
out_pt->r[d] = p->oldr[d];
}
if (pred<dist2*tol)
{
out->flags |= CONVERGED_FLAG;
out_pt->flags |= CONVERGED_FLAG;
}
return true;
}
@@ -369,7 +252,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
/* minimize ||resid - jac * dr||_2, with |dr| <= tr, |r0+dr|<=1
(exact solution of trust region problem) */
static MFEM_HOST_DEVICE void newton_face( findptsElementPoint_t *const out,
static MFEM_HOST_DEVICE void newton_face( findptsElementPoint_t *const out_pt,
const double jac[sDIM*rDIM],
const double rhes[3],
const double resid[sDIM],
@@ -540,19 +423,19 @@ newton_face_constrained:
}
newton_face_fin:
out->dist2p = -2*v;
out_pt->dist2p = -2*v;
dr[0] = r[0] - p->r[0];
dr[1] = r[1] - p->r[1];
if ( fabs(dr[0])+fabs(dr[1]) < tol)
{
new_flags |= CONVERGED_FLAG;
}
out->r[0] = r[0], out->r[1] = r[1];
out->flags = new_flags | ((p->flags & FLAG_MASK)<<5);
out_pt->r[0] = r[0], out_pt->r[1] = r[1];
out_pt->flags = new_flags | ((p->flags & FLAG_MASK)<<5);
}
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
out,
out_pt,
const double jac[sDIM*rDIM],
const double rhes,
const double resid[sDIM],
@@ -637,10 +520,10 @@ newton_edge_fin:
{
new_flags |= CONVERGED_FLAG;
}
out->r[de] = nr;
out->r[dn] = p->r[dn];
out->dist2p = -v;
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<5);
out_pt->r[de] = nr;
out_pt->r[dn] = p->r[dn];
out_pt->dist2p = -v;
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<5);
#undef EVAL
}
@@ -676,26 +559,27 @@ static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
// global memory access of element coordinates.
// Are the structs being stored in "local memory" or registers?
template<int T_D1D = 0>
static void FindPointsSurfLocal3D_Kernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
static void FindPointsSurfLocal3DKernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const bool obb_check,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
@@ -753,22 +637,36 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
{
const unsigned int el = *elp;
// construct obbox on the fly
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
bool pass_bb = true;
obbox_t box;
int n_box_ents = 3*sDIM + sDIM2;
for (int idx = 0; idx < sDIM; ++idx)
if (obb_check)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
// construct obbox on the fly
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
pass_bb = (bbox_test(&box, x_i) >= 0);
}
else
{
for (int d = 0; d < sDIM; ++d)
{
box.x[d].min = boxinfo[n_box_ents*el + d];
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
}
pass_bb = (AABB_test(&box, x_i) >= 0);
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
if (bbox_test(&box, x_i) < 0) { continue; }
if (!pass_bb) { continue; }
//// findpts_local ////
{
@@ -968,13 +866,19 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
double *hes_T = jac + sDIM*rDIM;
double *hes = hes_T + hes_count*sDIM;
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
edge.dxdn[d] = constraint_workspace + d*D1D
+ sDIM*D1D;
edge.d2xdn[d] = constraint_workspace + d*D1D
+ 2*sDIM*D1D;
}
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
{
// utilized first D1D threads
edge = get_edge(elx, wtend, ei,
constraint_workspace, edge_init, j,
D1D);
// One thread per physical component and edge DOF.
get_edge(elx, wtend, ei, edge_init, j, D1D, edge);
}
MFEM_SYNC_THREAD;
@@ -1045,7 +949,15 @@ static void FindPointsSurfLocal3D_Kernel(const int npt,
steep *= tmp->r[dn];
if (steep<0)
{
newton_face( fpt,jac,hes,resid,tmp->flags&CONVERGED_FLAG,tmp,tol);
double face_hes[3] =
{
dn == 0 ? hes[2] : hes[0],
hes[1],
dn == 0 ? hes[0] : hes[2]
};
newton_face(fpt, jac, face_hes, resid,
tmp->flags & CONVERGED_FLAG,
tmp, tol);
}
else
{
@@ -1211,29 +1123,42 @@ void FindPointsGSLIB::FindPointsSurfLocal3(const Vector &point_pos,
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
const bool obb_chk = obb_check;
switch (DEV.dof1d)
{
case 2:
return FindPointsSurfLocal3D_Kernel<2>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsSurfLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
return FindPointsSurfLocal3D_Kernel<3>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsSurfLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
return FindPointsSurfLocal3D_Kernel<4>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
FindPointsSurfLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
return FindPointsSurfLocal3D_Kernel(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
FindPointsSurfLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
+190
View File
@@ -0,0 +1,190 @@
#ifndef MFEM_GSLIB_KERNEL_HELPERS_HPP
#define MFEM_GSLIB_KERNEL_HELPERS_HPP
#include "../../config/config.hpp"
#include <cmath>
namespace mfem
{
namespace gslib
{
struct dbl_range_t
{
double min, max;
};
template <int SDIM>
struct obbox_t
{
double c0[SDIM], A[SDIM * SDIM];
dbl_range_t x[SDIM];
};
template <int SDIM>
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[SDIM];
double fac[SDIM];
unsigned int *offset;
};
// Eval the ith Lagrange interpolant at x.
MFEM_HOST_DEVICE inline void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
const double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
// Eval the ith Lagrange interpolant and its first derivative at x.
MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
const double d_j = 2 * (x - z[j]);
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN + i] = 2.0 * lCoeff[i] * u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
const double d_j = 2 * (x - z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN + i] = 2.0 * lCoeff[i] * u1;
p0[2 * pN + i] = 8.0 * lCoeff[i] * u2;
}
// Solve Ax=y where A is a symmetric 2x2 matrix packed as {a00, a01, a11}.
MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
const double A[3],
const double y[2])
{
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
}
// Positive when the point is inside the axis-aligned bounding box.
template <int SDIM>
MFEM_HOST_DEVICE inline double AABB_test(const obbox_t<SDIM> *const b,
const double (&x)[SDIM])
{
double test = 1.0;
for (int d = 0; d < SDIM; ++d)
{
const double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
test = test < 0.0 ? test : b_d;
}
return test;
}
// Positive when the point is inside the oriented bounding box.
template <int SDIM>
MFEM_HOST_DEVICE inline double bbox_test(const obbox_t<SDIM> *const b,
const double (&x)[SDIM])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0.0)
{
return bxyz;
}
double dxyz[SDIM];
for (int d = 0; d < SDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1.0;
for (int d = 0; d < SDIM; ++d)
{
double rst = 0.0;
for (int e = 0; e < SDIM; ++e)
{
rst += b->A[d * SDIM + e] * dxyz[e];
}
const double brst = (rst + 1.0) * (1.0 - rst);
test = test < 0.0 ? test : brst;
}
return test;
}
// Hash index in the hash table for the point x.
template <int SDIM>
MFEM_HOST_DEVICE inline int hash_index(
const findptsLocalHashData_t<SDIM> *const p,
const double (&x)[SDIM])
{
const int n = p->hash_n;
int sum = 0;
for (int d = SDIM - 1; d >= 0; --d)
{
sum *= n;
const int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
}
return sum;
}
// Squared Euclidean norm.
template <int SDIM>
MFEM_HOST_DEVICE inline double l2norm2(const double (&x)[SDIM])
{
double sum = 0.0;
for (int d = 0; d < SDIM; ++d)
{
sum += x[d] * x[d];
}
return sum;
}
template <int SDIM>
MFEM_HOST_DEVICE inline double l2norm2(const double *x)
{
double sum = 0.0;
for (int d = 0; d < SDIM; ++d)
{
sum += x[d] * x[d];
}
return sum;
}
} // namespace gslib
} // namespace mfem
#endif
+22 -27
View File
@@ -11,7 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/kernels.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -33,17 +33,7 @@ namespace mfem
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j=0; j<p_Nq; ++j)
{
p_i *= j==i ? 1 : x-z[j];
}
p0[i] = lagrangeCoeff[i] * p_i;
}
using gslib::lagrange_eval;
template<int T_D1D = 0>
static void InterpolateLocal1DKernel(const double *const gf_in,
@@ -123,21 +113,26 @@ void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf, dof1Dsol);
case 2:
InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
}
}
#undef CODE_INTERNAL
+22 -27
View File
@@ -11,6 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -32,18 +33,7 @@ namespace mfem
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
using gslib::lagrange_eval;
template<int T_D1D = 0>
static void InterpolateLocal2DKernel(const double *const gf_in,
@@ -132,21 +122,26 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf, dof1Dsol);
case 2:
InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
}
}
+22 -27
View File
@@ -11,6 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -32,18 +33,7 @@ namespace mfem
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
using gslib::lagrange_eval;
template<int T_D1D = 0>
static void InterpolateLocal3DKernel(const double *const gf_in,
@@ -135,21 +125,26 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
npt, ncomp,
pgll, plcf, dof1Dsol);
case 2:
InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
}
}
@@ -10,6 +10,7 @@
// CONTRIBUTING.md for details.
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -19,6 +20,13 @@ namespace mfem
DiffusionIntegrator::Kernels::Kernels()
{
// 2D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<2,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<2,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<2,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<2,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,7,6>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<2,1,1>();
DiffusionIntegrator::AddSpecialization<2,2,2>();
@@ -40,7 +48,18 @@ DiffusionIntegrator::Kernels::Kernels()
DiffusionIntegrator::AddSpecialization<2,8,9>();
DiffusionIntegrator::AddSpecialization<2,9,10>();
// others
DiffusionIntegrator::AddSimplexSpecialization<2,2,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,6>();
// 3D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<3,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<3,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<3,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<3,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<3,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<3,7,6>();
DiffusionIntegrator::AddSimplexSpecialization<3,8,7>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<3,1,1>();
DiffusionIntegrator::AddSpecialization<3,2,2>();
+21 -16
View File
@@ -12,7 +12,6 @@
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#include "../kernel_dispatch.hpp"
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
@@ -20,6 +19,8 @@
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp"
namespace mfem
{
@@ -637,8 +638,8 @@ inline void SmemPADiffusionApply2D(const int NE,
const bool symmetric,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Array<real_t> &bt_,
const Array<real_t> &gt_,
const Array<real_t> &,
const Array<real_t> &,
const Vector &d_,
const Vector &x_,
Vector &y_,
@@ -1218,43 +1219,47 @@ inline void SmemPADiffusionApply3D(const int NE,
namespace
{
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
using ApplySimplexKernelType = DiffusionIntegrator::ApplySimplexKernelType;
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
}
inline
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
{
if (DIM == 2) { return internal::PADiffusionApply2D; }
else if (DIM == 3) { return internal::PADiffusionApply3D; }
if (dim == 2) { return internal::PADiffusionApply2D; }
else if (dim == 3) { return internal::PADiffusionApply3D; }
else { MFEM_ABORT(""); }
}
template<int DIM, int D1D, int Q1D>
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
MFEM_ABORT("");
else { MFEM_ABORT(""); }
return nullptr;
}
inline DiagonalKernelType
DiffusionIntegrator::DiagonalPAKernels::Fallback(int DIM, int, int)
DiffusionIntegrator::DiagonalPAKernels::Fallback(int dim, int, int)
{
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
if (dim == 2) { return internal::PADiffusionDiagonal2D; }
else if (dim == 3) { return internal::PADiffusionDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+31 -2
View File
@@ -15,6 +15,7 @@
#include "../../mesh/nurbs.hpp"
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp"
namespace mfem
{
@@ -68,6 +69,24 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
#endif // MFEM_USE_OCCA
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
return ApplySimplexPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
rmaps->lex_map,
rmaps->forward_map2d_diff,
rmaps->inverse_map2d_diff,
rmaps->forward_map3d_diff,
rmaps->inverse_map3d_diff,
rmaps->Ga1,
rmaps->Ga2,
rmaps->Ga3,
rmaps->Ga1t,
rmaps->Ga2t,
rmaps->Ga3t,
Dv, x, y, dofs1D, quad1D);
}
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
Gt, Dv, x, y, dofs1D, quad1D);
}
@@ -94,7 +113,8 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
fespace = &fes;
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, stroud);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -119,13 +139,22 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
dim = mesh->Dimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
const int sdim = mesh->SpaceDimension();
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
if (MQ) { coeff.ProjectTranspose(*MQ); }
else if (VQ) { coeff.Project(*VQ); }
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -91,15 +91,15 @@ void ElasticityAddMultPA(const int dim, const int nDofs,
void ElasticityAssembleDiagonalPA(const int dim, const int nDofs,
const CoefficientVector &lambda,
const CoefficientVector &mu, const GeometricFactors &geom,
const DofToQuad &maps, QuadratureFunction &QVec, Vector &diag)
const DofToQuad &maps, const IntegrationRule &ir, Vector &diag)
{
switch (dim)
{
case 2:
ElasticityAssembleDiagonalPA_<2>(nDofs, lambda, mu, geom, maps, QVec, diag);
ElasticityAssembleDiagonalPA_<2>(nDofs, lambda, mu, geom, maps, ir, diag);
break;
case 3:
ElasticityAssembleDiagonalPA_<3>(nDofs, lambda, mu, geom, maps, QVec, diag);
ElasticityAssembleDiagonalPA_<3>(nDofs, lambda, mu, geom, maps, ir, diag);
break;
default:
MFEM_ABORT("Only dimensions 2 and 3 supported.");
+44 -55
View File
@@ -38,7 +38,6 @@
#include "../../linalg/vector.hpp"
#include "../../linalg/tensor.hpp"
#include "../quadinterpolator.hpp"
#include "../bilininteg.hpp"
#include "../coefficient.hpp"
#include "../qfunction.hpp"
@@ -133,12 +132,12 @@ void ElasticityAssembleEA(const int dim, const int i_block, const int j_block,
/// @param[in] mu Quadrature function for second Lame param.
/// @param[in] geom Geometric factors corresponding to fespace.
/// @param[in] maps DofToQuad maps for one element (assume elements all same).
/// @param QVec Scratch Q-Vector. nQuad x dim x dim x dim x dim x numEls.
/// @param[in] ir Integration rule.
/// @param[out] diag diagonal of A. nDofs x dim x numEls.
void ElasticityAssembleDiagonalPA(const int dim, const int nDofs,
const CoefficientVector &lambda,
const CoefficientVector &mu, const GeometricFactors &geom,
const DofToQuad &maps, QuadratureFunction &QVec, Vector &diag);
const DofToQuad &maps, const IntegrationRule &ir, Vector &diag);
/// Templated implementation of ElasticityAddMultPA.
template<int dim, int i_block = -1, int j_block = -1>
@@ -280,77 +279,67 @@ void ElasticityAddMultPA_(const int nDofs, const FiniteElementSpace &fespace,
template<int dim>
void ElasticityAssembleDiagonalPA_(const int nDofs,
const CoefficientVector &lambda,
const CoefficientVector &mu, const GeometricFactors &geom,
const DofToQuad &maps, QuadratureFunction &QVec, Vector &diag)
const CoefficientVector &mu,
const GeometricFactors &geom,
const DofToQuad &maps,
const IntegrationRule &ir,
Vector &diag)
{
using future::tensor;
using future::make_tensor;
using future::det;
using future::inv;
using future::make_tensor;
using future::tensor;
// Assuming all elements are the same
const auto &ir = QVec.GetIntRule(0);
static constexpr int d = dim;
const int numPoints = ir.GetNPoints();
const int numEls = lambda.Size()/numPoints;
const int numEls = lambda.Size() / numPoints;
const auto lamDev = Reshape(lambda.Read(), numPoints, numEls);
const auto muDev = Reshape(mu.Read(), numPoints, numEls);
const auto J = Reshape(geom.J.Read(), numPoints, d, d, numEls);
auto Q = Reshape(QVec.ReadWrite(), numPoints, d,d, d, numEls);
const real_t *ipWeights = ir.GetWeights().Read();
mfem::forall_2D(numEls, numPoints,1, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(p, x,numPoints)
{
auto invJ = inv(make_tensor<d, d>(
[&](int i, int j) { return J(p, i, j, e); }));
const real_t w = ipWeights[p] /det(invJ);
for (int n = 0; n < d; n++)
{
for (int m = 0; m < d; m++)
{
for (int q = 0; q < d; q++)
{
// compute contraction of 4*sym(grad(u))sym(grad(v)) term.
// this contraction could be made slightly cheaper using Voigt
// notation, but repeated entries are summed for simplicity.
real_t contraction = 0.;
for (int a = 0; a < d; a++)
{
for (int b = 0; b < d; b++)
{
contraction += ((a == q)*invJ(m,b) + (b==q)*invJ(m,a))*((a == q)
*invJ(n, b) + (b==q)*invJ(n,a));
}
}
// lambda*div(u)*div(v) + 2*mu*sym(grad(u))*sym(grad(v))
// contraction = 4*sym(grad(u))sym(grad(v))
Q(p,m,n,q,e) = w*(lamDev(p, e)*invJ(m,q)*invJ(n,q)
+ 0.5*muDev(p, e)*contraction);
}
}
}
}
});
// Reduce quadrature function to an E-Vector
const auto QRead = Reshape(QVec.Read(), numPoints, d, d, d, numEls);
auto diagDev = Reshape(diag.Write(), nDofs, d, numEls);
const auto G = Reshape(maps.G.Read(), numPoints, d, nDofs);
auto diagDev = Reshape(diag.Write(), nDofs, d, numEls);
mfem::forall_2D(numEls, d, nDofs, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(i, y, nDofs)
MFEM_FOREACH_THREAD_DIRECT(i, y, nDofs)
{
MFEM_FOREACH_THREAD(q, x, d)
MFEM_FOREACH_THREAD_DIRECT(q, x, d)
{
real_t sum = 0.;
for (int n = 0; n < d; n++)
real_t sum = 0.0;
for (int p = 0; p < numPoints; p++)
{
for (int m = 0; m < d; m++)
const auto invJ = inv(make_tensor<d, d>([&](int r, int c)
{
for (int p = 0; p < numPoints; p++ )
return J(p, r, c, e);
}));
const real_t w = ipWeights[p] / det(invJ);
for (int n = 0; n < d; n++)
{
for (int m = 0; m < d; m++)
{
sum += QRead(p,m,n,q,e)*G(p,m,i)*G(p,n,i);
// compute contraction of 4*sym(grad(u))sym(grad(v)) term.
// this contraction could be made slightly cheaper using Voigt
// notation, but repeated entries are summed for simplicity.
real_t contraction = 0.0;
for (int a = 0; a < d; a++)
{
for (int b = 0; b < d; b++)
{
contraction +=
((a == q) * invJ(m, b) + (b == q) * invJ(m, a)) *
((a == q) * invJ(n, b) + (b == q) * invJ(n, a));
}
}
// lambda*div(u)*div(v) + 2*mu*sym(grad(u))*sym(grad(v))
// contraction = 4*sym(grad(u))sym(grad(v))
const real_t Q =
w * (lamDev(p, e) * invJ(m, q) * invJ(n, q)
+ 0.5 * muDev(p, e) * contraction);
sum += Q * G(p, m, i) * G(p, n, i);
}
}
}
+1 -3
View File
@@ -10,7 +10,6 @@
// CONTRIBUTING.md for details.
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
#include "bilininteg_elasticity_kernels.hpp"
@@ -59,9 +58,8 @@ void ElasticityIntegrator::AssemblePA(const FiniteElementSpace &fes)
void ElasticityIntegrator::AssembleDiagonalPA(Vector &diag)
{
q_vec->SetVDim(vdim*vdim*vdim*vdim);
internal::ElasticityAssembleDiagonalPA(vdim, ndofs, *lambda_quad, *mu_quad,
*geom, *maps, *q_vec, diag);
*geom, *maps, *IntRule, diag);
}
void ElasticityIntegrator::AddMultPA(const Vector &x, Vector &y) const
+3
View File
@@ -10,6 +10,7 @@
// CONTRIBUTING.md for details.
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -39,8 +40,10 @@ MassIntegrator::Kernels::Kernels()
MassIntegrator::AddSpecialization<2,9,10>();
// others
MassIntegrator::AddSpecialization<2,2,4>();
MassIntegrator::AddSpecialization<2,2,5>();
MassIntegrator::AddSpecialization<2,3,6>();
MassIntegrator::AddSpecialization<2,4,6>();
// 3D
// Q=P+1
MassIntegrator::AddSpecialization<3,1,1>();
+25 -17
View File
@@ -19,6 +19,8 @@
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "bilininteg_mass_pa_simplices.hpp"
namespace mfem
{
@@ -1408,51 +1410,57 @@ using ApplyKernelType = MassIntegrator::ApplyKernelType;
using DiagonalKernelType = MassIntegrator::DiagonalKernelType;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3)
{
constexpr int MDQ = T_D1D >= T_Q1D ? T_D1D : T_Q1D;
constexpr int MDQ = D1D >= Q1D ? D1D : Q1D;
// max 64 threads in z limit in cuda and hip
if constexpr (MDQ > 0)
{
return internal::SmemPAMassApply3D<T_D1D, T_Q1D,
return internal::SmemPAMassApply3D<D1D, Q1D,
internal::mass::NBZ3D(MDQ)>;
}
}
MFEM_ABORT("");
else { MFEM_ABORT(""); }
return nullptr;
}
inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
int DIM, int, int)
int dim, int, int)
{
if (DIM == 1) { return internal::PAMassApply1D; }
else if (DIM == 2) { return internal::PAMassApply2D; }
else if (DIM == 3) { return internal::PAMassApply3D; }
if (dim == 1) { return internal::PAMassApply1D; }
else if (dim == 2) { return internal::PAMassApply2D; }
else if (dim == 3) { return internal::PAMassApply3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
DiagonalKernelType MassIntegrator::DiagonalPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
}
inline DiagonalKernelType MassIntegrator::DiagonalPAKernels::Fallback(
int DIM, int, int)
int dim, int, int)
{
if (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (DIM == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (DIM == 3) { return internal::PAMassAssembleDiagonal3D; }
if (dim == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (dim == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (dim == 3) { return internal::PAMassAssembleDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+43 -5
View File
@@ -15,6 +15,7 @@
#include "../qfunction.hpp"
#include "../ceed/integrators/mass/mass.hpp"
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp"
namespace mfem
{
@@ -29,9 +30,11 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
dim = mesh->Dimension();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation *T0 = mesh->GetTypicalElementTransformation();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0);
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0, stroud);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -48,17 +51,25 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
return;
}
int map_type = el.GetMapType();
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::DETERMINANTS, mt);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, mt);
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
{
const int NE = ne;
const int NQ = nq;
@@ -147,9 +158,10 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const int D1D = dofs1D;
const int Q1D = quad1D;
const Vector &D = pa_data;
const Array<real_t> &B = maps->B;
const Array<real_t> &Bt = maps->Bt;
const Vector &D = pa_data;
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
@@ -164,7 +176,31 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
}
#endif // MFEM_USE_OCCA
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
const Array<real_t> &Ba1 = rmaps->Ba1;
const Array<real_t> &Ba2 = rmaps->Ba2;
const Array<real_t> &Ba3 = rmaps->Ba3;
const Array<real_t> &Ba1t = rmaps->Ba1t;
const Array<real_t> &Ba2t = rmaps->Ba2t;
const Array<real_t> &Ba3t = rmaps->Ba3t;
const Array<int> &lex_map = rmaps->lex_map;
const Array<int> &forward_map2d = rmaps->forward_map2d_mass;
const Array<int> &inverse_map2d = rmaps->inverse_map2d_mass;
const Array<int> &forward_map3d = rmaps->forward_map3d_mass;
const Array<int> &inverse_map3d = rmaps->inverse_map3d_mass;
ApplySimplexPAKernels::Run(dim, D1D, Q1D, ne, lex_map, forward_map2d,
inverse_map2d,
forward_map3d, inverse_map3d, Ba1, Ba2, Ba3, Ba1t, Ba2t, Ba3t,
D, x, y, D1D, Q1D);
}
else
{
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
}
}
}
@@ -177,6 +213,8 @@ void MassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
}
else
{
MFEM_VERIFY(!fespace->UsesRaggedTensorBasis(),
"AbsMultPA not implemented for ragged tensor basis");
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absB(maps->B);
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+365
View File
@@ -0,0 +1,365 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "../kernels.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
// Shared memory PA Divergence Apply 2D kernel
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
inline void SmemPADivergenceApply2D(const int NE,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Array<real_t> &bt_,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = b_.Read(), G = g_.Read(), Bt = bt_.Read();
const auto Q = Reshape(q_.Read(), Q1D, Q1D, 2, 2, NE);
const auto X = Reshape(x_.Read(), TR_D1D, TR_D1D, 2, NE);
auto Y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, 1, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
kernels::internal::vd_regs2d_t<2, 2, MQ1> g0, g1;
kernels::internal::v_regs2d_t<1, MQ1> r0, r1;
kernels::internal::LoadMatrix(TR_D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(TR_D1D, Q1D, G, sG);
kernels::internal::LoadDofs2d(e, TR_D1D, X, g0);
kernels::internal::Grad2d(TR_D1D, Q1D, smem, sB, sG, g0, g1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
r0[0][qy][qx] =
g1[0][0][qy][qx] * Q(qx, qy, 0, 0, e) +
g1[0][1][qy][qx] * Q(qx, qy, 1, 0, e) +
g1[1][0][qy][qx] * Q(qx, qy, 0, 1, e) +
g1[1][1][qy][qx] * Q(qx, qy, 1, 1, e);
}
}
MFEM_SYNC_THREAD;
kernels::internal::LoadMatrix<MQ1,true>(TE_D1D, Q1D, Bt, sB);
kernels::internal::EvalTranspose2d(TE_D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs2d(e, TE_D1D, r1, Y);
});
}
// Shared memory PA Divergence Apply 2D kernel transpose
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
inline void SmemPADivergenceApplyTranspose2D(const int NE,
const Array<real_t> &bt,
const Array<real_t> &gt,
const Array<real_t> &b,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto Bt = bt.Read(), Gt = gt.Read(), B = b.Read();
const auto Q = Reshape(q_.Read(), Q1D, Q1D, 2, 2, NE);
const auto X = Reshape(x_.Read(), TE_D1D, TE_D1D, 1, NE);
auto Y = Reshape(y_.ReadWrite(), TR_D1D, TR_D1D, 2, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
kernels::internal::v_regs2d_t<1, MQ1> r0, r1;
kernels::internal::vd_regs2d_t<2, 2, MQ1> g0, g1;
kernels::internal::LoadMatrix(TE_D1D, Q1D, B, sB);
kernels::internal::LoadDofs2d(e, TE_D1D, X, r0);
kernels::internal::Eval2d(TE_D1D, Q1D, smem, sB, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
g0[0][0][qy][qx] = r1[0][qy][qx] * Q(qx, qy, 0, 0, e);
g0[0][1][qy][qx] = r1[0][qy][qx] * Q(qx, qy, 1, 0, e);
g0[1][0][qy][qx] = r1[0][qy][qx] * Q(qx, qy, 0, 1, e);
g0[1][1][qy][qx] = r1[0][qy][qx] * Q(qx, qy, 1, 1, e);
}
}
MFEM_SYNC_THREAD;
kernels::internal::LoadMatrix<MQ1,true>(TR_D1D, Q1D, Bt, sB);
kernels::internal::LoadMatrix<MQ1,true>(TR_D1D, Q1D, Gt, sG);
kernels::internal::GradTranspose2d(TR_D1D, Q1D, smem, sB, sG, g0, g1);
kernels::internal::WriteDofs2d(e, TR_D1D, g1, Y);
});
}
// Shared memory PA Divergence Apply 3D kernel transpose
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
inline void SmemPADivergenceApplyTranspose3D(const int NE,
const Array<real_t> &bt,
const Array<real_t> &gt,
const Array<real_t> &b,
const Vector &q_,
const Vector &x_,
Vector &y_,
int tr_d1d = 0,
int te_d1d = 0,
int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto Bt = bt.Read(), Gt = gt.Read(), B = b.Read();
const auto Q = Reshape(q_.Read(), Q1D, Q1D, Q1D, 3, 3, NE);
const auto X = Reshape(x_.Read(), TE_D1D, TE_D1D, TE_D1D, 1, NE);
auto Y = Reshape(y_.ReadWrite(), TR_D1D, TR_D1D, TR_D1D, 3, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
kernels::internal::v_regs3d_t<1, MQ1> r0, r1;
kernels::internal::vd_regs3d_t<3, 3, MQ1> g0, g1;
kernels::internal::LoadMatrix(TE_D1D, Q1D, B, sB);
kernels::internal::LoadDofs3d(e, TE_D1D, X, r0);
kernels::internal::Eval3d(TE_D1D, Q1D, smem, sB, r0, r1);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const auto r = r1[0][qz][qy][qx];
g0[0][0][qz][qy][qx] = r * Q(qx, qy, qz, 0, 0, e);
g0[0][1][qz][qy][qx] = r * Q(qx, qy, qz, 1, 0, e);
g0[0][2][qz][qy][qx] = r * Q(qx, qy, qz, 2, 0, e);
g0[1][0][qz][qy][qx] = r * Q(qx, qy, qz, 0, 1, e);
g0[1][1][qz][qy][qx] = r * Q(qx, qy, qz, 1, 1, e);
g0[1][2][qz][qy][qx] = r * Q(qx, qy, qz, 2, 1, e);
g0[2][0][qz][qy][qx] = r * Q(qx, qy, qz, 0, 2, e);
g0[2][1][qz][qy][qx] = r * Q(qx, qy, qz, 1, 2, e);
g0[2][2][qz][qy][qx] = r * Q(qx, qy, qz, 2, 2, e);
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::LoadMatrix<MQ1,true>(TR_D1D, Q1D, Bt, sB);
kernels::internal::LoadMatrix<MQ1,true>(TR_D1D, Q1D, Gt, sG);
kernels::internal::GradTranspose3d(TR_D1D, Q1D, smem, sB, sG, g0, g1);
kernels::internal::WriteDofs3d(e, TR_D1D, g1, Y);
});
}
// Shared memory PA Divergence Apply 3D kernel
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
inline void SmemPADivergenceApply3D(const int NE,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Array<real_t> &bt_,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = b_.Read(), G = g_.Read(), Bt = bt_.Read();
const auto Q = Reshape(q_.Read(), Q1D, Q1D, Q1D, 3,3, NE);
const auto X = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, 3, NE);
auto Y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 1, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> g0, g1;
kernels::internal::v_regs3d_t<1, MQ1> r0, r1;
kernels::internal::LoadMatrix(TR_D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(TR_D1D, Q1D, G, sG);
kernels::internal::LoadDofs3d(e, TR_D1D, X, g0);
kernels::internal::Grad3d(TR_D1D, Q1D, smem, sB, sG, g0, g1);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
r0[0][qz][qy][qx] =
// c = 0
g1[0][0][qz][qy][qx] * Q(qx, qy, qz, 0, 0, e) +
g1[0][1][qz][qy][qx] * Q(qx, qy, qz, 1, 0, e) +
g1[0][2][qz][qy][qx] * Q(qx, qy, qz, 2, 0, e) +
// c = 1
g1[1][0][qz][qy][qx] * Q(qx, qy, qz, 0, 1, e) +
g1[1][1][qz][qy][qx] * Q(qx, qy, qz, 1, 1, e) +
g1[1][2][qz][qy][qx] * Q(qx, qy, qz, 2, 1, e) +
// c = 2
g1[2][0][qz][qy][qx] * Q(qx, qy, qz, 0, 2, e) +
g1[2][1][qz][qy][qx] * Q(qx, qy, qz, 1, 2, e) +
g1[2][2][qz][qy][qx] * Q(qx, qy, qz, 2, 2, e);
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::LoadMatrix<MQ1, true>(TE_D1D, Q1D, Bt, sB);
kernels::internal::EvalTranspose3d(TE_D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs3d(e, TE_D1D, r1, Y);
});
}
} // namespace internal
template<int DIM, int T_TR_D1D, int T_TE_D1D, int T_Q1D>
VectorDivergenceIntegrator::VectorDivergenceAddMultPAType
VectorDivergenceIntegrator::VectorDivergenceAddMultPA::Kernel()
{
static_assert(T_TR_D1D <= T_Q1D && T_TE_D1D <= T_Q1D);
if constexpr (DIM == 2)
{
return internal::SmemPADivergenceApply2D<T_TR_D1D, T_TE_D1D, T_Q1D>;
}
else if constexpr (DIM == 3)
{
return internal::SmemPADivergenceApply3D<T_TR_D1D, T_TE_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorDivergenceIntegrator::VectorDivergenceAddMultPAType
VectorDivergenceIntegrator::VectorDivergenceAddMultPA::Fallback
(int dim, int tr_d1d, int te_d1d, int q1d)
{
MFEM_VERIFY(tr_d1d <= q1d && te_d1d <= q1d, "");
MFEM_VERIFY(tr_d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(te_d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
if (dim == 2)
{
return internal::SmemPADivergenceApply2D;
}
else if (dim == 3)
{
return internal::SmemPADivergenceApply3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
template<int DIM, int T_TR_D1D, int T_TE_D1D, int T_Q1D>
VectorDivergenceIntegrator::VectorDivergenceAddMultTransposePAType
VectorDivergenceIntegrator::VectorDivergenceAddMultTransposePA::Kernel()
{
static_assert(T_TR_D1D <= T_Q1D && T_TE_D1D <= T_Q1D);
if constexpr (DIM == 2)
{
return internal::SmemPADivergenceApplyTranspose2D<T_TR_D1D, T_TE_D1D, T_Q1D>;
}
else if constexpr (DIM == 3)
{
return internal::SmemPADivergenceApplyTranspose3D<T_TR_D1D, T_TE_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorDivergenceIntegrator::VectorDivergenceAddMultTransposePAType
VectorDivergenceIntegrator::VectorDivergenceAddMultTransposePA::Fallback
(int dim, int tr_d1d, int te_d1d, int q1d)
{
MFEM_VERIFY(tr_d1d <= q1d && te_d1d <= q1d, "");
MFEM_VERIFY(tr_d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(te_d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
if (dim == 2)
{
return internal::SmemPADivergenceApplyTranspose2D;
}
else if (dim == 3)
{
return internal::SmemPADivergenceApplyTranspose3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+32 -149
View File
@@ -205,157 +205,40 @@ void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
template <const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal2D(const int NE,
const Array<real_t> &b,
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
{
constexpr int VDIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t temp[max_Q1D][max_D1D];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dy = 0; dy < D1D; ++dy)
{
temp[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
}
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
real_t temp1 = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
}
Y(dx, dy, 0, e) = temp1;
Y(dx, dy, 1, e) = temp1;
}
}
});
}
template <const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal3D(const int NE,
const Array<real_t> &B_,
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(B_.Read(), Q1D, D1D);
MFEM_VERIFY(pa_data.Size() == Q1D * Q1D * Q1D * NE, "pa_data size error");
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t temp[max_Q1D][max_Q1D][max_D1D];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int dz = 0; dz < D1D; ++dz)
{
temp[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
temp[qx][qy][dz] +=
B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
}
}
}
}
real_t temp2[max_Q1D][max_D1D][max_D1D];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
temp2[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp2[qx][dy][dz] +=
B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
}
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
real_t temp3 = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
temp3 += B(qx, dx) * B(qx, dx) * temp2[qx][dy][dz];
}
Y(dx, dy, dz, 0, e) = temp3;
Y(dx, dy, dz, 1, e) = temp3;
Y(dx, dy, dz, 2, e) = temp3;
}
}
}
});
}
static void PAVectorMassAssembleDiagonal(const int dim, const int D1D,
const int Q1D, const int NE,
const Array<real_t> &B,
const Vector &pa_data,
Vector &diag)
{
if (dim == 2)
{
return PAVectorMassAssembleDiagonal2D(NE, B, pa_data, diag, D1D, Q1D);
}
else if (dim == 3)
{
return PAVectorMassAssembleDiagonal3D(NE, B, pa_data, diag, D1D, Q1D);
}
MFEM_ABORT("Dimension not implemented.");
}
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (DeviceCanUseCeed()) { ceedOp->GetDiagonal(diag); }
else
{
MFEM_VERIFY(coeff_vdim == 1, "coeff_vdim != 1");
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported");
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
if (DeviceCanUseCeed()) { return ceedOp->GetDiagonal(diag); }
MFEM_VERIFY(coeff_vdim == 1, "coeff_vdim != 1");
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported");
// Add the VectorMassAssembleDiagonalPA specializations
static const auto vector_mass_assemble_diagonal_kernel_specializations =
( // 2D
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 2>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 3>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 4>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 5>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 6>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 7>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<2, 8>::Add(),
// 3D
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 2>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 3>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 4>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 5>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 6>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 7>::Add(),
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Specialization<3, 8>::Add(),
true);
MFEM_CONTRACT_VAR(vector_mass_assemble_diagonal_kernel_specializations);
VectorMassAssembleDiagonalPA::Run(dim, quad1D, // templated arguments
ne, dofs1D, quad1D,
maps->B.Read(),
pa_data.Read(),
diag.ReadWrite());
}
} // namespace mfem
+170 -2
View File
@@ -176,8 +176,146 @@ void SmemPAVectorMassApply3D(const int NE,
});
}
template <int T_Q1D = 0, int T_MDQ = 16>
static void SmemPAVectorMassAssembleDiagonal2D(const int ne,
const int d1d,
const int q1d,
const real_t *b_r,
const real_t *d_r,
real_t *y_rw)
{
constexpr int VDIM = 2;
const int D1D = d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(Q1D <= T_MDQ && D1D <= Q1D, "");
const auto B = Reshape(b_r, Q1D, D1D);
const auto D = Reshape(d_r, Q1D, Q1D, ne);
auto Y = Reshape(y_rw, D1D, D1D, VDIM, ne);
mfem::forall_2D<T_Q1D*T_Q1D>(
ne, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MDQ;
MFEM_SHARED real_t sm[MQ1][MQ1];
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
real_t u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
}
sm[qx][dy] = u;
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += B(qx, dx) * B(qx, dx) * sm[qx][dy];
}
Y(dx, dy, 0, e) += u;
Y(dx, dy, 1, e) += u;
}
}
});
}
// T_MDQ <= 10 so the Q1D^3 thread block stays within the 1024/block GPU limit
template <int T_Q1D = 0, int T_MDQ = 10>
static void SmemPAVectorMassAssembleDiagonal3D(const int ne,
const int d1d,
const int q1d,
const real_t *b_r,
const real_t *d_r,
real_t *y_rw)
{
constexpr int VDIM = 3;
const int D1D = d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(Q1D <= T_MDQ && D1D <= Q1D, "");
const auto B = Reshape(b_r, Q1D, D1D);
const auto D = Reshape(d_r, Q1D, Q1D, Q1D, ne);
auto Y = Reshape(y_rw, D1D, D1D, D1D, VDIM, ne);
mfem::forall_3D<T_Q1D*T_Q1D*T_Q1D>(
ne, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MDQ;
MFEM_SHARED real_t sm[2][MQ1][MQ1][MQ1];
MFEM_FOREACH_THREAD_DIRECT(dz, z, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
}
sm[0][dz][qy][qx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dz, z, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
real_t u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += B(qy, dy) * B(qy, dy) * sm[0][dz][qy][qx];
}
sm[1][dz][dy][qx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dz, z, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += B(qx, dx) * B(qx, dx) * sm[1][dz][dy][qx];
}
Y(dx, dy, dz, 0, e) += u;
Y(dx, dy, dz, 1, e) += u;
Y(dx, dy, dz, 2, e) += u;
}
}
}
});
}
} // namespace internal
// AddMultPA kernels
template<int DIM, int T_D1D, int T_Q1D>
VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Kernel()
@@ -190,11 +328,11 @@ VectorMassIntegrator::VectorMassAddMultPA::Kernel()
{
return internal::SmemPAVectorMassApply3D<T_D1D, T_Q1D>;
}
MFEM_ABORT("Unsupported kernel");
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int d1d, int q1d)
VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int, int)
{
if (dim == 2)
{
@@ -207,6 +345,36 @@ VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int d1d, int q1d)
else { MFEM_ABORT("Unsupported kernel"); }
}
// DiagonalPA kernels
template<int DIM, int T_Q1D>
VectorMassIntegrator::VectorMassAssembleDiagonalPAType
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Kernel()
{
if constexpr (DIM == 2)
{
return internal::SmemPAVectorMassAssembleDiagonal2D<T_Q1D>;
}
else if constexpr (DIM == 3)
{
return internal::SmemPAVectorMassAssembleDiagonal3D<T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorMassIntegrator::VectorMassAssembleDiagonalPAType
VectorMassIntegrator::VectorMassAssembleDiagonalPA::Fallback(int dim, int)
{
if (dim == 2)
{
return internal::SmemPAVectorMassAssembleDiagonal2D;
}
else if (dim == 3)
{
return internal::SmemPAVectorMassAssembleDiagonal3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+500
View File
@@ -307,6 +307,506 @@ DomainLFIntegrator::AssembleKernels::Kernel()
MFEM_ABORT("");
}
template <int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble2D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 2 * (d - 1) * d * ne, "");
constexpr int vdim = 2;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int vdim = 2;
if (M[e] == 0) { return; } // ignore
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
MFEM_SHARED real_t sQQ[vdim*Q*Q];
MFEM_SHARED real_t sQD[vdim*Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
const DeviceCube QQ(sQQ, q, q, vdim);
const DeviceCube QD(sQD, q, d, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const real_t cst_val_0 = C(0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
const real_t J0 = J(x,y,0,vd,e);
const real_t J1 = J(x,y,1,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,e);
QQ(x,y,vd) = W(x,y)*(J0*C0 + J1*C1);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t qd = 0.0;
for (int qx = 0; qx < q; ++qx)
{
qd += QQ(qx,qy,vd) * Btx(dx,qx);
}
QD(dx,qy,vd) = qd;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
DeviceTensor<4> Yxy(Y, nx, ny, vdim, ne);
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t dd = 0.0;
for (int qy = 0; qy < q; ++qy)
{
dd += QD(dx,qy,vd) * Bty(dy,qy);
}
Yxy(dx,dy,vd,e) += dd;
}
}
}
MFEM_SYNC_THREAD;
});
}
template <int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble3D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 3 * (d - 1) * (d - 1) * d * ne, "y wrong length");
constexpr int vdim = 3;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int vdim = 3;
if (M[e] == 0) { return; } // ignore
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
MFEM_SHARED real_t sm0[vdim*Q*Q*Q];
MFEM_SHARED real_t sm1[vdim*Q*Q*Q];
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const real_t cst_val_0 = C(0,0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0,0);
const real_t cst_val_2 = C(2,0,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
for (int z = 0; z < q; ++z)
{
const real_t J0 = J(x,y,z,0,vd,e);
const real_t J1 = J(x,y,z,1,vd,e);
const real_t J2 = J(x,y,z,2,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,z,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,z,e);
const real_t C2 = cst ? cst_val_2 : C(2,x,y,z,e);
QQQ(x,y,z,vd) = W(x,y,z)*(J0*C0 + J1*C1 + J2*C2);
}
}
}
}
MFEM_SYNC_THREAD;
// Apply Bt operator
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qx = 0; qx < q; ++qx)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += QQQ(qx,qy,qz,vd) * Btx(dx,qx);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DQQ(dx,qy,qz,vd) = u[qz]; }
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qy = 0; qy < q; ++qy)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += DQQ(dx,qy,qz,vd) * Bty(dy,qy);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DDQ(dx,dy,qz,vd) = u[qz]; }
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
const int nz = (vd == 2) ? d : d-1;
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
DeviceMatrix Btz = (vd == 2) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[D];
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { u[dz] = 0.0; }
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] += DDQ(dx,dy,qz,vd) * Btz(dz,qz);
}
}
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { Yxyz(dx,dy,dz,vd,e) += u[dz]; }
}
}
}
MFEM_SYNC_THREAD;
});
}
/// @param ne number of elements
/// @param markers array where entry markers[e] == 0 to skip assembly over
/// element e element
/// @param jac Spatial Jacobians evaluated at all quadrature points
/// @param weights 1D quadrature weights
/// @param testBO 1D open basis test functions
/// @param testBC 1D closed basis test functions
/// @param coeff coefficient values evaluated at quadrature points, possibly
/// compressed.
/// @param d number of 1D closed dofs
/// @param q number of 1D quadrature points
/// @tparam T_D1D maximum number of dofs along any direction, or 0
/// @tparam T_Q1D maximum number of quadrature points along any direction, or 0
template <int T_D1D = 0, int T_Q1D = 0>
static void HcurlDLFAssemble3D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 3 * (d - 1) * d * d * ne, "y wrong length");
constexpr int vdim = 3;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE(int e)
{
if (M[e] == 0)
{
// ignore
return;
}
constexpr int vdim = 3;
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
MFEM_SHARED real_t sBot[Q * D];
MFEM_SHARED real_t sBct[Q * D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d - 1, q);
kernels::internal::LoadB<D, Q>(d - 1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D, Q>(d, q, BC, sBct);
MFEM_SHARED real_t sm0[vdim * Q * Q * Q];
MFEM_SHARED real_t sm1[vdim * Q * Q * Q];
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
const real_t cst_val_0 = C(0, 0, 0, 0, 0);
const real_t cst_val_1 = C(1, 0, 0, 0, 0);
const real_t cst_val_2 = C(2, 0, 0, 0, 0);
MFEM_FOREACH_THREAD(vd, z, vdim)
{
MFEM_FOREACH_THREAD(y, y, q)
{
MFEM_FOREACH_THREAD(x, x, q)
{
for (int z = 0; z < q; ++z)
{
real_t curr[3];
curr[0] = cst ? cst_val_0 : C(0, x, y, z, e);
curr[1] = cst ? cst_val_1 : C(1, x, y, z, e);
curr[2] = cst ? cst_val_2 : C(2, x, y, z, e);
const real_t J11 = J(x, y, z, 0, 0, e);
const real_t J21 = J(x, y, z, 1, 0, e);
const real_t J31 = J(x, y, z, 2, 0, e);
const real_t J12 = J(x, y, z, 0, 1, e);
const real_t J22 = J(x, y, z, 1, 1, e);
const real_t J32 = J(x, y, z, 2, 1, e);
const real_t J13 = J(x, y, z, 0, 2, e);
const real_t J23 = J(x, y, z, 1, 2, e);
const real_t J33 = J(x, y, z, 2, 2, e);
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
const real_t A[9] = {A11, A12, A13, A21, A22,
A23, A31, A32, A33
};
QQQ(x, y, z, vd) = W(x, y, z) * (A[vd * vdim] * curr[0] +
A[vd * vdim + 1] * curr[1] +
A[vd * vdim + 2] * curr[2]);
}
}
}
}
MFEM_SYNC_THREAD;
// Apply Bt operator
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
DeviceMatrix Btx = (vd == 0) ? Bot : Bct;
MFEM_FOREACH_THREAD(qy, y, q)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qx = 0; qx < q; ++qx)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += QQQ(qx, qy, qz, vd) * Btx(dx, qx);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
DQQ(dx, qy, qz, vd) = u[qz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
const int ny = (vd == 1) ? d - 1 : d;
DeviceMatrix Bty = (vd == 1) ? Bot : Bct;
MFEM_FOREACH_THREAD(dy, y, ny)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qy = 0; qy < q; ++qy)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += DQQ(dx, qy, qz, vd) * Bty(dy, qy);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
DDQ(dx, dy, qz, vd) = u[qz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
const int ny = (vd == 1) ? d - 1 : d;
const int nz = (vd == 2) ? d - 1 : d;
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
DeviceMatrix Btz = (vd == 2) ? Bot : Bct;
MFEM_FOREACH_THREAD(dy, y, ny)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[D];
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] += DDQ(dx, dy, qz, vd) * Btz(dz, qz);
}
}
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
Yxyz(dx, dy, dz, vd, e) += u[dz];
}
}
}
}
MFEM_SYNC_THREAD;
});
}
template <FiniteElement::DerivType TestType, int DIM, int TEST_D1D, int Q1D>
VectorFEDomainLFIntegrator::AssembleKernelType
VectorFEDomainLFIntegrator::AssembleKernels::Kernel()
{
if constexpr (TestType == FiniteElement::DIV)
{
if constexpr (DIM == 2)
{
return HdivDLFAssemble2D<TEST_D1D, Q1D>;
}
if constexpr (DIM == 3)
{
return HdivDLFAssemble3D<TEST_D1D, Q1D>;
}
}
if constexpr (TestType == FiniteElement::CURL)
{
if constexpr (DIM == 3)
{
return HcurlDLFAssemble3D<TEST_D1D, Q1D>;
}
}
MFEM_ABORT("");
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+68 -301
View File
@@ -13,317 +13,76 @@
#include "../../fem/kernels.hpp"
#include "../fem.hpp"
#include "lininteg_domain_kernels.hpp"
namespace mfem
{
template<int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble2D(
const int ne, const int d, const int q, const int *markers, const real_t *bo,
const real_t *bc, const real_t *j, const real_t *weights,
const Vector &coeff, real_t *y)
VectorFEDomainLFIntegrator::Kernels::Kernels()
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 1, 1>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 2, 2>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 3, 3>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 4, 4>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 5, 5>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 6, 6>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 7, 7>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 2, 8, 8>();
static constexpr int vdim = 2;
const auto F = coeff.Read();
const auto M = Reshape(markers, ne);
const auto BO = Reshape(bo, q, d-1);
const auto BC = Reshape(bc, q, d);
const auto J = Reshape(j, q, q, vdim, vdim, ne);
const auto W = Reshape(weights, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
auto Y = Reshape(y, 2*(d-1)*d, ne);
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 1, 1>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 2, 2>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 3, 3>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 4, 4>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 5, 5>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 6, 6>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 7, 7>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::DIV, 3, 8, 8>();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
{
if (M(e) == 0) { return; } // ignore
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 1, 1>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 2, 2>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 3, 3>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 4, 4>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 5, 5>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 6, 6>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 7, 7>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 8, 8>();
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
MFEM_SHARED real_t sQQ[vdim*Q*Q];
MFEM_SHARED real_t sQD[vdim*Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
const DeviceCube QQ(sQQ, q, q, vdim);
const DeviceCube QD(sQD, q, d, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const real_t cst_val_0 = C(0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
const real_t J0 = J(x,y,0,vd,e);
const real_t J1 = J(x,y,1,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,e);
QQ(x,y,vd) = W(x,y)*(J0*C0 + J1*C1);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t qd = 0.0;
for (int qx = 0; qx < q; ++qx)
{
qd += QQ(qx,qy,vd) * Btx(dx,qx);
}
QD(dx,qy,vd) = qd;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
DeviceTensor<4> Yxy(Y, nx, ny, vdim, ne);
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t dd = 0.0;
for (int qy = 0; qy < q; ++qy)
{
dd += QD(dx,qy,vd) * Bty(dy,qy);
}
Yxy(dx,dy,vd,e) += dd;
}
}
}
MFEM_SYNC_THREAD;
});
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 1, 2>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 2, 3>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 3, 4>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 4, 5>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 5, 6>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 6, 7>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 7, 8>();
VectorFEDomainLFIntegrator::AddSpecialization<FiniteElement::CURL, 3, 8, 9>();
}
template<int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble3D(
const int ne, const int d, const int q, const int *markers, const real_t *bo,
const real_t *bc, const real_t *j, const real_t *weights,
const Vector &coeff, real_t *y)
/// \cond DO_NOT_DOCUMENT
VectorFEDomainLFIntegrator::AssembleKernelType
VectorFEDomainLFIntegrator::AssembleKernels::Fallback(
FiniteElement::DerivType TestType, int DIM, int, int)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
static constexpr int vdim = 3;
const auto F = coeff.Read();
const auto M = Reshape(markers, ne);
const auto BO = Reshape(bo, q, d-1);
const auto BC = Reshape(bc, q, d);
const auto J = Reshape(j, q, q, q, vdim, vdim, ne);
const auto W = Reshape(weights, q, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
auto Y = Reshape(y, 2*(d-1)*(d-1)*d, ne);
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
if (TestType == FiniteElement::DIV)
{
if (M(e) == 0) { return; } // ignore
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
MFEM_SHARED real_t sm0[vdim*Q*Q*Q];
MFEM_SHARED real_t sm1[vdim*Q*Q*Q];
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
if (DIM == 2)
{
const real_t cst_val_0 = C(0,0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0,0);
const real_t cst_val_2 = C(2,0,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
for (int z = 0; z < q; ++z)
{
const real_t J0 = J(x,y,z,0,vd,e);
const real_t J1 = J(x,y,z,1,vd,e);
const real_t J2 = J(x,y,z,2,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,z,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,z,e);
const real_t C2 = cst ? cst_val_2 : C(2,x,y,z,e);
QQQ(x,y,z,vd) = W(x,y,z)*(J0*C0 + J1*C1 + J2*C2);
}
}
}
return HdivDLFAssemble2D<0, 0>;
}
MFEM_SYNC_THREAD;
// Apply Bt operator
MFEM_FOREACH_THREAD(vd,z,vdim)
if (DIM == 3)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qx = 0; qx < q; ++qx)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += QQQ(qx,qy,qz,vd) * Btx(dx,qx);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DQQ(dx,qy,qz,vd) = u[qz]; }
}
}
return HdivDLFAssemble3D<0, 0>;
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qy = 0; qy < q; ++qy)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += DQQ(dx,qy,qz,vd) * Bty(dy,qy);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DDQ(dx,dy,qz,vd) = u[qz]; }
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
const int nz = (vd == 2) ? d : d-1;
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
DeviceMatrix Btz = (vd == 2) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[D];
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { u[dz] = 0.0; }
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] += DDQ(dx,dy,qz,vd) * Btz(dz,qz);
}
}
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { Yxyz(dx,dy,dz,vd,e) += u[dz]; }
}
}
}
MFEM_SYNC_THREAD;
});
}
static void HdivDLFAssemble(const FiniteElementSpace &fes,
const IntegrationRule *ir,
const Array<int> &markers,
const Vector &coeff,
Vector &y)
{
Mesh &mesh = *fes.GetMesh();
const int dim = mesh.Dimension();
const FiniteElement *el = fes.GetTypicalFE();
const auto *vel = dynamic_cast<const VectorTensorFiniteElement *>(el);
MFEM_VERIFY(vel != nullptr, "Must be VectorTensorFiniteElement");
const MemoryType mt = Device::GetDeviceMemoryType();
const DofToQuad &maps_o = vel->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
const DofToQuad &maps_c = vel->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int d = maps_c.ndof, q = maps_c.nqpt;
constexpr int flags = GeometricFactors::JACOBIANS;
const GeometricFactors *geom = mesh.GetGeometricFactors(*ir, flags, mt);
decltype(&HdivDLFAssemble2D<>) ker =
dim == 2 ? HdivDLFAssemble2D<> : HdivDLFAssemble3D<>;
if (dim==2)
{
if (d==1 && q==1) { ker=HdivDLFAssemble2D<1,1>; }
if (d==2 && q==2) { ker=HdivDLFAssemble2D<2,2>; }
if (d==3 && q==3) { ker=HdivDLFAssemble2D<3,3>; }
if (d==4 && q==4) { ker=HdivDLFAssemble2D<4,4>; }
if (d==5 && q==5) { ker=HdivDLFAssemble2D<5,5>; }
if (d==6 && q==6) { ker=HdivDLFAssemble2D<6,6>; }
if (d==7 && q==7) { ker=HdivDLFAssemble2D<7,7>; }
if (d==8 && q==8) { ker=HdivDLFAssemble2D<8,8>; }
}
if (dim==3)
else if (TestType == FiniteElement::CURL)
{
if (d==2 && q==2) { ker=HdivDLFAssemble3D<2,2>; }
if (d==3 && q==3) { ker=HdivDLFAssemble3D<3,3>; }
if (d==4 && q==4) { ker=HdivDLFAssemble3D<4,4>; }
if (d==5 && q==5) { ker=HdivDLFAssemble3D<5,5>; }
if (d==6 && q==6) { ker=HdivDLFAssemble3D<6,6>; }
if (d==7 && q==7) { ker=HdivDLFAssemble3D<7,7>; }
if (d==8 && q==8) { ker=HdivDLFAssemble3D<8,8>; }
if (DIM == 3)
{
return HcurlDLFAssemble3D<0, 0>;
}
}
MFEM_VERIFY(ker, "No kernel ndof " << d << " nqpt " << q);
const int ne = mesh.GetNE();
const int *M = markers.Read();
const real_t *Bo = maps_o.B.Read();
const real_t *Bc = maps_c.B.Read();
const real_t *J = geom->J.Read();
const real_t *W = ir->GetWeights().Read();
real_t *Y = y.ReadWrite();
ker(ne, d, q, M, Bo, Bc, J, W, coeff, Y);
MFEM_ABORT("");
}
/// \endcond DO_NOT_DOCUMENT
void VectorFEDomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
const Array<int> &markers,
@@ -337,15 +96,23 @@ void VectorFEDomainLFIntegrator::AssembleDevice(const FiniteElementSpace &fes,
QuadratureSpace qs(*fes.GetMesh(), *ir);
CoefficientVector coeff(QF, qs, CoefficientStorage::COMPRESSED);
const int fe_type = fe.GetDerivType();
if (fe_type == FiniteElement::DIV)
{
HdivDLFAssemble(fes, ir, markers, coeff, b);
}
else
{
MFEM_ABORT("Not implemented.");
}
const FiniteElement::DerivType fe_type =
static_cast<FiniteElement::DerivType>(fe.GetDerivType());
Mesh &mesh = *fes.GetMesh();
const int dim = mesh.Dimension();
const FiniteElement *el = fes.GetTypicalFE();
const auto *vel = dynamic_cast<const VectorTensorFiniteElement *>(el);
MFEM_VERIFY(vel != nullptr, "Must be VectorTensorFiniteElement");
const MemoryType mt = Device::GetDeviceMemoryType();
const DofToQuad &maps_o = vel->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
const DofToQuad &maps_c = vel->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int d = maps_c.ndof, q = maps_c.nqpt;
constexpr int flags = GeometricFactors::JACOBIANS;
const GeometricFactors *geom = mesh.GetGeometricFactors(*ir, flags, mt);
AssembleKernels::Run(fe_type, dim, d, q, mesh.GetNE(), markers, geom->J,
ir->GetWeights(), maps_o.B, maps_c.B, coeff, b, d, q);
}
} // namespace mfem
+143 -771
View File
@@ -9,21 +9,51 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../../general/forall.hpp"
#include "../nonlininteg.hpp"
#include "../ceed/integrators/nlconvection/nlconvection.hpp"
#include "./nonlininteg_vecconvection_pa.hpp" // IWYU pragma: keep
#include "./nonlininteg_vecconvection_pa_grad.hpp" // IWYU pragma: keep
#include "./nonlininteg_vecconvection_pa_diag.hpp" // IWYU pragma: keep
namespace mfem
{
VectorConvectionNLFIntegrator::Kernels::Kernels()
{
// 2D
VectorConvectionNLFIntegrator::AddSpecialization<2, 2, 2>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 2, 3>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 3, 4>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 3, 5>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 4, 5>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 4, 6>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 5, 7>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 5, 8>();
VectorConvectionNLFIntegrator::AddSpecialization<2, 6, 8>();
// 3D
VectorConvectionNLFIntegrator::AddSpecialization<3, 2, 3>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 2, 4>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 2, 5>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 3, 4>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 3, 5>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 3, 6>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 4, 5>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 4, 6>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 4, 7>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 4, 8>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 5, 6>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 5, 7>();
VectorConvectionNLFIntegrator::AddSpecialization<3, 5, 8>();
}
void VectorConvectionNLFIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation &T = *mesh->GetTypicalElementTransformation();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, T);
ElementTransformation &Tr = *mesh->GetTypicalElementTransformation();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Tr);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -39,769 +69,124 @@ void VectorConvectionNLFIntegrator::AssemblePA(const FiniteElementSpace &fes)
}
return;
}
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
ne = mesh->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
dim = mesh->Dimension();
MFEM_VERIFY(dim == 2 || dim == 3, "Dimension not supported");
const MemoryType mt = pa_mt == MemoryType::DEFAULT
? Device::GetDeviceMemoryType()
: pa_mt;
pa_adj.SetSize(ne * nq * dim * dim, mt);
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
pa_data.SetSize(ne * nq * dim * dim, Device::GetMemoryType());
real_t COEFF = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient *>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
COEFF = cQ->constant;
}
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
if (dim == 1)
{
MFEM_ABORT("dim==1 not supported!");
}
d1d = maps->ndof;
q1d = maps->nqpt;
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
const int nq1d = q1d * q1d * (dim==3 ? q1d : 1);
MFEM_VERIFY(coeff.Size() == 1 || coeff.Size() == nq1d*ne, "Invalid coeff");
MFEM_VERIFY(ir->GetWeights().Size() == nq1d, "Invalid weights size");
const auto w_r = ir->GetWeights().Read();
const bool const_coeff = coeff.Size() == 1;
if (dim == 2)
{
auto J = Reshape(geom->J.Read(), NQ, 2, 2, NE);
auto G = Reshape(pa_data.Write(), NQ, 2, 2, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const int Q1D = q1d;
constexpr int VDIM = 2, DIM = 2;
const auto W = Reshape(w_r, Q1D, Q1D);
const auto C = const_coeff ?
Reshape(coeff.Read(), 1, 1, 1) :
Reshape(coeff.Read(), Q1D, Q1D, ne);
const auto J = Reshape(geom->J.Read(), Q1D, Q1D, VDIM, DIM, ne);
auto A = Reshape(pa_adj.Write(), VDIM, DIM, Q1D, Q1D, ne);
mfem::forall_2D(ne, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
for (int q = 0; q < NQ; ++q)
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
const real_t J11 = J(q, 0, 0, e);
const real_t J12 = J(q, 0, 1, e);
const real_t J21 = J(q, 1, 0, e);
const real_t J22 = J(q, 1, 1, e);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = W[q] * COEFF * J22; // 1,1
G(q, 0, 1, e) = W[q] * COEFF * -J12; // 1,2
G(q, 1, 0, e) = W[q] * COEFF * -J21; // 2,1
G(q, 1, 1, e) = W[q] * COEFF * J11; // 2,2
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t J11 = J(qx, qy, 0, 0, e), J12 = J(qx, qy, 0, 1, e);
const real_t J21 = J(qx, qy, 1, 0, e), J22 = J(qx, qy, 1, 1, e);
// adj(J)
const real_t A11 = +J22, A12 = -J12;
const real_t A21 = -J21, A22 = +J11;
// Store w * coeff * adj(J)
const real_t w = W(qx, qy);
const real_t c = const_coeff ? C(0, 0, 0) : C(qx, qy, e);
A(0, 0, qx, qy, e) = w * c * A11;
A(1, 0, qx, qy, e) = w * c * A12;
A(0, 1, qx, qy, e) = w * c * A21;
A(1, 1, qx, qy, e) = w * c * A22;
}
}
});
}
if (dim == 3)
else if (dim == 3)
{
auto J = Reshape(geom->J.Read(), NQ, 3, 3, NE);
auto G = Reshape(pa_data.Write(), NQ, 3, 3, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const int Q1D = q1d;
constexpr int VDIM = 3, DIM = 3;
const auto W = Reshape(w_r, Q1D, Q1D, Q1D);
const auto C = const_coeff ?
Reshape(coeff.Read(), 1, 1, 1, 1) :
Reshape(coeff.Read(), Q1D, Q1D, Q1D, ne);
const auto J = Reshape(geom->J.Read(), Q1D, Q1D, Q1D, VDIM, DIM, ne);
auto A = Reshape(pa_adj.Write(), VDIM, DIM, Q1D, Q1D, Q1D, ne);
mfem::forall_3D(ne, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
for (int q = 0; q < NQ; ++q)
MFEM_FOREACH_THREAD_DIRECT(qz, z, Q1D)
{
const real_t J11 = J(q, 0, 0, e);
const real_t J21 = J(q, 1, 0, e);
const real_t J31 = J(q, 2, 0, e);
const real_t J12 = J(q, 0, 1, e);
const real_t J22 = J(q, 1, 1, e);
const real_t J32 = J(q, 2, 1, e);
const real_t J13 = J(q, 0, 2, e);
const real_t J23 = J(q, 1, 2, e);
const real_t J33 = J(q, 2, 2, e);
const real_t cw = W[q] * COEFF;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = cw * A11; // 1,1
G(q, 0, 1, e) = cw * A12; // 1,2
G(q, 0, 2, e) = cw * A13; // 1,3
G(q, 1, 0, e) = cw * A21; // 2,1
G(q, 1, 1, e) = cw * A22; // 2,2
G(q, 1, 2, e) = cw * A23; // 2,3
G(q, 2, 0, e) = cw * A31; // 3,1
G(q, 2, 1, e) = cw * A32; // 3,2
G(q, 2, 2, e) = cw * A33; // 3,3
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t J11 = J(qx, qy, qz, 0, 0, e),
J12 = J(qx, qy, qz, 0, 1, e),
J13 = J(qx, qy, qz, 0, 2, e);
const real_t J21 = J(qx, qy, qz, 1, 0, e),
J22 = J(qx, qy, qz, 1, 1, e),
J23 = J(qx, qy, qz, 1, 2, e);
const real_t J31 = J(qx, qy, qz, 2, 0, e),
J32 = J(qx, qy, qz, 2, 1, e),
J33 = J(qx, qy, qz, 2, 2, e);
const real_t c =
const_coeff ? C(0, 0, 0, 0) : C(qx, qy, qz, e);
const real_t cw = W(qx, qy, qz) * c;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// Store wq * coeff * adj(J)
A(0, 0, qx, qy, qz, e) = cw * A11;
A(1, 0, qx, qy, qz, e) = cw * A12;
A(2, 0, qx, qy, qz, e) = cw * A13;
A(0, 1, qx, qy, qz, e) = cw * A21;
A(1, 1, qx, qy, qz, e) = cw * A22;
A(2, 1, qx, qy, qz, e) = cw * A23;
A(0, 2, qx, qy, qz, e) = cw * A31;
A(1, 2, qx, qy, qz, e) = cw * A32;
A(2, 2, qx, qy, qz, e) = cw * A33;
}
}
}
});
}
}
// PA Convection NL 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply2D(const int NE,
const Array<real_t> &b,
const Array<real_t> &g,
const Array<real_t> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D, 2, 2, NE);
auto x = Reshape(x_.Read(), D1D, D1D, 2, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
else
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t data[max_Q1D][max_Q1D][2];
real_t grad0[max_Q1D][max_Q1D][2];
real_t grad1[max_Q1D][max_Q1D][2];
real_t Z[max_Q1D][max_Q1D][2];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] = 0.0;
data[qy][qx][1] = 0.0;
grad0[qy][qx][0] = 0.0;
grad0[qy][qx][1] = 0.0;
grad1[qy][qx][0] = 0.0;
grad1[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t dataX[max_Q1D][2];
real_t gradX0[max_Q1D][2];
real_t gradX1[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s0 = x(dx, dy, 0, e);
const real_t s1 = x(dx, dy, 1, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] += dataX[qx][0] * By;
data[qy][qx][1] += dataX[qx][1] * By;
grad0[qy][qx][0] += gradX0[qx][0] * By;
grad0[qy][qx][1] += gradX0[qx][1] * Gy;
grad1[qy][qx][0] += gradX1[qx][0] * By;
grad1[qy][qx][1] += gradX1[qx][1] * Gy;
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const real_t u1 = data[qy][qx][0];
const real_t u2 = data[qy][qx][1];
const real_t grad00 = grad0[qy][qx][0];
const real_t grad01 = grad0[qy][qx][1];
const real_t grad10 = grad1[qy][qx][0];
const real_t grad11 = grad1[qy][qx][1];
const real_t Dxu1 = grad00 * Q(q, 0, 0, e) + grad01 * Q(q, 1, 0, e);
const real_t Dyu1 = grad00 * Q(q, 0, 1, e) + grad01 * Q(q, 1, 1, e);
const real_t Dxu2 = grad10 * Q(q, 0, 0, e) + grad11 * Q(q, 1, 0, e);
const real_t Dyu2 = grad10 * Q(q, 0, 1, e) + grad11 * Q(q, 1, 1, e);
Z[qy][qx][0] = u1 * Dxu1 + u2 * Dyu1;
Z[qy][qx][1] = u1 * Dxu2 + u2 * Dyu2;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t Y[max_D1D][2];
for (int dx = 0; dx < D1D; ++dx)
{
Y[dx][0] = 0.0;
Y[dx][1] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Btx = Bt(dx, qx);
Y[dx][0] += Btx * Z[qy][qx][0];
Y[dx][1] += Btx * Z[qy][qx][1];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const real_t Bty = Bt(dy, qy);
y(dx, dy, 0, e) += Bty * Y[dx][0];
y(dx, dy, 1, e) += Bty * Y[dx][1];
}
}
}
});
}
// PA Convection NL 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply3D(const int NE,
const Array<real_t> &b,
const Array<real_t> &g,
const Array<real_t> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t data[max_Q1D][max_Q1D][max_Q1D][VDIM];
real_t grad0[max_Q1D][max_Q1D][max_Q1D][VDIM];
real_t grad1[max_Q1D][max_Q1D][max_Q1D][VDIM];
real_t grad2[max_Q1D][max_Q1D][max_Q1D][VDIM];
real_t Z[max_Q1D][max_Q1D][max_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] = 0.0;
data[qz][qy][qx][1] = 0.0;
data[qz][qy][qx][2] = 0.0;
grad0[qz][qy][qx][0] = 0.0;
grad0[qz][qy][qx][1] = 0.0;
grad0[qz][qy][qx][2] = 0.0;
grad1[qz][qy][qx][0] = 0.0;
grad1[qz][qy][qx][1] = 0.0;
grad1[qz][qy][qx][2] = 0.0;
grad2[qz][qy][qx][0] = 0.0;
grad2[qz][qy][qx][1] = 0.0;
grad2[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
real_t dataXY[max_Q1D][max_Q1D][VDIM];
real_t gradXY0[max_Q1D][max_Q1D][VDIM];
real_t gradXY1[max_Q1D][max_Q1D][VDIM];
real_t gradXY2[max_Q1D][max_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] = 0.0;
dataXY[qy][qx][1] = 0.0;
dataXY[qy][qx][2] = 0.0;
gradXY0[qy][qx][0] = 0.0;
gradXY0[qy][qx][1] = 0.0;
gradXY0[qy][qx][2] = 0.0;
gradXY1[qy][qx][0] = 0.0;
gradXY1[qy][qx][1] = 0.0;
gradXY1[qy][qx][2] = 0.0;
gradXY2[qy][qx][0] = 0.0;
gradXY2[qy][qx][1] = 0.0;
gradXY2[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t dataX[max_Q1D][VDIM];
real_t gradX0[max_Q1D][VDIM];
real_t gradX1[max_Q1D][VDIM];
real_t gradX2[max_Q1D][VDIM];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
dataX[qx][2] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX0[qx][2] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
gradX1[qx][2] = 0.0;
gradX2[qx][0] = 0.0;
gradX2[qx][1] = 0.0;
gradX2[qx][2] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s0 = x(dx, dy, dz, 0, e);
const real_t s1 = x(dx, dy, dz, 1, e);
const real_t s2 = x(dx, dy, dz, 2, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
dataX[qx][2] += s2 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX0[qx][2] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
gradX1[qx][2] += s1 * Bx;
gradX2[qx][0] += s2 * Gx;
gradX2[qx][1] += s2 * Bx;
gradX2[qx][2] += s2 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] += dataX[qx][0] * By;
dataXY[qy][qx][1] += dataX[qx][1] * By;
dataXY[qy][qx][2] += dataX[qx][2] * By;
gradXY0[qy][qx][0] += gradX0[qx][0] * By;
gradXY0[qy][qx][1] += gradX0[qx][1] * Gy;
gradXY0[qy][qx][2] += gradX0[qx][2] * By;
gradXY1[qy][qx][0] += gradX1[qx][0] * By;
gradXY1[qy][qx][1] += gradX1[qx][1] * Gy;
gradXY1[qy][qx][2] += gradX1[qx][2] * By;
gradXY2[qy][qx][0] += gradX2[qx][0] * By;
gradXY2[qy][qx][1] += gradX2[qx][1] * Gy;
gradXY2[qy][qx][2] += gradX2[qx][2] * By;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t Bz = B(qz, dz);
const real_t Gz = G(qz, dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] += dataXY[qy][qx][0] * Bz;
data[qz][qy][qx][1] += dataXY[qy][qx][1] * Bz;
data[qz][qy][qx][2] += dataXY[qy][qx][2] * Bz;
grad0[qz][qy][qx][0] += gradXY0[qy][qx][0] * Bz;
grad0[qz][qy][qx][1] += gradXY0[qy][qx][1] * Bz;
grad0[qz][qy][qx][2] += gradXY0[qy][qx][2] * Gz;
grad1[qz][qy][qx][0] += gradXY1[qy][qx][0] * Bz;
grad1[qz][qy][qx][1] += gradXY1[qy][qx][1] * Bz;
grad1[qz][qy][qx][2] += gradXY1[qy][qx][2] * Gz;
grad2[qz][qy][qx][0] += gradXY2[qy][qx][0] * Bz;
grad2[qz][qy][qx][1] += gradXY2[qy][qx][1] * Bz;
grad2[qz][qy][qx][2] += gradXY2[qy][qx][2] * Gz;
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + Q1D * (qy + qz * Q1D);
const real_t u1 = data[qz][qy][qx][0];
const real_t u2 = data[qz][qy][qx][1];
const real_t u3 = data[qz][qy][qx][2];
const real_t grad00 = grad0[qz][qy][qx][0];
const real_t grad01 = grad0[qz][qy][qx][1];
const real_t grad02 = grad0[qz][qy][qx][2];
const real_t grad10 = grad1[qz][qy][qx][0];
const real_t grad11 = grad1[qz][qy][qx][1];
const real_t grad12 = grad1[qz][qy][qx][2];
const real_t grad20 = grad2[qz][qy][qx][0];
const real_t grad21 = grad2[qz][qy][qx][1];
const real_t grad22 = grad2[qz][qy][qx][2];
const real_t Dxu1 = grad00 * Q(q, 0, 0, e)
+ grad01 * Q(q, 1, 0, e)
+ grad02 * Q(q, 2, 0, e);
const real_t Dyu1 = grad00 * Q(q, 0, 1, e)
+ grad01 * Q(q, 1, 1, e)
+ grad02 * Q(q, 2, 1, e);
const real_t Dzu1 = grad00 * Q(q, 0, 2, e)
+ grad01 * Q(q, 1, 2, e)
+ grad02 * Q(q, 2, 2, e);
const real_t Dxu2 = grad10 * Q(q, 0, 0, e)
+ grad11 * Q(q, 1, 0, e)
+ grad12 * Q(q, 2, 0, e);
const real_t Dyu2 = grad10 * Q(q, 0, 1, e)
+ grad11 * Q(q, 1, 1, e)
+ grad12 * Q(q, 2, 1, e);
const real_t Dzu2 = grad10 * Q(q, 0, 2, e)
+ grad11 * Q(q, 1, 2, e)
+ grad12 * Q(q, 2, 2, e);
const real_t Dxu3 = grad20 * Q(q, 0, 0, e)
+ grad21 * Q(q, 1, 0, e)
+ grad22 * Q(q, 2, 0, e);
const real_t Dyu3 = grad20 * Q(q, 0, 1, e)
+ grad21 * Q(q, 1, 1, e)
+ grad22 * Q(q, 2, 1, e);
const real_t Dzu3 = grad20 * Q(q, 0, 2, e)
+ grad21 * Q(q, 1, 2, e)
+ grad22 * Q(q, 2, 2, e);
Z[qz][qy][qx][0] = u1 * Dxu1 + u2 * Dyu1 + u3 * Dzu1;
Z[qz][qy][qx][1] = u1 * Dxu2 + u2 * Dyu2 + u3 * Dzu2;
Z[qz][qy][qx][2] = u1 * Dxu3 + u2 * Dyu3 + u3 * Dzu3;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
real_t opXY[max_D1D][max_D1D][VDIM];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
opXY[dy][dx][0] = 0.0;
opXY[dy][dx][1] = 0.0;
opXY[dy][dx][2] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t opX[max_D1D][VDIM];
for (int dx = 0; dx < D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
opX[dx][2] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Btx = Bt(dx, qx);
opX[dx][0] += Btx * Z[qz][qy][qx][0];
opX[dx][1] += Btx * Z[qz][qy][qx][1];
opX[dx][2] += Btx * Z[qz][qy][qx][2];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const real_t Bty = Bt(dy, qy);
opXY[dy][dx][0] += Bty * opX[dx][0];
opXY[dy][dx][1] += Bty * opX[dx][1];
opXY[dy][dx][2] += Bty * opX[dx][2];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const real_t Btz = Bt(dz, qz);
y(dx, dy, dz, 0, e) += Btz * opXY[dy][dx][0];
y(dx, dy, dz, 1, e) += Btz * opXY[dy][dx][1];
y(dx, dy, dz, 2, e) += Btz * opXY[dy][dx][2];
}
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX_D1D = 0, int T_MAX_Q1D = 0>
static void SmemPAConvectionNLApply3D(const int NE,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_SHARED real_t BG[2][MQ1 * MD1];
real_t(*B)[MD1] = (real_t(*)[MD1])(BG + 0);
real_t(*G)[MD1] = (real_t(*)[MD1])(BG + 1);
real_t(*Bt)[MQ1] = (real_t(*)[MQ1])(BG + 0);
MFEM_SHARED real_t U[2][MQ1][MQ1][MQ1];
MFEM_SHARED real_t sm0[3][MQ1 * MQ1 * MQ1];
MFEM_SHARED real_t sm1[3][MQ1 * MQ1 * MQ1];
real_t(*DDQ0)[MD1][MQ1] = (real_t(*)[MD1][MQ1])(sm0 + 0);
real_t(*DDQ1)[MD1][MQ1] = (real_t(*)[MD1][MQ1])(sm0 + 1);
real_t(*X)[MD1][MD1] = (real_t(*)[MD1][MD1])(sm0 + 2);
real_t(*DQQ0)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm1 + 0);
real_t(*DQQ1)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm1 + 1);
real_t(*DQQ2)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm1 + 2);
real_t(*QQQ0)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm0 + 0);
real_t(*QQQ1)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm0 + 1);
real_t(*QQQ2)[MQ1][MQ1] = (real_t(*)[MQ1][MQ1])(sm0 + 2);
real_t(*QQD0)[MQ1][MD1] = (real_t(*)[MQ1][MD1])(sm1 + 0);
real_t(*QDD0)[MD1][MD1] = (real_t(*)[MD1][MD1])(sm0 + 0);
MFEM_SHARED real_t Z[MQ1][MQ1][MQ1];
for (int cy = 0; cy < VDIM; ++cy)
{
if (tidz == 0)
{
MFEM_FOREACH_THREAD(q, x, Q1D)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
B[q][d] = b(q, d);
G[q][d] = g(q, d);
}
}
}
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D) { Z[qz][qy][qx] = 0.0; }
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
X[dz][dy][dx] = x(dx, dy, dz, cy, e);
U[0][dz][dy][dx] = x(dx, dy, dz, c, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u = 0.0;
real_t v = 0.0;
real_t z = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const real_t coord = X[dz][dy][dx];
const real_t value = U[0][dz][dy][dx];
u += coord * B[qx][dx];
v += coord * G[qx][dx];
z += value * B[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
U[1][dz][dy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u = 0.0;
real_t v = 0.0;
real_t w = 0.0;
real_t z = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
z += U[1][dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
U[0][dz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u = 0.0;
real_t v = 0.0;
real_t w = 0.0;
real_t z = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
z += U[0][dz][qy][qx] * B[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
U[1][qz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const real_t z = U[1][qz][qy][qx];
const real_t gX = QQQ0[qz][qy][qx];
const real_t gY = QQQ1[qz][qy][qx];
const real_t gZ = QQQ2[qz][qy][qx];
const real_t d = gX * D(q, 0, c, e) + gY * D(q, 1, c, e)
+ gZ * D(q, 2, c, e);
Z[qz][qy][qx] += z * d;
}
}
}
MFEM_SYNC_THREAD;
} // for each conv component
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[d][q] = b(q, d); }
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += Z[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
}
Y(dx, dy, dz, cy, e) += u;
}
}
}
MFEM_SYNC_THREAD;
}
});
MFEM_ABORT("dim " << dim << " not supported!");
}
}
void VectorConvectionNLFIntegrator::AddMultPA(const Vector &x, Vector &y) const
@@ -812,26 +197,13 @@ void VectorConvectionNLFIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
else
{
const int NE = ne;
const int D1D = maps->ndof;
const int Q1D = maps->nqpt;
const Vector &QV = pa_data;
const Array<real_t> &B = maps->B;
const Array<real_t> &G = maps->G;
const Array<real_t> &Bt = maps->Bt;
if (dim == 2)
{
return PAConvectionNLApply2D(NE, B, G, Bt, QV, x, y, D1D, Q1D);
}
if (dim == 3)
{
constexpr int T_MAX_D1D = 8;
constexpr int T_MAX_Q1D = 8;
MFEM_VERIFY(D1D <= T_MAX_D1D && Q1D <= T_MAX_Q1D, "Not yet implemented!");
return SmemPAConvectionNLApply3D<0, 0, T_MAX_D1D, T_MAX_Q1D>
(NE, B, G, QV, x, y, D1D, Q1D);
}
MFEM_ABORT("Not yet implemented!");
AddMultPAKernels::Run(dim, d1d, q1d, ne,
maps->B.Read(),
maps->G.Read(),
pa_adj.Read(),
x.Read(),
y.ReadWrite(),
d1d, q1d);
}
}
+209
View File
@@ -0,0 +1,209 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../kernels.hpp"
#include "../nonlininteg.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
// PA Convection NL 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLApply2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int VDIM = 2, DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, NE);
const auto X = Reshape(x, D1D, D1D, VDIM, NE);
auto Y = Reshape(y, D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1], sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<VDIM, DIM, MQ1> g0, g1;
kernels::internal::v_regs2d_t<VDIM, MQ1> r0, r1;
kernels::internal::v_regs2d_t<VDIM, MQ1> s0, s1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1); // u vector-value
kernels::internal::LoadDofs2d(e, D1D, X, g0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, g0, g1); // u vector-gradient
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const future::tensor<real_t, 2> U =
{
r1[0][qy][qx], r1[1][qy][qx]
};
const future::tensor<real_t, 2,2> gradU = {{
{g1[0][0][qy][qx], g1[1][0][qy][qx]},
{g1[0][1][qy][qx], g1[1][1][qy][qx]},
}
};
const future::tensor<real_t, 2,2> Q = {{
{A(0,0,qx,qy,e), A(1,0,qx,qy,e)},
{A(0,1,qx,qy,e), A(1,1,qx,qy,e)},
}
};
const future::tensor<real_t, 2> conv = transpose(gradU) * (Q * U);
s0[0][qy][qx] = conv[0];
s0[1][qy][qx] = conv[1];
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, s0, s1);
kernels::internal::WriteDofs2d(e, D1D, s1, Y);
});
}
// PA Convection NL 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLApply3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int VDIM = 3, DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, Q1D, NE);
const auto X = Reshape(x, D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y, D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1], sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs3d_t<VDIM, DIM, MQ1> g0, g1;
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1;
kernels::internal::v_regs3d_t<VDIM, MQ1> s0, s1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1); // u vector-value
kernels::internal::LoadDofs3d(e, D1D, X, g0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, g0, g1); // u vector-gradient
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const future::tensor<real_t, 3> U =
{
r1[0][qz][qy][qx], r1[1][qz][qy][qx], r1[2][qz][qy][qx]
};
const future::tensor<real_t, 3,3> gradU = {{
{g1[0][0][qz][qy][qx], g1[1][0][qz][qy][qx], g1[2][0][qz][qy][qx]},
{g1[0][1][qz][qy][qx], g1[1][1][qz][qy][qx], g1[2][1][qz][qy][qx]},
{g1[0][2][qz][qy][qx], g1[1][2][qz][qy][qx], g1[2][2][qz][qy][qx]}
}
};
const future::tensor<real_t, 3,3> Q = {{
{A(0,0,qx,qy,qz,e), A(1,0,qx,qy,qz,e), A(2,0,qx,qy,qz,e)},
{A(0,1,qx,qy,qz,e), A(1,1,qx,qy,qz,e), A(2,1,qx,qy,qz,e)},
{A(0,2,qx,qy,qz,e), A(1,2,qx,qy,qz,e), A(2,2,qx,qy,qz,e)}
}
};
const future::tensor<real_t, 3> conv = transpose(gradU) * (Q * U);
s0[0][qz][qy][qx] = conv[0];
s0[1][qz][qy][qx] = conv[1];
s0[2][qz][qy][qx] = conv[2];
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, s0, s1);
kernels::internal::WriteDofs3d(e, D1D, s1, Y);
});
}
} // namespace internal
template<int DIM, int T_D1D, int T_Q1D>
VectorConvectionNLFIntegrator::AddMultPAType
VectorConvectionNLFIntegrator::AddMultPAKernels::Kernel()
{
static_assert(T_D1D <= T_Q1D, "d1d > q1d is not supported");
if constexpr (DIM == 2)
{
return internal::SmemPAConvectionNLApply2D<T_D1D, T_Q1D>;
}
else if constexpr (DIM == 3)
{
return internal::SmemPAConvectionNLApply3D<T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorConvectionNLFIntegrator::AddMultPAType
VectorConvectionNLFIntegrator::AddMultPAKernels::Fallback
(int dim, int d1d, int q1d)
{
MFEM_VERIFY(d1d <= q1d, "d1d > q1d is not supported");
MFEM_VERIFY(d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
if (dim == 2)
{
return internal::SmemPAConvectionNLApply2D<>;
}
else if (dim == 3)
{
return internal::SmemPAConvectionNLApply3D<>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
@@ -0,0 +1,50 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../ceed/interface/util.hpp"
#include "./nonlininteg_vecconvection_pa_diag.hpp" // IWYU pragma: keep
namespace mfem
{
void VectorConvectionNLFIntegrator::AssembleGradDiagonalPA(Vector &de) const
{
MFEM_VERIFY(!DeviceCanUseCeed(),
"VectorConvectionNLFIntegrator PA gradients are not supported "
"with the libCEED backend");
if (dim == 2)
{
GradDiagPA2D::Run(d1d, q1d, ne,
maps->B.Read(),
maps->G.Read(),
pa_adj.Read(),
pa_u.Read(),
de.ReadWrite(),
d1d, q1d);
}
else if (dim == 3)
{
GradDiagPA3D::Run(d1d, q1d, ne,
maps->B.Read(),
maps->G.Read(),
pa_adj.Read(),
pa_u.Read(),
de.ReadWrite(),
d1d, q1d);
}
else
{
MFEM_ABORT("Unsupported dimension");
}
}
} // namespace mfem
@@ -0,0 +1,302 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../kernels.hpp"
#include "../nonlininteg.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLGradDiagonal2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *u,
real_t *de,
const int d1d,
const int q1d)
{
static constexpr int VDIM = 2, DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, NE);
const auto U = Reshape(u, D1D, D1D, VDIM, NE);
auto D = Reshape(de, D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t sM[3][MQ1][MQ1], sQ[3][MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::v_regs2d_t<VDIM, MQ1> r0, r1;
kernels::internal::vd_regs2d_t<VDIM, DIM, MQ1> g0, g1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, U, r0);
kernels::internal::Eval2d(D1D, Q1D, sM[0], sB, r0, r1);
kernels::internal::LoadDofs2d(e, D1D, U, g0);
kernels::internal::Grad2d(D1D, Q1D, sM[0], sB, sG, g0, g1);
for (int v = 0; v < VDIM; ++v)
{
future::tensor<real_t, VDIM> e_v = {};
e_v[v] = real_t(1);
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
const future::tensor<real_t, VDIM> u_val =
{
r1[0][qy][qx], r1[1][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> Q_adj =
{
{ { A(0, 0, qx, qy, e), A(1, 0, qx, qy, e) },
{ A(0, 1, qx, qy, e), A(1, 1, qx, qy, e) }
}
};
const future::tensor<real_t, VDIM, DIM> grad_U =
{
{ { g1[0][0][qy][qx], g1[1][0][qy][qx] },
{ g1[0][1][qy][qx], g1[1][1][qy][qx] }
}
};
const auto one = Q_adj * u_val;
const auto two = transpose(grad_U) * (Q_adj * e_v);
sQ[0][qx][qy] = one[0];
sQ[1][qx][qy] = one[1];
sQ[2][qx][qy] = two[v];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
real_t s[3] = {};
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = sB[dy][qy], Gy = sG[dy][qy];
s[0] += By * By * sQ[0][qx][qy];
s[1] += Gy * By * sQ[1][qx][qy];
s[2] += By * By * sQ[2][qx][qy];
}
sM[0][qx][dy] = s[0];
sM[1][qx][dy] = s[1];
sM[2][qx][dy] = s[2];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = sB[dx][qx], Gx = sG[dx][qx];
d += Gx * Bx * sM[0][qx][dy] +
Bx * Bx * sM[1][qx][dy] +
Bx * Bx * sM[2][qx][dy];
}
D(dx, dy, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLGradDiagonal3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *u,
real_t *de,
const int d1d,
const int q1d)
{
static constexpr int VDIM = 3, DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, Q1D, NE);
const auto U = Reshape(u, D1D, D1D, D1D, VDIM, NE);
auto D = Reshape(de, D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D<T_Q1D * T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t sM[4][MQ1][MQ1], sQ[4][MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1;
kernels::internal::vd_regs3d_t<VDIM, DIM, MQ1> g0, g1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs3d(e, D1D, U, r0);
kernels::internal::Eval3d(D1D, Q1D, sM[0], sB, r0, r1);
kernels::internal::LoadDofs3d(e, D1D, U, g0);
kernels::internal::Grad3d(D1D, Q1D, sM[0], sB, sG, g0, g1);
for (int v = 0; v < VDIM; ++v)
{
future::tensor<real_t, VDIM> e_v = {};
e_v[v] = real_t(1);
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t s[4] = {};
for (int qz = 0; qz < Q1D; ++qz)
{
const future::tensor<real_t, VDIM> u_val =
{
r1[0][qz][qy][qx], r1[1][qz][qy][qx], r1[2][qz][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> Q_adj = {{
{A(0,0,qx,qy,qz,e), A(1,0,qx,qy,qz,e), A(2,0,qx,qy,qz,e)},
{A(0,1,qx,qy,qz,e), A(1,1,qx,qy,qz,e), A(2,1,qx,qy,qz,e)},
{A(0,2,qx,qy,qz,e), A(1,2,qx,qy,qz,e), A(2,2,qx,qy,qz,e)}
}
};
const future::tensor<real_t, VDIM, DIM> grad_U = {{
{g1[0][0][qz][qy][qx], g1[1][0][qz][qy][qx], g1[2][0][qz][qy][qx]},
{g1[0][1][qz][qy][qx], g1[1][1][qz][qy][qx], g1[2][1][qz][qy][qx]},
{g1[0][2][qz][qy][qx], g1[1][2][qz][qy][qx], g1[2][2][qz][qy][qx]}
}
};
const auto one = Q_adj * u_val;
const auto two = transpose(grad_U) * (Q_adj * e_v);
const real_t Bz = sB[dz][qz], Gz = sG[dz][qz];
s[0] += one[0] * Bz * Bz;
s[1] += one[1] * Bz * Bz;
s[2] += one[2] * Bz * Gz;
s[3] += two[v] * Bz * Bz;
}
sQ[0][qx][qy] = s[0];
sQ[1][qx][qy] = s[1];
sQ[2][qx][qy] = s[2];
sQ[3][qx][qy] = s[3];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t s[4] = {};
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = sB[dy][qy], Gy = sG[dy][qy];
s[0] += By * By * sQ[0][qx][qy];
s[1] += Gy * By * sQ[1][qx][qy];
s[2] += By * By * sQ[2][qx][qy];
s[3] += By * By * sQ[3][qx][qy];
}
sM[0][dy][qx] = s[0];
sM[1][dy][qx] = s[1];
sM[2][dy][qx] = s[2];
sM[3][dy][qx] = s[3];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = sB[dx][qx], Gx = sG[dx][qx];
d += Gx * Bx * sM[0][dy][qx];
d += Bx * Bx * sM[1][dy][qx];
d += Bx * Bx * sM[2][dy][qx];
d += Bx * Bx * sM[3][dy][qx];
}
D(dx, dy, dz, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
}
});
}
} // namespace internal
template<int T_D1D, int T_Q1D>
VectorConvectionNLFIntegrator::GradDiagPAType
VectorConvectionNLFIntegrator::GradDiagPA2D::Kernel()
{
static_assert(T_D1D <= T_Q1D, "d1d > q1d is not supported");
return internal::SmemPAConvectionNLGradDiagonal2D<T_D1D, T_Q1D>;
}
inline VectorConvectionNLFIntegrator::GradDiagPAType
VectorConvectionNLFIntegrator::GradDiagPA2D::Fallback(int d1d, int q1d)
{
MFEM_VERIFY(d1d <= q1d, "d1d > q1d is not supported");
MFEM_VERIFY(d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
return internal::SmemPAConvectionNLGradDiagonal2D<>;
}
template<int T_D1D, int T_Q1D>
VectorConvectionNLFIntegrator::GradDiagPAType
VectorConvectionNLFIntegrator::GradDiagPA3D::Kernel()
{
static_assert(T_D1D <= T_Q1D, "d1d > q1d is not supported");
return internal::SmemPAConvectionNLGradDiagonal3D<T_D1D, T_Q1D>;
}
inline VectorConvectionNLFIntegrator::GradDiagPAType
VectorConvectionNLFIntegrator::GradDiagPA3D::Fallback(int d1d, int q1d)
{
MFEM_VERIFY(d1d <= q1d, "d1d > q1d is not supported");
MFEM_VERIFY(d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
return internal::SmemPAConvectionNLGradDiagonal3D<>;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
@@ -0,0 +1,64 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../ceed/interface/util.hpp"
#include "./nonlininteg_vecconvection_pa_grad.hpp" // IWYU pragma: keep
namespace mfem
{
void VectorConvectionNLFIntegrator::AssembleGradPA(
const Vector &u, const FiniteElementSpace &fes)
{
MFEM_VERIFY(!DeviceCanUseCeed(),
"VectorConvectionNLFIntegrator PA gradients are not supported "
"with the libCEED backend");
this->pa_u = u;
AssemblePA(fes);
}
void VectorConvectionNLFIntegrator::AddMultGradPA(const Vector &x,
Vector &y) const
{
MFEM_VERIFY(!DeviceCanUseCeed(),
"VectorConvectionNLFIntegrator PA gradients are not supported "
"with the libCEED backend");
if (dim == 2)
{
AddMultGradPA2D::Run(d1d, q1d, ne,
maps->B.Read(),
maps->G.Read(),
pa_adj.Read(),
pa_u.Read(),
x.Read(),
y.ReadWrite(),
d1d, q1d);
}
else if (dim == 3)
{
AddMultGradPA3D::Run(d1d, q1d, ne,
maps->B.Read(),
maps->G.Read(),
pa_adj.Read(),
pa_u.Read(),
x.Read(),
y.ReadWrite(),
d1d, q1d);
}
else
{
MFEM_ABORT("Unsupported dimension");
}
}
} // namespace mfem
@@ -0,0 +1,257 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../kernels.hpp"
#include "../nonlininteg.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLGradApply2D(const int ne,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *u,
const real_t *du,
real_t *y,
const int d1d,
const int q1d)
{
static constexpr int VDIM = 2, DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, ne);
const auto U = Reshape(u, D1D, D1D, VDIM, ne);
const auto dU = Reshape(du, D1D, D1D, VDIM, ne);
auto Y = Reshape(y, D1D, D1D, VDIM, ne);
mfem::forall_2D<T_Q1D * T_Q1D>(ne, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<VDIM, DIM, MQ1> g0, g1, g2;
kernels::internal::v_regs2d_t<DIM, MQ1> r0, r1, r2;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, dU, g0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, g0, g1); // δu gradient
kernels::internal::LoadDofs2d(e, D1D, U, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r2); // u value
kernels::internal::LoadDofs2d(e, D1D, dU, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1); // δu value
kernels::internal::LoadDofs2d(e, D1D, U, g0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, g0, g2); // u gradient
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
// First part of the Jacobian: u·∇δu
const future::tensor<real_t, DIM> u_val =
{
r2[0][qy][qx], r2[1][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> Q_adj =
{
{ { A(0, 0, qx, qy, e), A(1, 0, qx, qy, e) },
{ A(0, 1, qx, qy, e), A(1, 1, qx, qy, e) }
}
};
const future::tensor<real_t, VDIM, DIM> grad_dU =
{
{ { g1[0][0][qy][qx], g1[1][0][qy][qx] },
{ g1[0][1][qy][qx], g1[1][1][qy][qx] }
}
};
const auto one = transpose(grad_dU) * (Q_adj * u_val);
// Second part of the Jacobian: δu·∇u
const future::tensor<real_t, DIM> du_val =
{
r1[0][qy][qx], r1[1][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> grad_U =
{
{ { g2[0][0][qy][qx], g2[1][0][qy][qx] },
{ g2[0][1][qy][qx], g2[1][1][qy][qx] }
}
};
const auto two = transpose(grad_U) * (Q_adj * du_val);
// u⋅∇δu + δu⋅∇u
r0[0][qy][qx] = one[0] + two[0];
r0[1][qy][qx] = one[1] + two[1];
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAConvectionNLGradApply3D(const int ne,
const real_t *b,
const real_t *g,
const real_t *a,
const real_t *u,
const real_t *du,
real_t *y,
const int d1d,
const int q1d)
{
static constexpr int VDIM = 3, DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto A = Reshape(a, VDIM, DIM, Q1D, Q1D, Q1D, ne);
const auto U = Reshape(u, D1D, D1D, D1D, VDIM, ne);
const auto dU = Reshape(du, D1D, D1D, D1D, VDIM, ne);
auto Y = Reshape(y, D1D, D1D, D1D, VDIM, ne);
mfem::forall_2D<T_Q1D * T_Q1D>(ne, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1, r2;
kernels::internal::vd_regs3d_t<VDIM, DIM, MQ1> g0, g1, g2;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs3d(e, D1D, dU, g0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, g0, g1); // δu gradient
kernels::internal::LoadDofs3d(e, D1D, U, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r2); // u value
kernels::internal::LoadDofs3d(e, D1D, dU, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1); // δu value
kernels::internal::LoadDofs3d(e, D1D, U, g0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, g0, g2); // u gradient
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
// First part of the Jacobian: u·∇δu
const future::tensor<real_t, DIM> u_val =
{
r2[0][qz][qy][qx],
r2[1][qz][qy][qx],
r2[2][qz][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> Q_adj = {{
{A(0,0,qx,qy,qz,e), A(1,0,qx,qy,qz,e), A(2,0,qx,qy,qz,e)},
{A(0,1,qx,qy,qz,e), A(1,1,qx,qy,qz,e), A(2,1,qx,qy,qz,e)},
{A(0,2,qx,qy,qz,e), A(1,2,qx,qy,qz,e), A(2,2,qx,qy,qz,e)}
}
};
const future::tensor<real_t, DIM, DIM> grad_dU = {{
{g1[0][0][qz][qy][qx], g1[1][0][qz][qy][qx], g1[2][0][qz][qy][qx]},
{g1[0][1][qz][qy][qx], g1[1][1][qz][qy][qx], g1[2][1][qz][qy][qx]},
{g1[0][2][qz][qy][qx], g1[1][2][qz][qy][qx], g1[2][2][qz][qy][qx]}
}
};
const auto one = transpose(grad_dU) * (Q_adj * u_val);
// Second part of the Jacobian: δu·∇u
const future::tensor<real_t, DIM> du_val =
{
r1[0][qz][qy][qx], r1[1][qz][qy][qx], r1[2][qz][qy][qx]
};
const future::tensor<real_t, VDIM, DIM> grad_U = {{
{g2[0][0][qz][qy][qx], g2[1][0][qz][qy][qx], g2[2][0][qz][qy][qx]},
{g2[0][1][qz][qy][qx], g2[1][1][qz][qy][qx], g2[2][1][qz][qy][qx]},
{g2[0][2][qz][qy][qx], g2[1][2][qz][qy][qx], g2[2][2][qz][qy][qx]}
}
};
const auto two = transpose(grad_U) * (Q_adj * du_val);
// u⋅∇δu + δu⋅∇u
r0[0][qz][qy][qx] = one[0] + two[0];
r0[1][qz][qy][qx] = one[1] + two[1];
r0[2][qz][qy][qx] = one[2] + two[2];
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
});
}
} // namespace internal
template<int T_D1D, int T_Q1D>
VectorConvectionNLFIntegrator::AddMultGradPAType
VectorConvectionNLFIntegrator::AddMultGradPA2D::Kernel()
{
static_assert(T_D1D <= T_Q1D, "d1d > q1d is not supported");
return internal::SmemPAConvectionNLGradApply2D<T_D1D, T_Q1D>;
}
inline VectorConvectionNLFIntegrator::AddMultGradPAType
VectorConvectionNLFIntegrator::AddMultGradPA2D::Fallback(int d1d, int q1d)
{
MFEM_VERIFY(d1d <= q1d, "d1d > q1d is not supported");
MFEM_VERIFY(d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
return internal::SmemPAConvectionNLGradApply2D<>;
}
template<int T_D1D, int T_Q1D>
VectorConvectionNLFIntegrator::AddMultGradPAType
VectorConvectionNLFIntegrator::AddMultGradPA3D::Kernel()
{
static_assert(T_D1D <= T_Q1D, "d1d > q1d is not supported");
return internal::SmemPAConvectionNLGradApply3D<T_D1D, T_Q1D>;
}
inline VectorConvectionNLFIntegrator::AddMultGradPAType
VectorConvectionNLFIntegrator::AddMultGradPA3D::Fallback(int d1d, int q1d)
{
MFEM_VERIFY(d1d <= q1d, "d1d > q1d is not supported");
MFEM_VERIFY(d1d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q1d <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
return internal::SmemPAConvectionNLGradApply3D<>;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+376
View File
@@ -236,6 +236,58 @@ IntegrationRule::ApplyToKnotIntervals(KnotVector const& kv) const
return kvir;
}
IntegrationRule IntegrationRule::Reorder(const Array<int> &ordering) const
{
const int np = GetNPoints();
MFEM_VERIFY(np == ordering.Size(), "Invalid permutation size");
IntegrationRule ir(np);
ir.SetOrder(GetOrder());
for (int i = 0; i < np; i++)
{
IntegrationPoint &ip_new = ir.IntPoint(i);
const IntegrationPoint &ip_old = IntPoint(ordering[i]);
ip_new.Set(ip_old.x, ip_old.y, ip_old.z, ip_old.weight);
}
return ir;
}
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim)
{
IntegrationRule ir_mapped(ir.GetNPoints());
ir_mapped.SetOrder(ir.GetOrder());
if (dim == 2)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
ip_mapped.x = ir.IntPoint(i).x;
ip_mapped.weight = ir.IntPoint(i).weight;
}
return ir_mapped;
}
else if (dim == 3)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
ip_mapped.z = ir.IntPoint(i).z * (1 - ir.IntPoint(i).x) * (1 - ir.IntPoint(
i).y);
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
ip_mapped.x = ir.IntPoint(i).x;
ip_mapped.weight = ir.IntPoint(i).weight;
}
return ir_mapped;
}
else
{
MFEM_ABORT("Duffy transformation not implemented for this dimension!");
}
}
#ifdef MFEM_USE_MPFR
// Class for computing hi-precision (HP) quadrature in 1D
@@ -433,6 +485,142 @@ public:
#endif // MFEM_USE_MPFR
void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
const real_t beta, IntegrationRule* ir)
{
/* The np-point Gauss-Jacobi quadrature rule is exact for polynomials of
degree 2np - 1 with weight function w(x) = (1-x)^alpha * x^beta. The
nodes are the zeros of the Jacobi polynomial P_{np}^{alpha,beta} and
the weights are
w_i = C / [(1 - x_i^2) * P'_{np}^{alpha,beta}(x_i)^2]
C = 2^{alpha + beta + 1} * Gamma(np + alpha + 1) * Gamma(np + beta + 1)
/ [Gamma(np + alpha + beta + 1) * Gamma(np + 1)].
The nodes are computed via nonlinear solve (Newton's method) with an
initial guess corresponding to Gatteschi's asymptotic expansions of the
Jacobi polynomial roots [1].
The current initial guess has been tested and performs well for
np <= 200 and -1 <= alpha, beta <= 4. For larger np, it may be necessary
utilize different initial guesses in the vicinity of x = -1,+1 [2].
[1] Gautschi, W., & Giordano, C. (2008). Luigi Gatteschis work on
asymptotics of special functions and their zeros. Numerical Algorithms,
49, 11-31.
[2] Hale, N., & Townsend, A. (2013). Fast and accurate computation of
Gauss--Legendre and Gauss--Jacobi quadrature nodes and weights.
SIAM Journal on Scientific Computing, 35(2), A652-A674.
*/
ir->SetSize(np);
ir->SetPointIndices();
ir->SetOrder(2*np - 1);
if (alpha <= -1.0 || beta <= -1.0)
{
MFEM_ABORT("Gauss-Jacobi quadrature only defined for alpha > -1 and beta > -1");
}
// Jacobi weight function is undefined whenever alpha <= -1 or beta <= -1
if (alpha > 4.0 || beta > 4.0)
{
MFEM_ABORT("Current Gauss-Jacobi quadrature implementation only tested for alpha <= 4 and beta <= 4");
}
// current asymptotic expansions for initial guess may perform poorly for large alpha, beta
switch (np)
{
case 1:
real_t x = (beta - alpha) / (alpha + beta + 2);
real_t w = pow(2, alpha + beta + 1) * tgamma(alpha + 2) * tgamma(
beta + 2) / (tgamma(alpha + beta + 2));
w = 0.5 * w / pow(2, alpha + beta);
// map weight to to [0,1], with additional 1/(2^(alpha + beta)) factor coming from mapping
// the weight (1-x)^alpha * (1+x)^beta to [0,1] as well.
ir->IntPoint(0).Set1w(0.5 * x + 0.5,
4.0 * w / ((1.0 - x*x) * (alpha + beta + 2) * (alpha + beta + 2)));
return;
}
#ifndef MFEM_USE_MPFR
const int n = np;
// common constants for Jacobi polynomials
real_t ab = alpha + beta;
real_t a2_minus_b2 = (alpha - beta) * (alpha + beta);
// roots of P^(alpha,beta)_n in the interval [-1,1]
for (int i = 1; i <= n; i++)
{
// rather than using Chebyshev points for initial guess, use Gatteschi's asymptotic expansion for roots of Jacobi
// polynomials
real_t n_ab_plus_1 = 2 * n + alpha + beta + 1;
real_t v = (2 * i + alpha - 0.5) * M_PI / n_ab_plus_1;
real_t theta = v + 1.0 / (n_ab_plus_1*n_ab_plus_1) * ((0.25 - alpha*alpha) *
1.0/tan(0.5*v) - (0.25 - beta*beta) * tan(0.5*v));
real_t z = cos(theta);
real_t pp, p1, dz, xi = 0.;
bool done = false;
while (1)
{
real_t p2 = 1;
p1 = ((alpha-beta) + (alpha + beta + 2) * z) / 2;
for (int j = 1; j <= n-1; j++)
{
real_t p3 = p2;
p2 = p1;
real_t jx2_ab = 2 * j + ab;
real_t an = (jx2_ab) * (jx2_ab + 2);
real_t bn = a2_minus_b2;
real_t cn = 2 * (j + alpha) * (j + beta) * (jx2_ab + 2) / (jx2_ab + 1);
real_t D = (jx2_ab + 1) / (2 * (j + 1) * (j + ab + 1) * (jx2_ab));
p1 = ((an * z + bn) * p2 - cn * p3) * D;
}
// p1 is Jacobi polynomial
pp = n * (alpha - beta - (2 * n + ab) * z) * p1 + 2 * (n + alpha) *
(n + beta) * p2;
pp = pp / ((2 * n + ab) * (1 - z*z));
// derivative of the Jacobi polynomial
if (done) { break; }
dz = p1/pp;
#ifdef MFEM_USE_SINGLE
if (std::abs(dz) < 1e-7)
#elif defined MFEM_USE_DOUBLE
if (std::abs(dz) < std::numeric_limits<real_t>::epsilon())
// this seems to cause trouble if we try std::abs(dz) < 1e-16
#else
MFEM_ABORT("Floating point type undefined");
// if (std::abs(dz) < 1e-16)
#endif
{
done = true;
xi = z - dz;
}
z -= dz;
}
real_t c0 = exp(lgamma(n + alpha + 1) - lgamma(n + ab + 1)) * exp(lgamma(
n + beta + 1) - lgamma(n + 1));
// ratio of gamma functions prone to overflow for large n, so compute logarithms
// of Gamma function instead, i.e. Gamma(a)/Gamma(b) = exp(lgamma(a) - lgamma(b))
ir->IntPoint(n-i).x = 0.5 * xi + 0.5;
ir->IntPoint(n-i).weight = 0.5 * c0 * pow(2.0,
ab + 1) / ((1.0 - xi*xi)*pp*pp) / pow(2, ab);
// map nodes and weights to the interval [0,1]
}
#else // MFEM_USE_MPFR is defined
MFEM_ABORT("MPFR implementation of Gauss-Jacobi quadrature not defined yet");
#endif // MFEM_USE_MPFR
}
void QuadratureFunctions1D::GaussLegendre(const int np, IntegrationRule* ir)
{
ir->SetSize(np);
@@ -2362,6 +2550,194 @@ IntegrationRule *IntegrationRules::CubeIntegrationRule(int Order)
return CubeIntRules[Order];
}
StroudIntegrationRules StroudIntRules;
StroudIntegrationRules::StroudIntegrationRules()
{
const MemoryType h_mt = MemoryType::HOST;
SquareStroudIntRules.SetSize(32, h_mt);
SquareStroudIntRules = NULL;
TriangleStroudIntRules.SetSize(32, h_mt);
TriangleStroudIntRules = NULL;
CubeStroudIntRules.SetSize(32, h_mt);
CubeStroudIntRules = NULL;
TetrahedronStroudIntRules.SetSize(32, h_mt);
TetrahedronStroudIntRules = NULL;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
IntRuleLocks.SetSize(Geometry::NUM_GEOMETRIES, h_mt);
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
{
omp_init_lock(&IntRuleLocks[i]);
}
#endif
}
const IntegrationRule &StroudIntegrationRules::Get(int GeomType, int Order)
{
Array<IntegrationRule *> *ir_array = NULL;
switch (GeomType)
{
case Geometry::TRIANGLE: ir_array = &TriangleStroudIntRules; break;
case Geometry::TETRAHEDRON: ir_array = &TetrahedronStroudIntRules; break;
case Geometry::INVALID:
case Geometry::NUM_GEOMETRIES:
MFEM_ABORT("Unknown type of reference element!");
default:
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
}
if (Order < 0)
{
Order = 0;
}
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
omp_set_lock(&IntRuleLocks[GeomType]);
#endif
if (!HaveIntRule(*ir_array, Order))
{
IntegrationRule *ir = GenerateIntegrationRule(GeomType, Order);
#ifdef MFEM_DEBUG
int RealOrder = Order;
while (RealOrder+1 < ir_array->Size() && (*ir_array)[RealOrder+1] == ir)
{
RealOrder++;
}
MFEM_VERIFY(RealOrder == ir->GetOrder(), "internal error");
#else
MFEM_CONTRACT_VAR(ir);
#endif
}
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
omp_unset_lock(&IntRuleLocks[GeomType]);
#endif
return *(*ir_array)[Order];
}
void StroudIntegrationRules::DeleteIntRuleArray(
Array<IntegrationRule *> &ir_array) const
{
// Many of the intrules have multiple contiguous copies in the ir_array
// so we have to be careful to not delete them twice.
IntegrationRule *ir = NULL;
for (int i = 0; i < ir_array.Size(); i++)
{
if (ir_array[i] != NULL && ir_array[i] != ir)
{
ir = ir_array[i];
delete ir;
}
}
}
StroudIntegrationRules::~StroudIntegrationRules()
{
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
{
omp_destroy_lock(&IntRuleLocks[i]);
}
#endif
DeleteIntRuleArray(SquareStroudIntRules);
DeleteIntRuleArray(TriangleStroudIntRules);
DeleteIntRuleArray(CubeStroudIntRules);
DeleteIntRuleArray(TetrahedronStroudIntRules);
}
IntegrationRule *StroudIntegrationRules::GenerateIntegrationRule(int GeomType,
int Order)
{
switch (GeomType)
{
case Geometry::TRIANGLE:
return TriangleStroudIntegrationRule(Order);
case Geometry::TETRAHEDRON:
return TetrahedronStroudIntegrationRule(Order);
case Geometry::INVALID:
case Geometry::NUM_GEOMETRIES:
MFEM_ABORT("Unknown type of reference element!");
default:
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
}
return NULL;
}
/* Integration rule in reference triangle according to tensor product Gauss-Jacobi rule.
The nodes and weights are used in the original form defined on the reference
square to evaluate the component 1D basis functions. Mapping to the reference
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
IntegrationRule *StroudIntegrationRules::TriangleStroudIntegrationRule(
int Order)
{
int RealOrder = GetSegmentRealOrder(Order);
// Order is one of {RealOrder-1,RealOrder}
// if (!HaveIntRule(SegmentIntRules, RealOrder))
// {
// SegmentIntegrationRule(RealOrder);
// }
IntegrationRule ir_0_0;
// Gauss-Jacobi is exact for 2*n-1
int n = RealOrder/2 + 1;
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
IntegrationRule ir_1_0;
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
AllocIntRule(TriangleStroudIntRules, RealOrder); // RealOrder >= Order
// create rule in unit square
TriangleStroudIntRules[RealOrder-1] =
TriangleStroudIntRules[RealOrder] =
new IntegrationRule(ir_1_0, ir_0_0);
// map rule to reference triangle
// TriangleStroudIntRules[RealOrder-1]->DuffyTrans(2);
*TriangleStroudIntRules[RealOrder-1] =
DuffyTrans(*TriangleStroudIntRules[RealOrder-1], 2);
return TriangleStroudIntRules[Order];
}
/* Integration rule in reference tetrahedron according to tensor product Gauss-Jacobi rule.
The nodes and weights are used in the original form defined on the reference
square to evaluate the component 1D basis functions. Mapping to the reference
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
IntegrationRule *StroudIntegrationRules::TetrahedronStroudIntegrationRule(
int Order)
{
int RealOrder = GetSegmentRealOrder(Order);
// Order is one of {RealOrder-1,RealOrder}
IntegrationRule ir_0_0;
int n = RealOrder/2 + 1;
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
IntegrationRule ir_1_0;
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
IntegrationRule ir_2_0;
QuadratureFunctions1D::GaussJacobi(n, 2.0, 0.0, &ir_2_0);
AllocIntRule(TetrahedronStroudIntRules, RealOrder); // RealOrder >= Order
// create rule in unit cube
TetrahedronStroudIntRules[RealOrder-1] =
TetrahedronStroudIntRules[RealOrder] =
new IntegrationRule(ir_2_0, ir_1_0, ir_0_0);
// map rule to reference tetrahedron
// TetrahedronStroudIntRules[RealOrder-1]->DuffyTrans(3);
*TetrahedronStroudIntRules[RealOrder-1] =
DuffyTrans(*TetrahedronStroudIntRules[RealOrder-1], 3);
return TetrahedronStroudIntRules[Order];
}
IntegrationRule& NURBSMeshRules::GetElementRule(const int elem,
const int patch, const int *ijk,
Array<const KnotVector*> const& kv) const
+68
View File
@@ -269,6 +269,13 @@ public:
/// applying this rule on each knot interval.
IntegrationRule* ApplyToKnotIntervals(KnotVector const& kv) const;
/** @brief Returns an integration rule such that the new IntegrationPoints
* are re-ordered based on @a ordering.
*
* @details In the new integration rule, ip_new[i] = ip_old[ordering[i]]
*/
IntegrationRule Reorder(const Array<int> &ordering) const;
/// Destroys an IntegrationRule object
~IntegrationRule() { }
};
@@ -378,6 +385,8 @@ public:
These methods calculate the actual points and weights for the different
types of quadrature rules. */
///@{
static void GaussJacobi(const int np, const real_t alpha, const real_t beta,
IntegrationRule* ir);
static void GaussLegendre(const int np, IntegrationRule* ir);
static void GaussLobatto(const int np, IntegrationRule *ir);
static void OpenUniform(const int np, IntegrationRule *ir);
@@ -487,12 +496,71 @@ public:
~IntegrationRules();
};
/// Container class for integration rules
class StroudIntegrationRules
{
private:
Array<IntegrationRule *> SquareStroudIntRules;
Array<IntegrationRule *> TriangleStroudIntRules;
Array<IntegrationRule *> CubeStroudIntRules;
Array<IntegrationRule *> TetrahedronStroudIntRules;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
Array<omp_lock_t> IntRuleLocks;
#endif
void AllocIntRule(Array<IntegrationRule *> &ir_array, int Order) const
{
if (ir_array.Size() <= Order)
{
ir_array.SetSize(Order + 1, NULL);
}
}
bool HaveIntRule(Array<IntegrationRule *> &ir_array, int Order) const
{
return (ir_array.Size() > Order && ir_array[Order] != NULL);
}
int GetSegmentRealOrder(int Order) const
{
return Order | 1; // valid for all quad_type's
}
void DeleteIntRuleArray(Array<IntegrationRule *> &ir_array) const;
/// The following methods allocate new IntegrationRule objects without
/// checking if they already exist. To avoid memory leaks use
/// IntegrationRules::Get(int GeomType, int Order) instead.
IntegrationRule *GenerateIntegrationRule(int GeomType, int Order);
IntegrationRule *TriangleStroudIntegrationRule(int Order);
IntegrationRule *TetrahedronStroudIntegrationRule(int Order);
public:
/// Sets initial sizes for the integration rule arrays, but rules
/// are defined the first time they are requested with the Get method.
explicit StroudIntegrationRules();
/// Returns a Stroud integration rule for given GeomType and Order.
const IntegrationRule &Get(int GeomType, int Order);
/// Destroys an StroudIntegrationRules object
~StroudIntegrationRules();
};
/// A global object with all integration rules (defined in intrules.cpp)
extern MFEM_EXPORT IntegrationRules IntRules;
/// A global object with all refined integration rules
extern MFEM_EXPORT IntegrationRules RefinedIntRules;
/// A global object with all Stroud integration rules (defined in intrules.cpp)
extern MFEM_EXPORT StroudIntegrationRules StroudIntRules;
/// Duffy Transformation of 2D and 3D tensor product rules of the form
/// $X(t) = \sum_{i=1}^{d+1} \lambda_i(t) * x_i$, where $x_i$ are the vertices
/// of the simplex and $\lambda_i = t_i * (1-\lambda_1-...-\lambda_{i-1})$, with
/// $t$ being the coordinates in the unit square/cube. This function is used only
/// in the partial assembly of Bernstein elements on simplices and does NOT
/// modify the quadrature weights.
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim);
}
#endif
+9 -2
View File
@@ -83,7 +83,7 @@ constexpr int SetMaxOf(int n) { return NextMultipleOf<4>(n); }
#endif // CUDA/HIP && DEVICE_COMPILE
/// Load 2D matrix into shared memory
template <int MQ1>
template <int MQ1, bool TRANSPOSE = false>
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
const real_t *M, real_t (*N)[MQ1])
{
@@ -91,7 +91,14 @@ inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
N[dy][qx] = M[dy * q1d + qx];
if constexpr (TRANSPOSE)
{
N[dy][qx] = M[qx * d1d + dy];
}
else
{
N[dy][qx] = M[dy * q1d + qx];
}
}
}
MFEM_SYNC_THREAD;
+7
View File
@@ -471,6 +471,13 @@ void VectorBoundaryLFIntegrator::AssembleRHSElementVect(
}
}
VectorFEDomainLFIntegrator::VectorFEDomainLFIntegrator(
VectorCoefficient &F, const IntegrationRule *ir)
: DeltaLFIntegrator(F, ir), QF(F)
{
static Kernels kernels{};
}
void VectorFEDomainLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
+36 -2
View File
@@ -369,8 +369,8 @@ private:
Vector vec;
public:
VectorFEDomainLFIntegrator(VectorCoefficient &F)
: DeltaLFIntegrator(F), QF(F) { }
VectorFEDomainLFIntegrator(VectorCoefficient &F,
const IntegrationRule *ir = nullptr);
void AssembleRHSElementVect(const FiniteElement &el,
ElementTransformation &Tr,
@@ -387,6 +387,40 @@ public:
Vector &b) override;
using LinearFormIntegrator::AssembleRHSElementVect;
/// @param ne number of elements
/// @param markers array where entry markers[e] == 0 to skip assembly over
/// element e element
/// @param jac Spatial Jacobians evaluated at all quadrature points
/// @param weights 1D quadrature weights
/// @param testBO 1D open basis test functions
/// @param testBC 1D closed basis test functions
/// @param coeff coefficient values evaluated at quadrature points, possibly
/// compressed.
/// @param d number of 1D closed dofs
/// @param q number of 1D quadrature points
using AssembleKernelType = void (*)(const int NE, const Array<int> &markers,
const Vector &jac,
const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC,
const Vector &coeff, Vector &y,
const int testd1d, const int q1d);
/// parameters: test_fetype, ndims, test_d1d, q1d
MFEM_REGISTER_KERNELS(AssembleKernels, AssembleKernelType,
(FiniteElement::DerivType, int, int, int));
struct Kernels
{
Kernels();
};
template <FiniteElement::DerivType TestType, int DIM, int TEST_D1D, int Q1D>
static void AddSpecialization()
{
AssembleKernels::Specialization<TestType, DIM, TEST_D1D, Q1D>::Add();
}
};
/// $ (Q, \mathrm{curl}(v))_{\Omega} $ for Nedelec Elements
+4 -4
View File
@@ -284,12 +284,12 @@ GeometricMultigrid::GeometricMultigrid(
ownedProlongations.SetSize(nlevels - 1);
ownedProlongations = have_ess_bdr;
if (have_ess_bdr)
essentialTrueDofs.SetSize(nlevels);
for (int level = 0; level < nlevels; ++level)
{
essentialTrueDofs.SetSize(nlevels);
for (int level = 0; level < nlevels; ++level)
essentialTrueDofs[level] = new Array<int>;
if (have_ess_bdr)
{
essentialTrueDofs[level] = new Array<int>;
fespaces.GetFESpaceAtLevel(level).GetEssentialTrueDofs(
ess_bdr, *essentialTrueDofs[level]);
}
+1 -2
View File
@@ -187,8 +187,7 @@ public:
/// mesh boundary element attributes that define the essential DOFs.
///
/// If @a ess_bdr is empty, or all its entries are 0, then no essential
/// boundary conditions are imposed and the protected array essentialTrueDofs
/// remains empty.
/// boundary conditions are imposed.
GeometricMultigrid(const FiniteElementSpaceHierarchy& fespaces_,
const Array<int> &ess_bdr);
+11
View File
@@ -100,6 +100,17 @@ PANonlinearFormExtension::Gradient::Gradient(const PANonlinearFormExtension &e):
void PANonlinearFormExtension::Gradient::AssembleGrad(const Vector &g)
{
if (DeviceCanUseCeed())
{
for (int i = 0; i < ext.dnfi.Size(); ++i)
{
MFEM_VERIFY(dynamic_cast<VectorConvectionNLFIntegrator *>
(ext.dnfi[i]) == nullptr,
"VectorConvectionNLFIntegrator PA gradients are not supported "
"with the libCEED backend");
}
}
ext.elemR->Mult(g, ext.xe);
for (int i = 0; i < ext.dnfi.Size(); ++i)
{
+70
View File
@@ -954,4 +954,74 @@ void SkewSymmetricVectorConvectionNLFIntegrator::AssembleElementGrad(
}
}
void ConvectiveVectorConvectionNLFIntegrator::AssemblePA(
const FiniteElementSpace &)
{
MFEM_ABORT("ConvectiveVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void ConvectiveVectorConvectionNLFIntegrator::AssembleGradPA(
const Vector &, const FiniteElementSpace &)
{
MFEM_ABORT("ConvectiveVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void ConvectiveVectorConvectionNLFIntegrator::AddMultPA(
const Vector &, Vector &) const
{
MFEM_ABORT("ConvectiveVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void ConvectiveVectorConvectionNLFIntegrator::AddMultGradPA(
const Vector &, Vector &) const
{
MFEM_ABORT("ConvectiveVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void ConvectiveVectorConvectionNLFIntegrator::AssembleGradDiagonalPA(
Vector &) const
{
MFEM_ABORT("ConvectiveVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void SkewSymmetricVectorConvectionNLFIntegrator::AssemblePA(
const FiniteElementSpace &)
{
MFEM_ABORT("SkewSymmetricVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void SkewSymmetricVectorConvectionNLFIntegrator::AssembleGradPA(
const Vector &, const FiniteElementSpace &)
{
MFEM_ABORT("SkewSymmetricVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void SkewSymmetricVectorConvectionNLFIntegrator::AddMultPA(
const Vector &, Vector &) const
{
MFEM_ABORT("SkewSymmetricVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void SkewSymmetricVectorConvectionNLFIntegrator::AddMultGradPA(
const Vector &, Vector &) const
{
MFEM_ABORT("SkewSymmetricVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
void SkewSymmetricVectorConvectionNLFIntegrator::AssembleGradDiagonalPA(
Vector &) const
{
MFEM_ABORT("SkewSymmetricVectorConvectionNLFIntegrator does not support "
"partial assembly; use VectorConvectionNLFIntegrator");
}
}
+70 -8
View File
@@ -18,6 +18,7 @@
#include "fespace.hpp"
#include "ceed/interface/operator.hpp"
#include "integrator.hpp"
#include "kernel_dispatch.hpp"
namespace mfem
{
@@ -384,15 +385,17 @@ private:
DenseMatrix dshape, dshapex, EF, gradEF, ELV, elmat_comp;
Vector shape;
// PA extension
Vector pa_data;
int dim, ne, nq, d1d, q1d;
Vector pa_adj, pa_u;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
public:
VectorConvectionNLFIntegrator(Coefficient &q): Q(&q) { }
struct Kernels { Kernels(); };
VectorConvectionNLFIntegrator() = default;
VectorConvectionNLFIntegrator(Coefficient &q): Q(&q) { static Kernels kernels; }
VectorConvectionNLFIntegrator() { static Kernels kernels; }
static const IntegrationRule &GetRule(const FiniteElement &fe,
const ElementTransformation &T);
@@ -411,12 +414,55 @@ public:
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleMF(const FiniteElementSpace &fes) override;
void AssembleGradPA(const Vector &x, const FiniteElementSpace &fes) override;
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultMF(const Vector &x, Vector &y) const override;
using AddMultPAType =
void(*)(const int ne, const real_t *B, const real_t *G, const real_t *A,
const real_t *x, real_t *y,
const int d1d, const int q1d);
MFEM_REGISTER_KERNELS(AddMultPAKernels, AddMultPAType, (int, int, int));
void AddMultGradPA(const Vector &x, Vector &y) const override;
using AddMultGradPAType =
void(*)(const int ne, const real_t *B, const real_t *G, const real_t *A,
const real_t *u, const real_t *x, real_t *y,
const int d1d, const int q1d);
MFEM_REGISTER_KERNELS(AddMultGradPA2D, AddMultGradPAType, (int, int));
MFEM_REGISTER_KERNELS(AddMultGradPA3D, AddMultGradPAType, (int, int));
void AssembleGradDiagonalPA(Vector &) const override;
using GradDiagPAType =
void (*)(const int ne, const real_t *B, const real_t *G, const real_t *A,
const real_t *u, real_t *y,
const int d1d, const int q1d);
MFEM_REGISTER_KERNELS(GradDiagPA2D, GradDiagPAType, (int, int));
MFEM_REGISTER_KERNELS(GradDiagPA3D, GradDiagPAType, (int, int));
template <int DIM, int D1D, int Q1D>
static void AddSpecialization()
{
AddMultPAKernels::Specialization<DIM, D1D, Q1D>::Add();
if constexpr (DIM == 2)
{
AddMultGradPA2D::Specialization<D1D, Q1D>::Add();
GradDiagPA2D::Specialization<D1D, Q1D>::Add();
}
else if constexpr (DIM == 3)
{
AddMultGradPA3D::Specialization<D1D, Q1D>::Add();
GradDiagPA3D::Specialization<D1D, Q1D>::Add();
}
}
void AssembleMF(const FiniteElementSpace &fes) override;
void AddMultMF(const Vector &x, Vector &y) const override;
protected:
const IntegrationRule* GetDefaultIntegrationRule(
@@ -430,7 +476,8 @@ protected:
/** This class is used to assemble the convective form of the nonlinear term
arising in the Navier-Stokes equations $(u \cdot \nabla v, w )$ */
arising in the Navier-Stokes equations $(u \cdot \nabla v, w )$.
Partial assembly is not supported; use VectorConvectionNLFIntegrator. */
class ConvectiveVectorConvectionNLFIntegrator :
public VectorConvectionNLFIntegrator
{
@@ -448,12 +495,20 @@ public:
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat) override;
using NonlinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleGradPA(const Vector &x, const FiniteElementSpace &fes) override;
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultGradPA(const Vector &x, Vector &y) const override;
void AssembleGradDiagonalPA(Vector &diag) const override;
};
/** This class is used to assemble the skew-symmetric form of the nonlinear term
arising in the Navier-Stokes equations
$.5*(u \cdot \nabla v, w ) - .5*(u \cdot \nabla w, v )$ */
$.5*(u \cdot \nabla v, w ) - .5*(u \cdot \nabla w, v )$.
Partial assembly is not supported; use VectorConvectionNLFIntegrator. */
class SkewSymmetricVectorConvectionNLFIntegrator :
public VectorConvectionNLFIntegrator
{
@@ -471,6 +526,13 @@ public:
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat) override;
using NonlinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleGradPA(const Vector &x, const FiniteElementSpace &fes) override;
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultGradPA(const Vector &x, Vector &y) const override;
void AssembleGradDiagonalPA(Vector &diag) const override;
};
}
+354 -63
View File
@@ -10,6 +10,7 @@
// CONTRIBUTING.md for details.
#include "particleset.hpp"
#include "../general/forall.hpp"
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
@@ -225,6 +226,7 @@ void ParticleSet::AddParticles(const Array<IDType> &new_ids,
}
}
// Add new ids
ids.HostReadWrite();
ids.Append(new_ids);
// Update data
@@ -244,6 +246,102 @@ void ParticleSet::AddParticles(const Array<IDType> &new_ids,
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
/// \cond DO_NOT_DOCUMENT
// Static helper: gather selected particle-vector entries into a compact buffer.
// nvcc does not allow extended host/device lambdas in non-public members.
static void GatherParticleVectorDevice(const ParticleVector &pv,
const Array<int> &send_idxs,
Vector &send_data,
int nsend)
{
const int vdim = pv.GetVDim();
const int ordering = pv.GetOrdering();
const int num_particles = pv.GetNumParticles();
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
send_data.SetSize(nsend*vdim);
real_t *d_send_data =
send_data.GetMemory().Write(device_mc, send_data.Size());
const real_t *d_src = pv.GetMemory().Read(device_mc, pv.Size());
const int *d_send_idxs = send_idxs.GetMemory().Read(device_mc, nsend);
mfem::forall(nsend, [=] MFEM_HOST_DEVICE (int i)
{
const int p = d_send_idxs[i];
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
const int stride = (ordering == Ordering::byVDIM) ? 1 : num_particles;
for (int c = 0; c < vdim; c++)
{
d_send_data[i*vdim + c] = d_src[offset + c*stride];
}
});
}
// Static helper: gather selected tag values into a compact buffer.
// nvcc does not allow extended host/device lambdas in non-public members.
static void GatherParticleTagsDevice(const Array<int> &tag,
const Array<int> &send_idxs,
Array<int> &send_tag,
int nsend)
{
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
send_tag.SetSize(nsend);
int *d_send_tag = send_tag.GetMemory().Write(device_mc, nsend);
const int *d_tag = tag.GetMemory().Read(device_mc, tag.Size());
const int *d_send_idxs = send_idxs.GetMemory().Read(device_mc, nsend);
mfem::forall(nsend, [=] MFEM_HOST_DEVICE (int i)
{
d_send_tag[i] = d_tag[d_send_idxs[i]];
});
}
// Static helper: scatter compact particle-vector entries to particle storage.
// nvcc does not allow extended host/device lambdas in non-public members.
static void ScatterParticleVectorDevice(ParticleVector &pv,
const Vector &recv_data,
const Array<int> &recv_locs,
int nrecv)
{
const int vdim = pv.GetVDim();
const int ordering = pv.GetOrdering();
const int num_particles = pv.GetNumParticles();
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
const real_t *d_recv_data =
recv_data.GetMemory().Read(device_mc, recv_data.Size());
const int *d_recv_locs = recv_locs.GetMemory().Read(device_mc, nrecv);
real_t *d_dst = pv.GetMemory().ReadWrite(device_mc, pv.Size());
mfem::forall(nrecv, [=] MFEM_HOST_DEVICE (int i)
{
const int p = d_recv_locs[i];
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
const int stride = (ordering == Ordering::byVDIM) ? 1 : num_particles;
for (int c = 0; c < vdim; c++)
{
d_dst[offset + c*stride] = d_recv_data[i*vdim + c];
}
});
}
// Static helper: scatter compact tag values to particle storage.
// nvcc does not allow extended host/device lambdas in non-public members.
static void ScatterParticleTagsDevice(Array<int> &tag,
const Array<int> &recv_tag,
const Array<int> &recv_locs,
int nrecv)
{
const MemoryClass device_mc = Device::GetDeviceMemoryClass();
const int *d_recv_tag = recv_tag.GetMemory().Read(device_mc, nrecv);
const int *d_recv_locs = recv_locs.GetMemory().Read(device_mc, nrecv);
int *d_tag = tag.GetMemory().ReadWrite(device_mc, tag.Size());
mfem::forall(nrecv, [=] MFEM_HOST_DEVICE (int i)
{
d_tag[d_recv_locs[i]] = d_recv_tag[i];
});
}
template<size_t NBytes>
void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
const Array<int> &send_idxs,
@@ -266,37 +364,108 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
array_init(parr_t, &gsl_arr, send_idxs.Size());
pdata_arr = (parr_t*) gsl_arr.ptr;
int nparticles = pset.GetNParticles();
int nsend = send_idxs.Size();
gsl_arr.n = send_idxs.Size();
const int *h_send_idxs_initial = send_idxs.HostRead();
const IDType *h_ids = pset.GetIDs().HostRead();
for (int i = 0; i < send_idxs.Size(); i++)
{
parr_t &pdata = pdata_arr[i];
pdata.id = pset.GetIDs()[send_idxs[i]];
pdata.id = h_ids[h_send_idxs_initial[i]];
}
// Copy particle data directly into pdata
size_t counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
// Pack coords and fields into the GSLIB send buffer. Device-resident data
// is first gathered into a compact device buffer so that only selected
// particles are copied back to host. Host-resident data is packed directly.
int max_vdim = pset.Coords().GetVDim();
for (int f = 0; f < pset.GetNFields(); f++)
{
int f_vdim = pset.Field(f).GetVDim();
if (f_vdim > max_vdim) { max_vdim = f_vdim; }
}
Vector send_data;
Array<int> send_tag;
if (Device::IsEnabled())
{
send_data.SetSize(nsend * max_vdim); // allocate max size over all fields
send_tag.SetSize(nsend);
}
size_t counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
{
const ParticleVector &pv = f == -1 ? pset.Coords() : pset.Field(f);
const int vdim = pv.GetVDim();
const int ordering = pv.GetOrdering();
const int num_particles = pv.GetNumParticles();
const bool use_dev = Device::IsEnabled() && pv.UseDevice();
if (use_dev)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
for (int c = 0; c < pv.GetVDim(); c++)
GatherParticleVectorDevice(pv, send_idxs, send_data, nsend);
const real_t *h_send_data = send_data.HostRead();
for (int i = 0; i < nsend; i++)
{
std::memcpy(pdata.data.data() + counter, &pv(send_idxs[i], c),
sizeof(real_t));
counter += sizeof(real_t);
std::memcpy(pdata_arr[i].data.data() + counter,
h_send_data + i*vdim, vdim * sizeof(real_t));
}
}
else
{
const real_t *h_src = pv.HostRead();
const int *h_send_idxs = send_idxs.HostRead();
for (int i = 0; i < nsend; i++)
{
parr_t &pdata = pdata_arr[i];
const int p = h_send_idxs[i];
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
const int stride = (ordering == Ordering::byVDIM) ? 1 :
num_particles;
for (int c = 0; c < vdim; c++)
{
std::memcpy(pdata.data.data() + counter + c*sizeof(real_t),
h_src + offset + c*stride, sizeof(real_t));
}
}
}
// Copy tags
for (int t = 0; t < pset.GetNTags(); t++)
{
Array<int> &tag_arr = pset.Tag(t);
std::memcpy(pdata.data.data() + counter, &tag_arr[send_idxs[i]],
sizeof(int));
counter += sizeof(int);
}
counter += vdim*sizeof(real_t);
}
int nparticles = pset.GetNParticles();
int nsend = send_idxs.Size();
// Pack tags after all real_t data. Each tag uses the same selective
// device gather path when its Array is device-resident.
for (int t = 0; t < pset.GetNTags(); t++)
{
const Array<int> &tag = pset.Tag(t);
const size_t tag_counter = counter + t*sizeof(int);
const bool use_dev = Device::IsEnabled() && tag.UseDevice();
if (use_dev)
{
GatherParticleTagsDevice(tag, send_idxs, send_tag, nsend);
const int *h_send_tag = send_tag.HostRead();
for (int i = 0; i < nsend; i++)
{
std::memcpy(pdata_arr[i].data.data() + tag_counter,
h_send_tag + i, sizeof(int));
}
}
else
{
const int *h_tag = tag.HostRead();
const int *h_send_idxs = send_idxs.HostRead();
for (int i = 0; i < nsend; i++)
{
std::memcpy(pdata_arr[i].data.data() + tag_counter,
h_tag + h_send_idxs[i], sizeof(int));
}
}
}
// Transfer particles
sarray_transfer_ext(parr_t, &gsl_arr, send_ranks.GetData(),
@@ -304,11 +473,20 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
// Make sure we have enough space for received particles
int nrecv = (int) gsl_arr.n;
Vector recv_data;
Array<int> recv_tag;
if (Device::IsEnabled())
{
recv_data.SetSize(nrecv * max_vdim);
recv_tag.SetSize(nrecv);
}
int ndelete = nsend - nrecv;
if (ndelete > 0)
{
// Remove unneeded particles
auto datap = const_cast<int*>(send_idxs.GetData());
auto datap = const_cast<int*>(send_idxs.HostRead());
Array<int> delete_idxs(datap + nrecv, ndelete);
pset.RemoveParticles(delete_idxs);
}
@@ -319,47 +497,133 @@ void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
pdata_arr = (parr_t*) gsl_arr.ptr;
// Add newly-recvd data directly to active state
// Make a list of new IDs to add
int num_new = nrecv > nsend ? nrecv - nsend : 0;
Array<IDType> new_ids(num_new);
for (int i = 0; i < num_new; i++)
{
new_ids[i] = pdata_arr[nsend + i].id;
}
// Add particles in batch
Array<int> new_indices;
if (num_new > 0)
{
pset.AddParticles(new_ids, &new_indices);
}
// Map each received packet to the local particle slot it updates.
Array<int> recv_locs(nrecv);
int *h_recv_locs = recv_locs.HostWrite();
const int *h_send_idxs_recv = send_idxs.HostRead();
for (int i = 0; i < nrecv; i++)
{
parr_t &pdata = pdata_arr[i];
IDType id = pdata.id;
int new_loc_idx;
if (i < nsend) // update existing particle
{
new_loc_idx = send_idxs[i];
pset.UpdateID(new_loc_idx, id);
h_recv_locs[i] = h_send_idxs_recv[i];
pset.UpdateID(h_recv_locs[i], pdata.id);
}
else
{
// add new particle
Array<int> idx_temp;
pset.AddParticles(Array<IDType>({id}), &idx_temp);
new_loc_idx = idx_temp[0]; // Get index of newly-added particle
h_recv_locs[i] = new_indices[i - nsend];
}
}
size_t counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
// Unpack coords and fields from GSLIB host packets. Device-resident
// destinations use a compact host buffer followed by a device scatter.
size_t recv_counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
const int vdim = pv.GetVDim();
const int ordering = pv.GetOrdering();
const int num_particles = pv.GetNumParticles();
const bool use_dev = Device::IsEnabled() && pv.UseDevice();
if (use_dev)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
for (int c = 0; c < pv.GetVDim(); c++)
recv_data.SetSize(nrecv*vdim);
real_t *h_recv_data = recv_data.HostWrite();
for (int i = 0; i < nrecv; i++)
{
real_t& val = pv(new_loc_idx, c);
std::memcpy(&val, pdata.data.data() + counter, sizeof(real_t));
counter += sizeof(real_t);
std::memcpy(h_recv_data + i*vdim,
pdata_arr[i].data.data() + recv_counter,
vdim*sizeof(real_t));
}
ScatterParticleVectorDevice(pv, recv_data, recv_locs, nrecv);
}
else
{
real_t *h_dst = pv.HostReadWrite();
const int *h_recv_locs_read = recv_locs.HostRead();
for (int i = 0; i < nrecv; i++)
{
parr_t &pdata = pdata_arr[i];
const int p = h_recv_locs_read[i];
const int offset = (ordering == Ordering::byVDIM) ? p * vdim : p;
const int stride = (ordering == Ordering::byVDIM) ? 1 :
num_particles;
for (int c = 0; c < vdim; c++)
{
std::memcpy(h_dst + offset + c*stride,
pdata.data.data() + recv_counter + c*sizeof(real_t),
sizeof(real_t));
}
}
}
for (int t = 0; t < pset.GetNTags(); t++)
recv_counter += vdim*sizeof(real_t);
}
// Unpack tags after all real_t data, using the same compact scatter path
// for device-resident tag arrays.
for (int t = 0; t < pset.GetNTags(); t++)
{
Array<int> &tag = pset.Tag(t);
const size_t tag_counter = recv_counter + t*sizeof(int);
const bool use_dev = Device::IsEnabled() && tag.UseDevice();
if (use_dev)
{
Array<int> &tag_arr = pset.Tag(t);
std::memcpy(&tag_arr[new_loc_idx],
pdata.data.data() + counter, sizeof(int));
counter += sizeof(int);
recv_tag.SetSize(nrecv);
int *h_recv_tag = recv_tag.HostWrite();
for (int i = 0; i < nrecv; i++)
{
std::memcpy(h_recv_tag + i,
pdata_arr[i].data.data() + tag_counter, sizeof(int));
}
ScatterParticleTagsDevice(tag, recv_tag, recv_locs, nrecv);
}
else
{
int *h_tag = tag.HostReadWrite();
const int *h_recv_locs_read = recv_locs.HostRead();
for (int i = 0; i < nrecv; i++)
{
std::memcpy(h_tag + h_recv_locs_read[i],
pdata_arr[i].data.data() + tag_counter, sizeof(int));
}
}
}
array_free(&gsl_arr);
// Restore Device validity if needed
for (int f = -1; f < pset.GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
pv.ReadWrite(pv.UseDevice());
}
for (int t = 0; t < pset.GetNTags(); t++)
{
Array<int> &tag_arr = pset.Tag(t);
if (tag_arr.UseDevice()) { tag_arr.ReadWrite(true); }
}
}
template<size_t NBytes>
@@ -526,11 +790,14 @@ ParticleSet::ParticleSet(int id_stride_, IDType id_counter_, int num_particles,
int dim, Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
const Array<const char*> &tag_names_,
bool use_device)
: id_stride(id_stride_),
id_counter(id_counter_),
coords(dim, coords_ordering)
{
if (use_device) { coords.UseDevice(true); }
// Initialize fields
for (int f = 0; f < field_vdims.Size(); f++)
{
@@ -580,21 +847,22 @@ bool ParticleSet::IsValidParticle(const Particle &p) const
}
ParticleSet::ParticleSet(int num_particles, int dim,
Ordering::Type coords_ordering)
Ordering::Type coords_ordering,
bool use_device)
: ParticleSet(1, 0, num_particles, dim, coords_ordering, Array<int>(),
Array<Ordering::Type>(), Array<const char*>(), 0,
Array<const char*>())
Array<const char*>(), use_device)
{
}
ParticleSet::ParticleSet(int num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering)
Ordering::Type all_ordering, bool use_device)
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
GetEmptyNameArray(field_vdims.Size()), num_tags,
GetEmptyNameArray(num_tags))
GetEmptyNameArray(num_tags), use_device)
{
}
@@ -602,11 +870,11 @@ ParticleSet::ParticleSet(int num_particles, int dim,
const Array<int> &field_vdims, const Array<const
char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_,
Ordering::Type all_ordering)
Ordering::Type all_ordering, bool use_device)
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
field_names_, num_tags,
tag_names_)
tag_names_, use_device)
{
}
@@ -616,9 +884,9 @@ ParticleSet::ParticleSet(int num_particles, int dim,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
const Array<const char*> &tag_names_, bool use_device)
: ParticleSet(1, 0, num_particles, dim, coords_ordering, field_vdims,
field_orderings, field_names_, num_tags, tag_names_)
field_orderings, field_names_, num_tags, tag_names_, use_device)
{
}
@@ -627,21 +895,21 @@ ParticleSet::ParticleSet(int num_particles, int dim,
#ifdef MFEM_USE_MPI
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering)
Ordering::Type coords_ordering, bool use_device)
: ParticleSet(comm_, rank_num_particles, dim, coords_ordering, Array<int>(),
Array<Ordering::Type>(), Array<const char*>(), 0,
Array<const char*>())
Array<const char*>(), use_device)
{
};
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering)
Ordering::Type all_ordering, bool use_device)
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
GetEmptyNameArray(field_vdims.Size()), num_tags,
GetEmptyNameArray(num_tags))
GetEmptyNameArray(num_tags), use_device)
{
}
@@ -650,11 +918,11 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, const Array<const
char*> &field_names_,
int num_tags, const Array<const char*> &tag_names_,
Ordering::Type all_ordering)
Ordering::Type all_ordering, bool use_device)
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
field_names_, num_tags,
tag_names_)
tag_names_, use_device)
{
}
@@ -664,7 +932,7 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
const Array<const char*> &tag_names_, bool use_device)
: ParticleSet(GetSize(comm_), (IDType)GetRank(comm_),
rank_num_particles,
dim,
@@ -673,7 +941,7 @@ ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
field_orderings,
field_names_,
num_tags,
tag_names_)
tag_names_, use_device)
{
comm = comm_;
#ifdef MFEM_USE_GSLIB
@@ -705,6 +973,7 @@ int ParticleSet::AddField(int vdim, Ordering::Type field_ordering,
}
fields.emplace_back(std::make_unique<ParticleVector>(vdim, field_ordering,
GetNParticles()));
if (coords.UseDevice()) { fields.back()->UseDevice(true); }
field_names.emplace_back(field_name_str);
return GetNFields() - 1;
@@ -718,6 +987,7 @@ int ParticleSet::AddTag(const char* tag_name)
tag_name_str = GetDefaultTagName(tag_names.size());
}
tags.emplace_back(std::make_unique<Array<int>>(GetNParticles()));
if (coords.UseDevice()) { tags.back()->GetMemory().UseDevice(true); }
tag_names.emplace_back(tag_name_str);
return GetNTags() - 1;
@@ -782,7 +1052,7 @@ Particle ParticleSet::GetParticle(int i) const
for (int t = 0; t < GetNTags(); t++)
{
p.Tag(t) = Tag(t)[i];
p.Tag(t) = Tag(t).HostRead()[i];
}
return p;
@@ -790,13 +1060,21 @@ Particle ParticleSet::GetParticle(int i) const
bool ParticleSet::IsParticleRefValid() const
{
if (coords.GetOrdering() == Ordering::byNODES)
if (coords.GetOrdering() == Ordering::byNODES || coords.UseDevice())
{
return false;
}
for (int f = 0; f < GetNFields(); f++)
{
if (fields[f]->GetOrdering() == Ordering::byNODES)
if (fields[f]->GetOrdering() == Ordering::byNODES ||
fields[f]->UseDevice())
{
return false;
}
}
for (int t = 0; t < GetNTags(); t++)
{
if (tags[t]->UseDevice())
{
return false;
}
@@ -806,6 +1084,10 @@ bool ParticleSet::IsParticleRefValid() const
Particle ParticleSet::GetParticleRef(int i)
{
MFEM_ASSERT(IsParticleRefValid(),
"GetParticleRef is only valid when coordinates and fields are "
"ordered byVDIM and particle data is host-resident.");
Particle p = CreateParticle();
Coords().GetValuesRef(i, p.Coords());
@@ -839,7 +1121,7 @@ void ParticleSet::SetParticle(int i, const Particle &p)
for (int t = 0; t < GetNTags(); t++)
{
Tag(t)[i] = p.Tag(t);
Tag(t).HostReadWrite()[i] = p.Tag(t);
}
}
@@ -900,6 +1182,15 @@ void ParticleSet::PrintCSV(const char *fname, const Array<int> &field_idxs,
#ifdef MFEM_USE_MPI
int rank = GetRank(comm);
#endif // MFEM_USE_MPI
// make sure we can read tag data on host. fields and coords will be read as
// needed in the loop below, so we don't need to pre-read them here.
for (int i = 0; i < GetNTags(); i++)
{
tags[i]->HostRead();
}
ids.HostRead();
// Write particle data
for (int i = 0; i < GetNParticles(); i++)
{
ss_data << ids[i];
+49 -12
View File
@@ -211,6 +211,12 @@ public:
* byVDIM). The unique_ptrs to all the ParticleVectors are stored in the
* std::vector \ref fields.
*
* @par Device Behavior:
* When a ParticleSet is constructed with \p use_device=true, \ref coords and
* all ParticleVector fields are marked to use device memory. Fields added
* later through \ref AddField inherit the current device mode (through
* \ref coords).
*
* @par Tags:
* Tags represent integers associated with each particle. For a given tag,
* all particle data are stored in a single Array<int>. The unique_ptrs to all
@@ -369,7 +375,10 @@ protected:
* ID of a particle.
*/
void UpdateID(int local_idx, IDType new_global_id)
{ ids[local_idx] = new_global_id; }
{
ids.HostReadWrite();
ids[local_idx] = new_global_id;
}
/** @brief Create a Particle object with the same spatial dimension,
* number of fields and field vdims, and number of tags as this ParticleSet.
@@ -399,12 +408,14 @@ protected:
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
* @param[in] use_device Use device memory for particle fields.
*/
ParticleSet(int id_stride_, IDType id_counter_, int num_particles, int dim,
Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
const Array<const char*> &tag_names_,
bool use_device);
public:
@@ -413,9 +424,12 @@ public:
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering Ordering of coordinates.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(int num_particles, int dim,
Ordering::Type coords_ordering=Ordering::byVDIM);
Ordering::Type coords_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Construct a serial ParticleSet with specified fields and tags at
* construction.
@@ -426,9 +440,12 @@ public:
* @param[in] num_tags Number of tags to register.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
int num_tags, Ordering::Type all_ordering=Ordering::byVDIM);
int num_tags, Ordering::Type all_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Construct a serial ParticleSet with specified fields and tags at
* construction, with names.
@@ -441,11 +458,14 @@ public:
* @param[in] tag_names_ Array of tag names.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_,
Ordering::Type all_ordering=Ordering::byVDIM);
Ordering::Type all_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Comprehensive serial constructor of ParticleSet.
*
@@ -457,12 +477,15 @@ public:
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(int num_particles, int dim, Ordering::Type coords_ordering,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
const Array<const char*> &tag_names_,
bool use_device=false);
#ifdef MFEM_USE_MPI
/** @brief Construct a parallel ParticleSet.
@@ -471,9 +494,12 @@ public:
* @param[in] rank_num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering (Optional) Ordering of coordinates.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering=Ordering::byVDIM);
Ordering::Type coords_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Construct a parallel ParticleSet with specified fields and tags
* at construction.
@@ -485,10 +511,13 @@ public:
* @param[in] num_tags Number of tags to register.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering=Ordering::byVDIM);
Ordering::Type all_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Construct a parallel ParticleSet with specified fields and tags
* at construction, with names (for PrintCSV()).
@@ -502,12 +531,15 @@ public:
* @param[in] tag_names_ Array of tag names.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims,
const Array<const char*> &field_names_,
int num_tags, const Array<const char*> &tag_names_,
Ordering::Type all_ordering=Ordering::byVDIM);
Ordering::Type all_ordering=Ordering::byVDIM,
bool use_device=false);
/** @brief Comprehensive parallel constructor of ParticleSet.
*
@@ -520,12 +552,15 @@ public:
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
* @param[in] use_device (Optional) Use device memory for particle
* fields.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
const Array<const char*> &tag_names_,
bool use_device=false);
/// Get the MPI communicator for this ParticleSet.
MPI_Comm GetComm() const { return comm; };
@@ -545,6 +580,8 @@ public:
* @param[in] field_ordering (Optional) Ordering::Type of the field.
* @param[in] field_name (Optional) Name of the field.
*
* @note New fields inherit the current device mode of \ref coords.
*
* @return Index of the newly-added field.
*/
int AddField(int vdim, Ordering::Type field_ordering=Ordering::byVDIM,
@@ -637,8 +674,8 @@ public:
/** @brief Determine if GetParticleRef is valid.
*
* If coordinates and all fields are ordered byVDIM, then returns true.
* Otherwise, false.
* Returns true when coordinates and all fields are ordered byVDIM and
* particle data is host-resident. Otherwise, false.
*/
bool IsParticleRefValid() const;
+31 -3
View File
@@ -349,6 +349,7 @@ void ParFiniteElementSpace::GetGroupComm(
}
}
bool have_sign_flips = false;
if (g_ldof_sign)
{
g_ldof_sign->SetSize(GetNDofs());
@@ -428,6 +429,7 @@ void ParFiniteElementSpace::GetGroupComm(
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -466,6 +468,7 @@ void ParFiniteElementSpace::GetGroupComm(
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -504,6 +507,7 @@ void ParFiniteElementSpace::GetGroupComm(
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -527,12 +531,18 @@ void ParFiniteElementSpace::GetGroupComm(
group_ldof.GetI()[gr+1] = group_ldof_counter;
}
if (g_ldof_sign && have_sign_flips == false)
{
g_ldof_sign->DeleteAll();
}
gc.Finalize();
}
void ParFiniteElementSpace::ApplyLDofSigns(Array<int> &dofs) const
{
MFEM_ASSERT(Conforming(), "wrong code path");
if (!HaveDofSigns()) { return; }
for (int i = 0; i < dofs.Size(); i++)
{
@@ -559,6 +569,24 @@ void ParFiniteElementSpace::ApplyLDofSigns(Table &el_dof) const
ApplyLDofSigns(all_dofs);
}
void ParFiniteElementSpace::ApplyDofSigns(real_t *h_data) const
{
if (!HaveDofSigns()) { return; }
const bool byvdim = (ordering == Ordering::byVDIM);
for (int i = 0; i < ndofs; i++)
{
if (ldof_sign[i] < 0)
{
for (int d = 0; d < vdim; d++)
{
const int idx = byvdim ? d+vdim*i : i+ndofs*d;
h_data[idx] = -h_data[idx];
}
}
}
}
void ParFiniteElementSpace::GetElementDofs(int i, Array<int> &dofs,
DofTransformation &doftrans) const
{
@@ -1193,15 +1221,15 @@ void ParFiniteElementSpace::GetEssentialTrueDofsVar(const Array<int>
MFEM_VERIFY(IsVariableOrder() && R,
"GetEssentialTrueDofsVar is only for variable-order spaces");
true_ess_dofs.SetSize(R->Height(), Device::GetDeviceMemoryType());
true_ess_dofs.SetSize(R->Height());
true_ess_dofs.HostWrite();
true_ess_dofs = 0;
const int ntdofs = tdof2ldof.Size();
MFEM_VERIFY(vdim * ntdofs == R->NumRows() &&
vdim * ntdofs == true_ess_dofs.Size(), "");
MFEM_VERIFY(ldof_ltdof.Size() == ndofs && ess_dofs.Size() == vdim * ndofs, "");
true_ess_dofs = 0;
const bool bynodes = (ordering == Ordering::byNODES);
const int vdim_factor = bynodes ? 1 : vdim;
const int num_true_dofs = R->NumRows() / vdim;
+14 -2
View File
@@ -340,8 +340,20 @@ public:
inline ParMesh *GetParMesh() const { return pmesh; }
int GetDofSign(int i)
{ return NURBSext || Nonconforming() ? 1 : ldof_sign[VDofToDof(i)]; }
/** @brief Return true if the parallel FE space has DOFs with signs opposite
of the DOFs in the respective serial FE space. */
bool HaveDofSigns() const { return ldof_sign.Size() != 0; }
/** @brief Apply the DOF signs to the given host data @a h_data which must be
of size GetVSize() if HaveDofSigns() is true. If HaveDofSigns() is false,
this method is no-op and returns immediately. */
void ApplyDofSigns(real_t *h_data) const;
/** @brief Return -1 if the given (vector) DOF @a i has a sign opposite of
the DOF in the respecive serial FE space. Otherwise, return 1. */
int GetDofSign(int i) const
{ return !HaveDofSigns() ? 1 : ldof_sign[VDofToDof(i)]; }
HYPRE_BigInt *GetDofOffsets() const { return dof_offsets; }
HYPRE_BigInt *GetTrueDofOffsets() const { return tdof_offsets; }
HYPRE_BigInt GlobalVSize() const
+18 -9
View File
@@ -80,6 +80,8 @@ ParGridFunction::ParGridFunction(ParMesh *pmesh, std::istream &input)
fes->GetOrdering());
delete fes;
fes = pfes;
pfes->ApplyDofSigns(HostReadWrite());
}
void ParGridFunction::Update()
@@ -1082,18 +1084,17 @@ real_t ParGridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
void ParGridFunction::Save(std::ostream &os) const
{
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < size; i++)
{
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
}
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t *h_data = const_cast<real_t*>(HostRead());
pfes->ApplyDofSigns(h_data);
GridFunction::Save(os);
for (int i = 0; i < size; i++)
{
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
}
pfes->ApplyDofSigns(h_data);
}
void ParGridFunction::Save(const char *fname, int precision) const
@@ -1264,7 +1265,13 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
int *nfdofs = new int[NRanks];
int *nrdofs = new int[NRanks];
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t * h_data = const_cast<real_t *>(this->HostRead());
pfes->ApplyDofSigns(h_data); // temporarily flip the dof signs
values[0] = h_data;
nv[0] = pfes -> GetVSize();
@@ -1371,6 +1378,8 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
MPI_Send(h_data, nv[0], MPITypeMap<real_t>::mpi_type, 0, 460, MyComm);
}
pfes->ApplyDofSigns(h_data); // restore the original h_data
delete [] values;
delete [] nv;
delete [] nvdofs;
+9 -2
View File
@@ -50,14 +50,21 @@ ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
const FiniteElement *fe = fes.GetFE(e);
auto el_t = dynamic_cast<const TensorBasisElement*>(fe);
auto el_n = dynamic_cast<const NodalFiniteElement*>(fe);
if (el_t || el_n) { continue; }
auto el_p = dynamic_cast<const H1Pos_TriangleElement*>(fe) ||
dynamic_cast<const H1Pos_TetrahedronElement*>(fe);
if (el_t || el_n || el_p) { continue; }
MFEM_ABORT("Finite element not suitable for lexicographic ordering");
}
const FiniteElement *fe = fes.GetTypicalFE();
auto el_t = dynamic_cast<const TensorBasisElement*>(fe);
auto el_n = dynamic_cast<const NodalFiniteElement*>(fe);
auto el_p_tri = dynamic_cast<const H1Pos_TriangleElement*>(fe);
auto el_p_tet = dynamic_cast<const H1Pos_TetrahedronElement*>(fe);
const Array<int> &fe_dof_map =
(el_t) ? el_t->GetDofMap() : el_n->GetLexicographicOrdering();
el_n ? el_n->GetLexicographicOrdering() :
el_t ? el_t->GetDofMap() :
el_p_tri ? el_p_tri->GetDofMap() :
el_p_tet->GetDofMap();
MFEM_VERIFY(fe_dof_map.Size() > 0, "invalid dof map");
dof_map = fe_dof_map.HostRead();
}
+303 -113
View File
@@ -3758,7 +3758,8 @@ void TMOP_Integrator::SetInitialMeshPos(const GridFunction *x0)
TMOP_Integrator::~TMOP_Integrator()
{
delete lim_func;
delete adapt_lim_gf;
for (int i = 0; i < adapt_lim_gf.Size(); i++) { delete adapt_lim_gf[i]; }
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { delete adapt_lim_gf0[i]; }
delete surf_fit_gf;
delete surf_fit_limiter;
delete surf_fit_grad;
@@ -3800,20 +3801,13 @@ void TMOP_Integrator::EnableAdaptiveLimiting(const GridFunction &z0,
AdaptivityEvaluator &ae,
real_t delta_max)
{
MFEM_VERIFY(delta_max > 0.0,
"EnableAdaptiveLimiting requires delta_max > 0.0.");
adapt_lim_gf0 = &z0;
delete adapt_lim_gf;
adapt_lim_gf = new GridFunction(z0);
adapt_lim_coeff = &coeff;
adapt_lim_eval = &ae;
adapt_lim_delta_max = delta_max;
adapt_lim_eval->SetSerialMetaInfo(*z0.FESpace()->GetMesh(),
*z0.FESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
Array<const GridFunction *> z0_arr(1);
Array<Coefficient *> c_arr(1);
Array<real_t> d_arr(1);
z0_arr[0] = &z0;
c_arr[0] = &coeff;
d_arr[0] = delta_max;
EnableAdaptiveLimiting(z0_arr, c_arr, ae, d_arr);
}
#ifdef MFEM_USE_MPI
@@ -3822,21 +3816,111 @@ void TMOP_Integrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
AdaptivityEvaluator &ae,
real_t delta_max)
{
MFEM_VERIFY(delta_max > 0.0,
"EnableAdaptiveLimiting requires delta_max > 0.0.");
Array<const ParGridFunction *> z0_arr(1);
Array<Coefficient *> c_arr(1);
Array<real_t> d_arr(1);
z0_arr[0] = &z0;
c_arr[0] = &coeff;
d_arr[0] = delta_max;
EnableAdaptiveLimiting(z0_arr, c_arr, ae, d_arr);
}
#endif
adapt_lim_gf0 = &z0;
adapt_lim_pgf0 = &z0;
delete adapt_lim_gf;
adapt_lim_gf = new GridFunction(z0);
adapt_lim_coeff = &coeff;
void TMOP_Integrator::
EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(z0.Size() > 0, "Requires at least one field.");
MFEM_VERIFY(z0.Size() == coeff.Size(), "Requires one Coefficient per field.");
MFEM_VERIFY(z0.Size() == delta_max.Size(), "Requires one delta_max per field.");
for (int i = 0; i < delta_max.Size(); i++)
{
MFEM_VERIFY(delta_max[i] > 0.0, "Requires delta_max > 0.0.");
}
// Verify compatibility of input fields.
const FiniteElementSpace *sfes = z0[0]->FESpace();
MFEM_VERIFY(sfes->GetVDim() == 1, "Expects scalar input GridFunctions.");
const int ndofs = sfes->GetVSize();
Mesh *mesh = sfes->GetMesh();
MFEM_VERIFY(mesh->GetNodes(), "EnableAdaptiveLimiting requires mesh Nodes.");
for (int i = 0; i < z0.Size(); i++)
{
MFEM_VERIFY(z0[i], "NULL GridFunction pointer.");
const FiniteElementSpace *fes_i = z0[i]->FESpace();
MFEM_VERIFY(fes_i->GetVDim() == 1, "Expects scalar input GridFunctions.");
MFEM_VERIFY(fes_i->GetVSize() == ndofs,
"All fields must be on the same FE space.");
MFEM_VERIFY(fes_i->GetMesh() == mesh,
"All fields must be on the same Mesh.");
MFEM_VERIFY(coeff[i], "NULL Coefficient pointer.");
}
// Delete previous adaptive limiting data.
for (int i = 0; i < adapt_lim_gf.Size(); i++) { delete adapt_lim_gf[i]; }
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { delete adapt_lim_gf0[i]; }
adapt_lim_coeff.SetSize(coeff.Size());
for (int i = 0; i < coeff.Size(); i++) { adapt_lim_coeff[i] = coeff[i]; }
adapt_lim_eval = &ae;
adapt_lim_delta_max = delta_max;
adapt_lim_init_nodes = *mesh->GetNodes();
adapt_lim_eval->SetParMetaInfo(*z0.ParFESpace()->GetParMesh(),
*z0.ParFESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
// Use one internal vector field (vdim = #fields) so remapping can be done in
// one call and incremental remap state (when provided by the evaluator) is
// preserved across TMOP iterations.
//
// Use Ordering::byNODES for the packed vector field so packing / unpacking
// can be done with contiguous sub-vector copies (device-friendly).
const int nal = z0.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
// Setup the evaluator.
#ifdef MFEM_USE_MPI
if (auto pfes = dynamic_cast<const ParFiniteElementSpace *>(sfes))
{
auto *pm = pfes->GetParMesh();
MFEM_VERIFY(pm, "Invalid ParMesh.");
ParFiniteElementSpace vfes(pm, pfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetParMetaInfo(*pm, vfes);
}
else
#endif
{
FiniteElementSpace vfes(mesh, sfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetSerialMetaInfo(*mesh, vfes);
}
// Copy the initial fields; remapped fields are initialized to the same data.
adapt_lim_gf0.SetSize(z0.Size());
adapt_lim_gf.SetSize(z0.Size());
for (int i = 0; i < z0.Size(); i++)
{
adapt_lim_gf0[i] = new GridFunction(*z0[i]);
adapt_lim_gf[i] = new GridFunction(*z0[i]);
}
// Initialize the evaluator with the packed vector field.
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::
EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
Array<const GridFunction *> z0_base(z0.Size());
for (int i = 0; i < z0.Size(); i++) { z0_base[i] = z0[i]; }
EnableAdaptiveLimiting(z0_base, coeff, ae, delta_max);
}
#endif
@@ -4157,26 +4241,61 @@ void TMOP_Integrator::GetSurfaceFittingErrors(const Vector &d_loc,
void TMOP_Integrator::UpdateAfterMeshTopologyChange()
{
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_gf->Update();
adapt_lim_eval->SetSerialMetaInfo(*adapt_lim_gf->FESpace()->GetMesh(),
*adapt_lim_gf->FESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { adapt_lim_gf0[i]->Update(); }
for (int i = 0; i < adapt_lim_gf.Size(); i++) { adapt_lim_gf[i]->Update(); }
Mesh *mesh = adapt_lim_gf[0]->FESpace()->GetMesh();
// Same setup as in EnableAdaptiveLimiting().
const int nal = adapt_lim_coeff.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
FiniteElementSpace vfes(mesh, adapt_lim_gf[0]->FESpace()->FEColl(), nal,
packed_ord);
adapt_lim_eval->SetSerialMetaInfo(*mesh, vfes);
adapt_lim_init_nodes = *mesh->GetNodes();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::ParUpdateAfterMeshTopologyChange()
{
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_gf->Update();
adapt_lim_eval->SetParMetaInfo(*adapt_lim_pgf0->ParFESpace()->GetParMesh(),
*adapt_lim_pgf0->ParFESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { adapt_lim_gf0[i]->Update(); }
for (int i = 0; i < adapt_lim_gf.Size(); i++) { adapt_lim_gf[i]->Update(); }
// Same setup as in EnableAdaptiveLimiting().
auto *pfes = dynamic_cast<ParFiniteElementSpace *>(adapt_lim_gf[0]->FESpace());
MFEM_VERIFY(pfes, "internal error");
ParMesh *pmesh = pfes->GetParMesh();
const int nal = adapt_lim_coeff.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
ParFiniteElementSpace vfes(pmesh, pfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetParMetaInfo(*pmesh, vfes);
adapt_lim_init_nodes = *pmesh->GetNodes();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
}
#endif
@@ -4208,7 +4327,8 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// No adaptive limiting / surface fitting terms if the function is called
// as part of a FD derivative computation (because we include the exact
// derivatives of these terms in FD computations).
const bool adaptive_limiting = (adapt_lim_gf && fd_call_flag == false);
const bool adaptive_limiting = (adapt_lim_gf.Size() > 0 &&
fd_call_flag == false);
const bool surface_fit = (surf_fit_marker && fd_call_flag == false);
DSh.SetSize(dof, dim);
@@ -4271,11 +4391,21 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// the physical coordinates (i.e. changes in 'elfun'), e.g. when the
// coefficient is a ConstantCoefficient or a GridFunctionCoefficient.
const int nal = adapt_lim_coeff.Size();
const int nqp = ir.GetNPoints();
Vector adapt_lim_gf_q, adapt_lim_gf0_q;
if (adaptive_limiting)
{
adapt_lim_gf->GetValues(el_id, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(el_id, ir, adapt_lim_gf0_q);
adapt_lim_gf_q.SetSize(nal * nqp);
adapt_lim_gf0_q.SetSize(nal * nqp);
Vector zc, z0c;
for (int c = 0; c < nal; c++)
{
zc.MakeRef(adapt_lim_gf_q, c * nqp, nqp);
z0c.MakeRef(adapt_lim_gf0_q, c * nqp, nqp);
adapt_lim_gf[c]->GetValues(el_id, ir, zc);
adapt_lim_gf0[c]->GetValues(el_id, ir, z0c);
}
}
for (int i = 0; i < ir.GetNPoints(); i++)
@@ -4307,9 +4437,13 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// Contribution from the adaptive limiting term.
if (adaptive_limiting)
{
const real_t diff = (adapt_lim_gf_q(i) - adapt_lim_gf0_q(i)) /
adapt_lim_delta_max;
val += adapt_lim_coeff->Eval(*Tpr, ip) * lim_normal * diff * diff;
for (int c = 0; c < nal; c++)
{
const int idx = c * nqp + i;
const real_t diff = (adapt_lim_gf_q(idx) - adapt_lim_gf0_q(idx)) /
adapt_lim_delta_max[c];
val += adapt_lim_coeff[c]->Eval(*Tpr, ip) * lim_normal * diff * diff;
}
}
energy += weight * val;
@@ -4602,7 +4736,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (metric_coeff || lim_coeff || adapt_lim_gf ||
if (metric_coeff || lim_coeff || adapt_lim_gf.Size() > 0 ||
surf_fit_gf || surf_fit_pos || exact_action)
{
Tpr = new IsoparametricTransformation;
@@ -4700,7 +4834,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
}
}
if (adapt_lim_gf) { AssembleElemVecAdaptLim(el, *Tpr, ir, weights, PMatO); }
if (adapt_lim_gf.Size() > 0) { AssembleElemVecAdaptLim(el, *Tpr, ir, weights, PMatO); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemVecSurfFit(el, *Tpr, PMatO); }
delete Tpr;
@@ -4774,7 +4908,8 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (metric_coeff || lim_coeff || adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (metric_coeff || lim_coeff || adapt_lim_gf.Size() > 0 ||
surf_fit_gf || surf_fit_pos)
{
Tpr = new IsoparametricTransformation;
Tpr->SetFE(&el);
@@ -4829,7 +4964,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
}
}
if (adapt_lim_gf) { AssembleElemGradAdaptLim(el, *Tpr, ir, weights, elmat); }
if (adapt_lim_gf.Size() > 0) { AssembleElemGradAdaptLim(el, *Tpr, ir, weights, elmat); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemGradSurfFit(el, *Tpr, elmat);}
delete Tpr;
@@ -4842,34 +4977,42 @@ void TMOP_Integrator::AssembleElemVecAdaptLim(const FiniteElement &el,
DenseMatrix &mat)
{
const int dof = el.GetDof(), dim = el.GetDim(), nqp = weights.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q, adapt_lim_gf0_q(nqp);
const int nal = adapt_lim_coeff.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q(nqp), adapt_lim_gf0_q(nqp);
Array<int> dofs;
adapt_lim_gf->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
adapt_lim_gf->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf[0]->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
// Project the gradient of adapt_lim_gf in the same space.
// The FE coefficients of the gradient go in adapt_lim_gf_grad_e.
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
el.ProjectGrad(el, Tpr, grad_phys);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
Vector adapt_lim_gf_grad_q(dim);
for (int q = 0; q < nqp; q++)
for (int c = 0; c < nal; c++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
const real_t delta2 = adapt_lim_delta_max[c] * adapt_lim_delta_max[c];
adapt_lim_gf[c]->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
adapt_lim_gf_grad_q *= 2.0 * (adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) /
adapt_lim_delta_max / adapt_lim_delta_max;
adapt_lim_gf_grad_q *= weights(q) * lim_normal * adapt_lim_coeff->Eval(Tpr, ip);
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
AddMultVWt(shape, adapt_lim_gf_grad_q, mat);
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
adapt_lim_gf_grad_q *= 2.0 * (adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) /
delta2;
adapt_lim_gf_grad_q *=
weights(q) * lim_normal * adapt_lim_coeff[c]->Eval(Tpr, ip);
AddMultVWt(shape, adapt_lim_gf_grad_q, mat);
}
}
}
@@ -4880,60 +5023,66 @@ void TMOP_Integrator::AssembleElemGradAdaptLim(const FiniteElement &el,
DenseMatrix &mat)
{
const int dof = el.GetDof(), dim = el.GetDim(), nqp = weights.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q, adapt_lim_gf0_q(nqp);
const int nal = adapt_lim_coeff.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q(nqp), adapt_lim_gf0_q(nqp);
Array<int> dofs;
adapt_lim_gf->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
adapt_lim_gf->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf[0]->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
// Project the gradient of adapt_lim_gf in the same space.
// The FE coefficients of the gradient go in adapt_lim_gf_grad_e.
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
el.ProjectGrad(el, Tpr, grad_phys);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
// Project the gradient of each gradient of adapt_lim_gf in the same space.
// The FE coefficients of the second derivatives go in adapt_lim_gf_hess_e.
DenseMatrix adapt_lim_gf_hess_e(dof*dim, dim);
Mult(grad_phys, adapt_lim_gf_grad_e, adapt_lim_gf_hess_e);
// Reshape to be more convenient later (no change in the data).
adapt_lim_gf_hess_e.SetSize(dof, dim*dim);
Vector adapt_lim_gf_grad_q(dim);
DenseMatrix adapt_lim_gf_hess_q(dim, dim);
for (int q = 0; q < nqp; q++)
for (int c = 0; c < nal; c++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
const real_t delta2 = adapt_lim_delta_max[c] * adapt_lim_delta_max[c];
adapt_lim_gf[c]->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
Vector gg_ptr(adapt_lim_gf_hess_q.GetData(), dim*dim);
adapt_lim_gf_hess_e.MultTranspose(shape, gg_ptr);
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
const real_t coeff = adapt_lim_coeff->Eval(Tpr, ip);
const real_t factor =
weights(q) * lim_normal * coeff * 2.0 /
(adapt_lim_delta_max * adapt_lim_delta_max);
// Project the gradient of each gradient of adapt_lim_gf in the same space.
// The FE coefficients of the second derivatives go in adapt_lim_gf_hess_e.
DenseMatrix adapt_lim_gf_hess_e(dof*dim, dim);
Mult(grad_phys, adapt_lim_gf_grad_e, adapt_lim_gf_hess_e);
// Reshape to be more convenient later (no change in the data).
adapt_lim_gf_hess_e.SetSize(dof, dim*dim);
for (int i = 0; i < dof * dim; i++)
for (int q = 0; q < nqp; q++)
{
const int idof = i % dof, idim = i / dof;
for (int j = 0; j <= i; j++)
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
Vector gg_ptr(adapt_lim_gf_hess_q.GetData(), dim*dim);
adapt_lim_gf_hess_e.MultTranspose(shape, gg_ptr);
const real_t coeff_q = adapt_lim_coeff[c]->Eval(Tpr, ip);
const real_t factor =
weights(q) * lim_normal * coeff_q * 2.0 /
delta2;
for (int i = 0; i < dof * dim; i++)
{
const int jdof = j % dof, jdim = j / dof;
const real_t entry =
factor *
(adapt_lim_gf_grad_q(idim) * shape(idof) *
adapt_lim_gf_grad_q(jdim) * shape(jdof) +
(adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) *
adapt_lim_gf_hess_q(idim, jdim) * shape(idof) * shape(jdof));
mat(i, j) += entry;
if (i != j) { mat(j, i) += entry; }
const int idof = i % dof, idim = i / dof;
for (int j = 0; j <= i; j++)
{
const int jdof = j % dof, jdim = j / dof;
const real_t entry =
factor *
(adapt_lim_gf_grad_q(idim) * shape(idof) *
adapt_lim_gf_grad_q(jdim) * shape(jdof) +
(adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) *
adapt_lim_gf_hess_q(idim, jdim) * shape(idof) * shape(jdof));
mat(i, j) += entry;
if (i != j) { mat(j, i) += entry; }
}
}
}
}
@@ -5206,7 +5355,7 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
fd_call_flag = false;
// Contributions from adaptive limiting, surface fitting (exact derivatives).
if (adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (adapt_lim_gf.Size() > 0 || surf_fit_gf || surf_fit_pos)
{
const IntegrationRule &ir = ActionIntegrationRule(el);
const int nqp = ir.GetNPoints();
@@ -5230,7 +5379,7 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
}
PMatO.UseExternalData(elvect.GetData(), dof, dim);
if (adapt_lim_gf) { AssembleElemVecAdaptLim(el, Tpr, ir, weights, PMatO); }
if (adapt_lim_gf.Size() > 0) { AssembleElemVecAdaptLim(el, Tpr, ir, weights, PMatO); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemVecSurfFit(el, Tpr, PMatO); }
}
}
@@ -5316,7 +5465,7 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
fd_call_flag = false;
// Contributions from adaptive limiting.
if (adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (adapt_lim_gf.Size() > 0 || surf_fit_gf || surf_fit_pos)
{
const IntegrationRule &ir = GradientIntegrationRule(el);
const int nqp = ir.GetNPoints();
@@ -5339,7 +5488,7 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
ir.IntPoint(q).weight;
}
if (adapt_lim_gf) { AssembleElemGradAdaptLim(el, Tpr, ir, weights, elmat); }
if (adapt_lim_gf.Size() > 0) { AssembleElemGradAdaptLim(el, Tpr, ir, weights, elmat); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemGradSurfFit(el, Tpr, elmat); }
}
}
@@ -5686,9 +5835,22 @@ UpdateAfterMeshPositionChange(const Vector &d, const FiniteElementSpace &d_fes)
}
// Update adapt_lim_gf if adaptive limiting is enabled.
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_eval->ComputeAtNewPosition(x_loc, *adapt_lim_gf, ordering);
// All adapt_lim_gf are remapped as a multi-component vector.
const int nal = adapt_lim_coeff.Size();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector new_field_vec;
new_field_vec.SetSize(nal * ndofs, *adapt_lim_gf[0]);
new_field_vec.UseDevice(adapt_lim_gf[0]->UseDevice());
adapt_lim_eval->ComputeAtNewPosition(x_loc, new_field_vec, ordering);
for (int c = 0; c < nal; c++)
{
const real_t *src = new_field_vec.Read() + c * ndofs;
real_t *dst = adapt_lim_gf[c]->Write();
internal::device_copy(dst, src, ndofs);
}
if (PA.enabled)
{
PA.AL_grads_assembled = false;
@@ -5698,9 +5860,17 @@ UpdateAfterMeshPositionChange(const Vector &d, const FiniteElementSpace &d_fes)
// Refresh PA.ALF from the updated adapt_lim_gf.
const ElementDofOrdering ord = ElementDofOrdering::LEXICOGRAPHIC;
const Operator *alf_R =
adapt_lim_gf->FESpace()->GetElementRestriction(ord);
alf_R->Mult(*adapt_lim_gf, PA.ALF);
const FiniteElementSpace *alfes = adapt_lim_gf[0]->FESpace();
const Operator *alf_R = alfes->GetElementRestriction(ord);
const int Esize = alf_R->Height();
Vector ALFc;
for (int c = 0; c < nal; c++)
{
MFEM_VERIFY(adapt_lim_gf[c]->Size() == ndofs, "internal error");
ALFc.MakeRef(PA.ALF, c * Esize, Esize);
alf_R->Mult(*adapt_lim_gf[c], ALFc);
}
// Step 2 of PA.ALFmF0 update: add the new ALF.
PA.ALFmF0 += PA.ALF;
@@ -5917,7 +6087,7 @@ ComputeUntangleMetricQuantiles(const Vector &d, const FiniteElementSpace &fes)
dynamic_cast<const ParFiniteElementSpace *>(&fes);
#endif
if (wcuo && wcuo->GetBarrierType() ==
if (wcuo->GetBarrierType() ==
TMOP_WorstCaseUntangleOptimizer_Metric::BarrierType::Shifted)
{
real_t min_detT = ComputeMinDetT(x_loc, fes);
@@ -5929,7 +6099,7 @@ ComputeUntangleMetricQuantiles(const Vector &d, const FiniteElementSpace &fes)
MPITypeMap<real_t>::mpi_type, MPI_MIN, pfes->GetComm());
}
#endif
if (wcuo) { wcuo->SetMinDetT(min_detT_all); }
wcuo->SetMinDetT(min_detT_all);
}
real_t max_muT = ComputeUntanglerMaxMuBarrier(x_loc, fes);
@@ -5975,6 +6145,16 @@ void TMOPComboIntegrator::EnableAdaptiveLimiting(const GridFunction &z0,
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
void TMOPComboIntegrator::
EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(tmopi.Size() > 0, "No TMOP_Integrators were added.");
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
#ifdef MFEM_USE_MPI
void TMOPComboIntegrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
Coefficient &coeff,
@@ -5985,6 +6165,16 @@ void TMOPComboIntegrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
void TMOPComboIntegrator::
EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(tmopi.Size() > 0, "No TMOP_Integrators were added.");
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
#endif
void TMOPComboIntegrator::SetLimitingNodes(const GridFunction &n0)
+34 -11
View File
@@ -2038,14 +2038,17 @@ protected:
real_t lim_normal;
// Adaptive limiting.
const GridFunction *adapt_lim_gf0; // Not owned.
#ifdef MFEM_USE_MPI
const ParGridFunction *adapt_lim_pgf0;
#endif
GridFunction *adapt_lim_gf; // Owned. Updated by adapt_lim_eval.
Coefficient *adapt_lim_coeff; // Not owned.
AdaptivityEvaluator *adapt_lim_eval; // Not owned.
real_t adapt_lim_delta_max = 1.0;
// Adaptive limiting fields. Each field adds a term to the integral:
// int [ c_k (z_k(x) - z_k0(x0))^2 / delta_max_k^2 ] dx
// with one Coefficient per field. The fields z_k(x) are remapped from their
// initial values z_k0(x0) through a single AdaptivityEvaluator instance.
// All GridFunctions must use the same FE space.
Array<GridFunction *> adapt_lim_gf0; // Owned. Initial fields z_k0(x0).
Array<GridFunction *> adapt_lim_gf; // Owned. Remapped fields z_k(x).
Vector adapt_lim_init_nodes; // Owned. Initial mesh nodes (ldofs).
Array<Coefficient *> adapt_lim_coeff; // Not owned, one per field.
AdaptivityEvaluator *adapt_lim_eval; // Not owned. Used for all fields.
Array<real_t> adapt_lim_delta_max; // Per-field delta_max_k (>0).
// Surface fitting.
const Array<bool> *surf_fit_marker; // Not owned. Nodes to fit.
@@ -2141,13 +2144,13 @@ protected:
{
bool enabled;
int dim, ne, nq;
int nal = 0; // number of adaptive limiting fields
mutable DenseTensor Jtr;
mutable bool Jtr_needs_update;
mutable bool Jtr_debug_grad;
mutable Vector E, O, X0, XL, H, C0, LD, H0, MC, ALC,
ALF, ALFmF0, ALFG, ALFH;
ALF, ALFmF0, ALFG, ALFH, ALD;
mutable bool AL_grads_assembled;
real_t al_delta;
const DofToQuad *maps;
const DofToQuad *maps_lim = nullptr;
const DofToQuad *maps_nodes = nullptr;
@@ -2314,7 +2317,6 @@ public:
integ_order(-1), metric_coeff(NULL), metric_normal(1.0),
lim_nodes0(NULL), lim_coeff(NULL),
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
adapt_lim_gf0(NULL), adapt_lim_gf(NULL), adapt_lim_coeff(NULL),
adapt_lim_eval(NULL),
surf_fit_marker(NULL), surf_fit_coeff(NULL),
surf_fit_gf(NULL), surf_fit_eval(NULL),
@@ -2403,10 +2405,21 @@ public:
Smaller values activate the term faster. */
void EnableAdaptiveLimiting(const GridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field adaptive limiting with per-field delta_max values. All
/// GridFunctions must be on the same FiniteElementSpace.
void EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#ifdef MFEM_USE_MPI
/// Parallel support for adaptive limiting.
void EnableAdaptiveLimiting(const ParGridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field parallel adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#endif
/** @brief Fitting of certain DOFs to the zero level set of a function.
@@ -2632,10 +2645,20 @@ public:
/// Adds the adaptive limiting term to the first integrator.
void EnableAdaptiveLimiting(const GridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#ifdef MFEM_USE_MPI
/// Parallel support for adaptive limiting.
void EnableAdaptiveLimiting(const ParGridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field parallel adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#endif
+31 -14
View File
@@ -119,15 +119,14 @@ void TMOP_AssembleDiagPA_AdaptLim_2D(const real_t lim_normal,
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t weight = W(qx, qy) * detJtr;
const real_t coeff = const_coeff ? ALC(0, 0, 0) : ALC(qx, qy, e);
const real_t coeff = const_coeff ? ALC(0,0,0) : ALC(qx, qy, e);
const real_t factor = weight * coeff * normal_inv_delta_sq;
const real_t diff = alf_quad(qy, qx);
const real_t grad_v = ALF_grad(v, qx, qy, e);
const real_t hess_vv = ALF_hess(v, v, qx, qy, e);
const real_t hdiag = factor * (grad_v * grad_v + diff * hess_vv);
QD(qx, dy) += bb * hdiag;
QD(qx, dy) += bb * factor * (grad_v*grad_v + diff * hess_vv);
}
}
}
@@ -176,27 +175,45 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagAdaptLim2D);
void TMOP_Integrator::AssembleDiagonalPA_AdaptLim_2D(Vector &diagonal) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto *B = PA.maps->B.Read();
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 2, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiagAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const int ALFH_stride = 2 * 2 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 2, 2, q, q, NE);
TMOPAssembleDiagAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
}
}
} // namespace mfem
+32 -13
View File
@@ -183,15 +183,15 @@ void TMOP_AssembleDiagPA_AdaptLim_3D(const real_t lim_normal,
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t weight = W(qx, qy, qz) * detJtr;
const real_t coeff = const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t coeff =
const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t factor = weight * coeff * normal_inv_delta_sq;
const real_t diff = alf_quad(qz, qy, qx);
const real_t grad_v = ALF_grad(v, qx, qy, qz, e);
const real_t hess_vv = ALF_hess(v, v, qx, qy, qz, e);
const real_t hdiag = factor * (grad_v * grad_v + diff * hess_vv);
u += bb * hdiag;
u += bb * factor * (grad_v * grad_v + diff * hess_vv);
}
r0[dz][qy][qx] = u;
}
@@ -265,26 +265,45 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagAdaptLim3D);
void TMOP_Integrator::AssembleDiagonalPA_AdaptLim_3D(Vector &diagonal) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto *B = PA.maps->B.Read();
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 3, q, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiagAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 3 * nqp_el * NE;
const int ALFH_stride = 3 * 3 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 3, q, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 3, 3, q, q, q, NE);
TMOPAssembleDiagAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
}
}
} // namespace mfem
+22 -7
View File
@@ -113,7 +113,7 @@ void TMOP_AssembleGradPA_C0_2D(const real_t lim_normal,
});
}
// Assemble gradient and Hessian of ALF field at quadrature points for AdaptLim (2D)
// Assemble gradient and Hessian of ALF field at quad points for AdaptLim (2D).
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_AdaptLim_2D(const int NE,
const real_t *B_nodes,
@@ -185,7 +185,7 @@ void TMOP_AssembleGradPA_AdaptLim_2D(const int NE,
}
MFEM_SYNC_THREAD;
// Compute/interpolate gradient and Hessian one vector component at a time.
// Compute/interpolate gradient and Hessian, one component at a time.
for (int c = 0; c < 2; c++)
{
kernels::internal::s_regs2d_t<MD1> rgrad_nodes, ddalf_dx_n, ddalf_dy_n;
@@ -326,16 +326,31 @@ void TMOP_Integrator::AssembleGradPA_AdaptLim_2D(const Vector &x) const
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const auto *B_nodes = PA.maps_nodes->B.Read(),
*G_nodes = PA.maps_nodes->G.Read();
const auto *B = PA.maps->B.Read();
const auto X = Reshape(x.Read(), d, d, 2, NE);
const auto ALF = Reshape(PA.ALF.Read(), d, d, NE);
auto ALF_grad = Reshape(PA.ALFG.Write(), 2, q, q, NE);
auto ALF_hess = Reshape(PA.ALFH.Write(), 2, 2, q, q, NE);
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const int ALFH_stride = 2 * 2 * nqp_el * NE;
TMOPAssembleGradAdaptLim2D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
const real_t *ALF_all = PA.ALF.Read();
real_t *ALFG_all = PA.ALFG.Write();
real_t *ALFH_all = PA.ALFH.Write();
for (int c = 0; c < nal; c++)
{
const auto ALF = Reshape(ALF_all + c * ALF_stride, d, d, NE);
auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 2, 2, q, q, NE);
TMOPAssembleGradAdaptLim2D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
}
PA.AL_grads_assembled = true;
}

Some files were not shown because too many files have changed in this diff Show More