Compare commits

...
903 Commits
Author SHA1 Message Date
Ketan Mittal eeaf33d17c Merge branch 'master' of https://github.com/mfem/mfem into mesh-dist 2026-06-09 10:00:05 -07:00
Ketan Mittal 70cc72c462 initial commit 2026-06-09 09:59:57 -07:00
Tzanio Kolev c6ea42a681 Merge pull request #5045 from mfem/stefanozampini/petsc-requires-hypre
PETSc: Remove ad-hoc code to support PETSc without hypre
2026-06-06 16:52:11 -07:00
Tzanio Kolev 61bb7755a3 Merge pull request #5324 from mfem/multigrid-bcs-fix
Bugfix for GeometricMultigrid with no essential BCs
2026-06-05 16:17:51 -07:00
Tzanio Kolev a7fa61464c Merge pull request #5267 from mfem/print-interfaces-dev
Optional output of material interfaces in parallel
2026-06-05 16:13:18 -07:00
Tzanio Kolev 93329a7ba0 Merge pull request #5124 from yuyangdai/cudss-dev
Add support for parallel NVIDIA's GPU-accelerated direct sparse solver cuDSS solver[cudss-dev]
2026-06-05 16:10:03 -07:00
yuyangdai fb93e21a0c Update INSTALL file to correct cuDSS library options 2026-06-05 09:16:00 +08:00
Stefano Zampini 19d60f95ec PETSc: Remove ad-hoc code to support PETSc without hypre 2026-06-04 11:27:36 +01:00
Will Pazner 861d318162 Fix Doxygen comment in GeometricMultigrid 2026-06-03 15:48:39 -07:00
Tzanio Kolev d29e914146 Merge branch 'master' into multigrid-bcs-fix 2026-06-02 18:20:52 -07:00
yuyangdai b5e99d8d83 Fix CMake script to correctly locate cuDSS library and allocate device memory for csr values when set the cudssMatrix. 2026-06-02 13:31:25 +08:00
daiyuyangandAndrew Ho 91f58c293d Update linalg/cudss.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-06-02 13:17:09 +08:00
Tzanio Kolev 6451e64637 Merge branch 'master' into cudss-dev 2026-06-01 15:55:15 -07:00
Tzanio Kolev b7ac1963bb Merge pull request #5340 from mfem/update-contact-miniapp-readme
Update Tribol shared libraries
2026-06-01 15:53:50 -07:00
Veselin Dobrev baefb786fb Correction in CHANGELOG 2026-06-01 08:58:54 -07:00
Veselin Dobrev 99523b3a97 Address reviewer feedback: add/improve Doxygen documentation. 2026-06-01 08:43:12 -07:00
Tzanio Kolev 59d6820ff3 Merge branch 'master' into cudss-dev 2026-05-28 14:21:30 -07:00
Tzanio Kolev 9ec379f176 Merge branch 'master' into update-contact-miniapp-readme 2026-05-28 14:19:45 -07:00
Tzanio Kolev 72cc503eb9 Updated CHANGELOG 2026-05-28 12:53:25 -07:00
Veselin Dobrev 5aff935c98 Minor future-proofing suggested by Copilot.
MFEM_ASSERT does not need to explicitly print the name of the function
because that is already done automatically.
2026-05-27 10:18:49 -07:00
Tzanio KolevandCopilot Autofix powered by AI d56298ba17 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 19:02:02 -07:00
Tzanio KolevandCopilot Autofix powered by AI b9d67dec34 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 19:00:55 -07:00
Tzanio Kolev 5b664e393d Merge branch 'master' into print-interfaces-dev 2026-05-26 18:43:14 -07:00
Veselin Dobrev c582282084 Merge pull request #5345 from mfem/copilot-dev
Updates in developer docs + instructions for the GitHub Copilot reviews
2026-05-26 18:27:49 -07:00
Tzanio Kolev 944ece4090 TODO item for ConduitDataCollection::Save() 2026-05-26 17:35:13 -07:00
Veselin DobrevandTzanio Kolev f225d0c3ef Apply suggestions from code review
Remove FIXME comments -- no actions needed.

Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-05-26 16:58:36 -07:00
Tzanio Kolev a9c98c2e3a Merge branch 'master' into print-interfaces-dev 2026-05-26 10:35:15 -07:00
Tzanio KolevandCopilot Autofix powered by AI ac8c2948e7 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 10:02:50 -07:00
Tzanio KolevandCopilot Autofix powered by AI f75b4c10d2 Potential fix for pull request finding
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-26 10:02:10 -07:00
Veselin Dobrev ecf6b0b44f Address some reviewer suggestions and comments 2026-05-26 08:41:10 -07:00
Tzanio Kolev 5a52c676e1 Small changes in Copilot instructions 2026-05-25 15:49:19 -07:00
Tzanio Kolev 5be9693235 Initial Copilot instructions 2026-05-25 15:34:45 -07:00
Tzanio Kolev 5421abc4c4 Fixes and updates in CONTRIBUTING.md 2026-05-25 12:58:57 -07:00
Tzanio Kolev f2d3fb45dc Fixes and updates in INSTALL 2026-05-25 12:04:26 -07:00
John Camier 239c83a742 Merge branch 'master' into cudss-dev 2026-05-24 20:14:04 -07:00
maxpaik16 c315028a09 Merge branch 'master' into update-contact-miniapp-readme 2026-05-24 19:03:34 -07:00
yuyangdai 99b66f4f45 Update CHANGELOG 2026-05-25 09:20:19 +08:00
Tzanio Kolev e7be50eb91 Merge pull request #5333 from nmnobre/gslib
FindPointsGSLIB: use parallel-aware ProjectDiscCoefficient
2026-05-24 10:58:16 -07:00
Tzanio Kolev 2dd3915ad5 Merge branch 'master' into gslib 2026-05-23 10:15:21 -07:00
Tzanio Kolev 29b572819a Merge branch 'master' into cudss-dev 2026-05-23 10:14:37 -07:00
Tzanio Kolev 0a0acfda66 Merge pull request #5329 from nmnobre/host
Fix GetEssentialTrueDofsVar memory allocation
2026-05-22 20:15:51 -07:00
maxpaik16 7dbce44472 Update tribol libraries 2026-05-22 16:36:51 -07:00
maxpaik16 c67bd21219 Update tribol shared libraries 2026-05-22 16:35:29 -07:00
Tzanio Kolev 38f5b93520 Merge branch 'master' into print-interfaces-dev 2026-05-21 09:45:23 -07:00
Tzanio Kolev a078dfd59e Reviewer comments 2026-05-21 09:45:02 -07:00
Tzanio Kolev 66818f5525 Merge branch 'master' into gslib 2026-05-21 07:41:18 -07:00
Tzanio Kolev b810a5e540 Merge branch 'master' into host 2026-05-21 07:41:14 -07:00
Tzanio Kolev e63e421343 Merge branch 'master' into cudss-dev 2026-05-21 07:40:59 -07:00
Tzanio Kolev 176958144b Merge pull request #5107 from mfem/findpts-surface
FindPointsGSLIB for surface meshes
2026-05-19 13:29:29 -07:00
Tzanio Kolev 9d191edf06 Merge pull request #5269 from mfem/adapt-lim-pa
PA kernels for adaptive limiting in TMOP (aka interface tangential relaxation)
2026-05-19 13:28:21 -07:00
Tzanio Kolev 86609f139b Merge pull request #5137 from mfem/gmsh-v4
Gmsh v4.1 support (ASCII and binary)
2026-05-19 13:27:54 -07:00
Veselin Dobrev 420fcba457 Merge pull request #5316 from lindsayad/move-attribute-names
Fix Mesh::Swap to preserve named attribute sets
2026-05-19 12:07:15 -07:00
Nuno Nobre 95acb1f85f Remove unnecessary namespace qualification 2026-05-19 17:15:03 +01:00
Nuno Nobre 1c7261db07 Fence code using parallel objs w/ MPI-conditional directive 2026-05-18 23:53:35 +01:00
Nuno Nobre e1b678664a FindPointsGSLIB: use parallel-aware ProjectDiscCoefficient 2026-05-18 19:02:30 +01:00
Tzanio Kolev d3e43a6423 Merge pull request #5332 from mfem/update-gh-actions-artifacts
Update some GH actions to the latest versions
2026-05-16 14:08:37 -07:00
Nuno NobreandAndrew Ho 8f06539b6e Do not assume true_ess_dofs is empty nor host-allocated
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-05-16 00:32:25 +01:00
Veselin Dobrev 1f6d115d78 Update to the latest versions the actions upload-artifact and download-artifact 2026-05-15 12:19:51 -07:00
Ketan Mittal 686c8416c2 Merge branch 'master' into findpts-surface 2026-05-15 09:28:59 -07:00
Nuno Nobre 30a8eb5ccf Fix GetEssentialTrueDofsVar memory allocation 2026-05-14 23:58:41 +01:00
John Camier 078e59a33b Merge branch 'master' into cudss-dev 2026-05-14 13:20:48 -07:00
Veselin Dobrev 4b61294dc2 Merge pull request #5327 from mfem/hotfix-5200
Revert PR 5200
2026-05-13 18:14:16 -07:00
Tzanio Kolev fced53cd29 Revert PR 5020 2026-05-13 15:33:20 -07:00
Vladimir Z Tomov f47447d92d tolerance 2026-05-13 14:01:40 -07:00
Vladimir Z Tomov a523710117 Go to 1st order for the tmop unit tests. 2026-05-13 12:11:32 -07:00
Tzanio KolevandVeselin Dobrev da4e1b5137 Update makefile
Co-authored-by: Veselin Dobrev <v-dobrev@users.noreply.github.com>
2026-05-12 13:54:55 -07:00
Vladimir Z Tomov 54b0a83ffd added missing .mesh to makefile 2026-05-11 09:12:01 -07:00
Tzanio Kolev b5a1660c2d Merge branch 'master' into cudss-dev 2026-05-09 10:41:10 -07:00
Tzanio Kolev 20ba3f3d0c Merge branch 'master' into findpts-surface 2026-05-09 10:37:16 -07:00
Tzanio Kolev 07cd99fc3d Merge branch 'master' into adapt-lim-pa 2026-05-09 10:37:13 -07:00
Tzanio Kolev f6eb88574f Merge branch 'master' into gmsh-v4 2026-05-09 10:37:10 -07:00
Tzanio Kolev 2631ba93ca Merge pull request #5200 from nmnobre/hypremat
Ensure hypre_CSRMatrixSetRownnz() allocs on host if ownership set to -1
2026-05-09 10:36:59 -07:00
Tzanio Kolev 5ea36c8fd6 Merge pull request #5257 from mfem/hypre-init-bug
missing hypre init in parallel miniapps
2026-05-09 10:36:16 -07:00
Mittal, Ketan b07fc2bb8e make style 2026-05-08 10:50:33 -07:00
Mittal, Ketan 53b1b8f9a9 update serial miniapp to also use surface mesh capability 2026-05-08 10:48:55 -07:00
Will Pazner 5ffc2ef502 Bugfix for GeometricMultigrid with no essential BCs 2026-05-07 18:04:46 -07:00
Vladimir Z Tomov a709bdb9ee style 2026-05-06 15:23:07 -07:00
Vladimir Z Tomov 46c01f196a Use ALF and ALFmF0 instead of ALF and ALF0. 2026-05-06 15:20:42 -07:00
Vladimir Z Tomov 240955c2cb optimized alf - alf0 computations as Ketan suggested. 2026-05-06 14:10:18 -07:00
Vladimir Z Tomov da4a8e3412 added comments 2026-05-06 13:37:04 -07:00
yuyangdai 59cef5f9e3 Add a link for communication layer library in cuDSS 2026-05-06 10:07:31 +08:00
John Camier 05be944a86 Merge branch 'master' into cudss-dev 2026-05-05 15:33:09 -07:00
John Camier 258bd917ad Merge branch 'master' into adapt-lim-pa 2026-05-05 15:33:01 -07:00
Will Pazner b9cf853dd3 Fix orientation issue in low-order periodic Gmsh meshes 2026-05-05 12:01:00 -07:00
Will Pazner 915967925c Avoid use of tellg in Gmsh reader
With zlib enabled, tellg will not work reliably with ifgzstream
2026-05-05 12:00:07 -07:00
Tzanio Kolev 88bc3b5833 Merge branch 'master' into hypre-init-bug 2026-05-05 09:13:27 -07:00
Tzanio Kolev bdd36c8982 Merge pull request #5318 from mfem/ai-policy
AI policy
2026-05-05 07:39:53 -07:00
John Camier c103cfa84a Merge branch 'master' into cudss-dev 2026-05-05 06:34:53 -07:00
John Camier 40bcad05c4 Merge branch 'master' into hypremat 2026-05-05 06:25:44 -07:00
John Camier 6c1c98e4fb Merge branch 'master' into adapt-lim-pa 2026-05-05 06:21:23 -07:00
Mittal, Ketan 3888cba7c4 minor 2026-05-04 14:58:49 -07:00
Mittal, Ketan 932b30e163 Merge branch 'findpts-surface' of https://github.com/mfem/mfem into findpts-surface 2026-05-04 14:47:33 -07:00
Mittal, Ketan 395e4b0d0e Merge branch 'master' of https://github.com/mfem/mfem into findpts-surface 2026-05-04 14:34:11 -07:00
Tzanio Kolev a7988aa845 Merge branch 'master' into ai-policy 2026-05-04 14:30:08 -07:00
Tzanio Kolev 4ec768c82b Merge pull request #5322 from mfem/fix-changelog
Fix CHANGELOG
2026-05-04 14:28:58 -07:00
Mittal, Ketan e32ea54e00 fix changelog 2026-05-04 14:11:52 -07:00
Veselin Dobrev 630a75440f Merge pull request #5299 from mfem/batchmass3d
Add element batching capabilities to 3D MassIntegrator
2026-05-04 13:47:24 -07:00
Veselin Dobrev 3ef3c8e6b4 Merge pull request #5306 from mfem/gslib-gitlab-testing
Include gslib testing on Dane
2026-05-04 13:43:23 -07:00
Will Pazner fe01ebf36c Fix bug in Gmsh nodes reader 2026-05-04 09:55:09 -07:00
Will Pazner 905de04020 Properly handle files with CRLF in Gmsh reader 2026-05-02 21:15:42 -07:00
Tzanio Kolev 145efc313d Merge pull request #5320 from mfem/fix-cmake-libceed-test
Fix a CMake test of libCEED
2026-05-02 12:55:41 -07:00
Veselin Dobrev 26b2aa5cea In .gitlab/scripts/baseline, use srun to run scripts since salloc
does NOT run the script in the allocation as does srun.

Revert the change in the number of build tasks in dane-baseline.yml.
2026-05-01 11:10:09 -07:00
Veselin Dobrev 476c148949 Adjust the number of build tasks in dane-baseline.yml 2026-05-01 09:13:40 -07:00
camierjs faa3e22816 Include 2.0 * lim_normal in the normal_inv_delta_sq factor for TMOP diag, grad & mult kernels 2026-05-01 08:02:56 -07:00
Tzanio KolevandVeselin Dobrev 8ed259be31 Update CONTRIBUTING.md
Co-authored-by: Veselin Dobrev <v-dobrev@users.noreply.github.com>
2026-04-30 11:56:57 -07:00
Tzanio Kolev 67025d49ff AI policy updates based on feedback 2026-04-30 11:56:57 -07:00
Tzanio Kolev de1dea610e AI policy updates based on feedback 2026-04-30 11:56:57 -07:00
Tzanio Kolev 9f3f5c0372 Suggested AI policy 2026-04-30 11:56:56 -07:00
John Camier 29caf08098 Merge branch 'master' into cudss-dev 2026-04-29 17:05:01 -07:00
camierjs 76d225439a Avoid recomputing Jpr_inv in TMOP_AssembleGradPA_AdaptLim_3D 2026-04-29 16:46:17 -07:00
camierjs 5ee3f03902 Avoid recomputing Jpr_inv in TMOP_AssembleGradPA_AdaptLim_2D 2026-04-29 16:31:05 -07:00
camierjs d3a1144d10 Avoid recomputing some constants 2026-04-29 15:45:44 -07:00
camierjs 20e38f3b10 Use same temporary registers for the computation of ralf & ralf0
Remove unused input vector in GetLocalStateEnergyPA_AdaptLim functions
2026-04-29 15:39:41 -07:00
Tzanio Kolev 9205efab48 Merge pull request #5319 from mfem/gslib-gnu-make-updates
GSLIB related updates to the GNU make build system
2026-04-29 15:04:10 -07:00
Veselin Dobrev 1ccc27226a Fix a CMake test of libCEED 2026-04-29 10:54:49 -07:00
Will Pazner b20f61b3b8 Use friend class for Gmsh reader; improve Doxygen documentation 2026-04-29 09:02:34 -07:00
Ketan Mittal 1d0b49e5dd Merge branch 'master' into adapt-lim-pa 2026-04-29 08:46:14 -07:00
Ketan Mittal a9dcb20e84 Merge branch 'master' into findpts-surface 2026-04-29 08:35:23 -07:00
Tzanio Kolev e52948f9e5 Merge pull request #4714 from mfem/dc-ofstream-fix-minor
Verify that `ofstream` is open
2026-04-29 08:18:43 -06:00
Veselin Dobrev c860bf20ea Merge pull request #5294 from mfem/bugfix/lorentz-test-runs
Fixing typos in lorentz miniapp test runs
2026-04-28 15:47:46 -07:00
Mittal, Ketan 68f6ce14a6 move gslib in NOTICE 2026-04-28 11:36:07 -07:00
Mittal, Ketan 78905d471c minor 2026-04-28 09:44:51 -07:00
Mittal, Ketan 61806ff1f7 add gslib to notice and license text to gslib/bb_grid_map 2026-04-28 08:27:09 -07:00
Veselin Dobrev 0d3195e69b Fix issue #5314 and other tweaks.
* 'make style' now checks if all git source files are selected for formatting.
* In examples/makefile, propagate the target 'test-noclean' to subdirectories.
* In miniapps/plasma/makefile, use logic similar to examples/makefile to
  propagate targets to subdirectories.
* Other small fixes.
2026-04-28 06:32:57 -07:00
daiyuyang e217864f16 Merge branch 'master' into cudss-dev 2026-04-28 15:17:58 +08:00
yuyangdai de8aacddce Replace enum class with enum 2026-04-28 14:59:09 +08:00
Alex Lindsay 7b1656e19f Don't hard code array comparisons 2026-04-27 15:59:09 -07:00
Mittal, Ketan c228538c17 fix loop range in interpolate_local_1 2026-04-27 10:42:23 -07:00
Mittal, Ketan a8f5fac0bb Merge branch 'findpts-surface' of https://github.com/mfem/mfem into findpts-surface 2026-04-27 10:25:36 -07:00
Mittal, Ketan 206eb51618 remove unused argument from interpolate kernels 2026-04-27 10:25:24 -07:00
Ketan MittalandJohn Camier 02226f934b Fix typos
Co-authored-by: John Camier <camierjs@gmail.com>
2026-04-27 10:16:01 -07:00
Mittal, Ketan 3d0ba2251a fix scratch space size used for bounding box calculation and some other cosmetic changes to the bounding box methods 2026-04-27 09:34:06 -07:00
Mittal, Ketan b400ee6741 fix volume kernels, and add some undef 2026-04-27 09:27:13 -07:00
Mittal, Ketan f40335f9e7 fix include and flags in findpts kernels 2026-04-27 09:21:55 -07:00
Mittal, Ketan f0de33e33b remove unused argument and header include 2026-04-27 08:51:05 -07:00
Veselin Dobrev f37a596173 Fix a build issue: in the top makefile ensure miniapps/common is built
before building miniapps/gslib.
2026-04-27 07:26:12 -07:00
Tzanio Kolev 7ff0bd3bb0 Merge branch 'master' into dc-ofstream-fix-minor 2026-04-26 14:30:17 -06:00
Alex LindsayandClaude Sonnet 4.6 c982aa2448 Add unit test for Mesh::Swap preserving named attribute sets
Covers the regression where attr_sets maps were not swapped, silently
dropping all named element/boundary sets on any move or swap.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-24 13:15:09 -07:00
Alex LindsayandClaude Sonnet 4.6 94d1238637 Fix Mesh::Swap to preserve named attribute sets
Mesh::Swap swapped attributes and bdr_attributes but omitted the
attr_sets maps inside attribute_sets and bdr_attribute_sets, causing
all named boundary/element sets to be silently lost on any move or
swap of an mfem::Mesh.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-24 13:00:00 -07:00
Will Pazner d218d38af3 Factor out ReadBinaryOrASCII and Skip to binaryio.hpp 2026-04-24 12:13:28 -07:00
Andrew Ho 04dd962b6d review comments 2026-04-23 15:58:25 -07:00
Mittal, Ketan 383914db9a use MFEM's Mpi class to initialize instead of MPI_Init directly 2026-04-23 14:27:01 -07:00
Andrew HoandJohn Camier f77d238a5d Update fem/dgmassinv_kernels.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2026-04-23 10:52:43 -07:00
John Camier 84996ce32f Merge branch 'master' into batchmass3d 2026-04-23 06:26:09 -07:00
camierjs 5fa7ab3602 revert back tol_fe to 1e-5 for parallel runs 2026-04-22 20:21:24 -07:00
camierjs 48cb5996b7 Merge branch 'master' into adapt-lim-pa 2026-04-22 18:47:58 -07:00
camierjs 8e36285a98 make style 2026-04-22 18:47:28 -07:00
camierjs 27db27b088 GPU runs w/ 1e-6 tol_fe 2026-04-22 17:56:28 -07:00
camierjs 8339ee0fe9 wip gpu grad limit 2026-04-22 15:26:14 -07:00
Mittal, Ketan f2b64de28f Merge branch 'master' of https://github.com/mfem/mfem into gslib-gitlab-testing 2026-04-22 12:27:50 -07:00
Mittal, Ketan 3415b0f3d4 run serial miniapps on 1 run when mfem is built with MPI 2026-04-22 12:26:45 -07:00
Mittal, Ketan 9f18d7e044 documentation, use newt_tol instead of new variable tol, and update surface tolerance based on experiments in other branch 2026-04-22 09:53:52 -07:00
Veselin Dobrev 3e31395f85 Fix another minor compiler warning.
Update the CMake tests in miniapps/electromagnetics to match the makefile.
2026-04-21 23:39:11 -07:00
Will Pazner 90820b76cf Don't need PairHash now that general PairHasher is merged 2026-04-21 20:04:39 -07:00
Will Pazner 3adb2add4c Don't pass istream to ReadGmsh2Mesh or ReadGmsh4Mesh 2026-04-21 16:37:26 -07:00
Will Pazner 5979dd1cce Merge remote-tracking branch 'origin/master' into gmsh-v4
# Conflicts:
#	mesh/mesh_readers.cpp
2026-04-21 16:35:38 -07:00
Mittal, Ketan be999694b0 Merge branch 'master' of https://github.com/mfem/mfem into findpts-surface 2026-04-21 15:42:23 -07:00
Mittal, Ketan 5d3b9ea727 reviewer comments, minor documentation fix, and global map empty rank fix 2026-04-21 15:42:09 -07:00
Mittal, Ketan a4d01470d8 initialize interpolation vector, and fix empty ranks for the grid maps, and some documentation 2026-04-21 13:06:37 -07:00
Tzanio Kolev b2de4c4ba1 Merge pull request #5239 from nmnobre/ir
A few small fixes and feature additions
2026-04-21 12:27:07 -06:00
John Camier a055c7ec63 Merge branch 'master' into adapt-lim-pa 2026-04-21 10:11:08 -07:00
Andrew Ho 6ea799e385 Merge branch 'master' into batchmass3d 2026-04-20 09:04:25 -07:00
Ketan Mittal 4e83a1604c Merge branch 'master' into findpts-surface 2026-04-19 08:40:28 -07:00
Tzanio Kolev a713e386c2 Merge pull request #5278 from mfem/fix-umpire-dep
fix CMake umpire build and install
2026-04-18 18:57:23 -06:00
Tzanio Kolev 4155b0bdda Merge pull request #4645 from mfem/ex37
Enhancements, optimization for ex37
2026-04-18 18:55:39 -06:00
Ketan Mittal dd0d879e7b Merge branch 'master' into findpts-surface 2026-04-17 20:56:17 -07:00
Andrew Ho f1561e47d1 Merge branch 'master' into batchmass3d 2026-04-17 10:07:18 -07:00
Veselin Dobrev f1174bfbf5 Added new methods in class ParFiniteElementSpace: HaveDofSigns and
ApplyDofSigns.

Used the new methods to fix bugs in:
* the ParGridFunction constructor that reads input from a stream
* the method ParGridFunction::SaveAsOne
2026-04-17 09:08:19 -07:00
Mittal, Ketan abf5fedc5b include hypre with cuda on matrix 2026-04-16 21:03:11 -07:00
John Camier 4195e4ea2f Merge branch 'master' into adapt-lim-pa 2026-04-16 14:26:00 -07:00
Mittal, Ketan d183f43c96 Merge branch 'gslib-gitlab-testing' of https://github.com/mfem/mfem into gslib-gitlab-testing 2026-04-16 12:56:36 -07:00
Mittal, Ketan a545b94ad7 enable testing on matrix as well 2026-04-16 12:56:08 -07:00
John Camier 94828dbdd0 Merge branch 'master' into hypremat 2026-04-16 04:46:18 -07:00
Ketan Mittal 12eefe3c41 Merge branch 'master' into gslib-gitlab-testing 2026-04-14 12:58:08 -07:00
Tzanio Kolev dbbd425a22 Merge pull request #5186 from mfem/particles-pic-dev-pr
Electrostatic PIC
2026-04-14 13:54:29 -06:00
Tzanio Kolev 156f338e49 Merge pull request #4626 from mfem/mfem-v13-mesh-reader-fix-issue-4625
[BUG] Fix for mfem v13 mesh format reader
2026-04-14 13:38:27 -06:00
Mittal, Ketan 8e33891c07 initial commit 2026-04-14 12:13:15 -07:00
Ketan Mittal 62fbabe3a5 Merge branch 'master' into findpts-surface 2026-04-14 11:49:44 -07:00
Will Pazner 53581cb5b7 Merge pull request #5300 from mfem/fix-nvcc-static_cast-warnings
Fix nvcc warnings
2026-04-14 11:04:36 -07:00
Andrew Ho 7b4df2d374 Merge branch 'master' into batchmass3d 2026-04-13 09:29:35 -07:00
Andrew Ho 12509fda28 Merge branch 'master' into fix-umpire-dep 2026-04-13 09:29:28 -07:00
Nuno Nobre a9b36b1e5e Merge branch 'master' into test 2026-04-13 12:05:37 +01:00
Veselin Dobrev 64ef39bbe6 Merge pull request #5291 from lindsayad/fix-petsc-b-allocation
Don't attempt to allocate B if b is non-empty in PetscNonlinearSolver
2026-04-12 19:03:40 -07:00
Veselin Dobrev 7985a225bb Fix nvcc warnings about static_cast<const int> 2026-04-12 18:11:16 -07:00
Andrew Ho 2d7460bde1 fixed bug in how tidz was set
128 seems to offer a slightly better balance for low and high orders
2026-04-11 10:21:10 -07:00
Veselin Dobrev 65acd08d38 Small tweak in ParMesh::Print 2026-04-10 19:50:10 -07:00
Veselin Dobrev 0acc85d962 In ParMesh::PrintAsOne() fix the interface attribute to make it different
from all real boundary attributes.
2026-04-10 17:41:26 -07:00
Andrew Ho 3c45d59813 cap CPU version to batch size 1 2026-04-10 17:23:48 -07:00
Andrew Ho 63acbeb8c0 use the same batching pattern as elsewhere, hopefully fixes bugs 2026-04-10 14:39:20 -07:00
Andrew Ho 9bf6819f7a Merge remote-tracking branch 'base/fix-umpire-dep' into batchmass3d 2026-04-10 13:53:32 -07:00
Andrew Ho bed2cc5735 implemented 3D element batching for mass integrator 2026-04-10 13:48:16 -07:00
Veselin Dobrev 4e35641c28 Fix GCC warning 2026-04-10 12:57:48 -07:00
Veselin Dobrev 8594867ab6 Modify ParMesh::PrintAsOne() to use the settings of SetPrintShared() and
SetPrintInterfaces().

Added some suggestions/questions as FIXME comments.
2026-04-10 12:21:02 -07:00
John Camier 6e0fbbd3bf Merge branch 'master' into cudss-dev 2026-04-09 06:43:58 -07:00
John Camier d8fd6d95c0 Merge branch 'master' into adapt-lim-pa 2026-04-09 06:39:09 -07:00
Veselin Dobrev 72f83edd53 Fix compiler warnings when GSLIB is enabled with some extra warning flags 2026-04-08 18:41:57 -07:00
Andrew Ho 6c5f513eaa Merge branch 'master' into fix-umpire-dep 2026-04-08 16:37:59 -07:00
Mark L. Stowell 75cc8433e9 Merge branch 'master' into bugfix/lorentz-test-runs 2026-04-08 15:19:45 -04:00
Veselin Dobrev f700d97549 Merge pull request #5287 from mfem/pr-4626-tweaks
Proposed tweaks for PR 4626
2026-04-08 11:16:23 -07:00
Veselin Dobrev ec39b3509c Merge pull request #5292 from mfem/saveAsOneAttrs
Propagate named attribute sets in ParMesh::GetSerialMesh
2026-04-08 11:13:00 -07:00
Veselin Dobrev 449ec725e2 Merge branch 'master' into particles-pic-dev-pr 2026-04-08 11:04:40 -07:00
Will Pazner 399d8e1e9b Merge pull request #5293 from mfem/ci-fix-brew-info
GitHub CI fix
2026-04-08 11:00:36 -07:00
yuyangdai a2b34ab650 Add conditional compilation for cudss_solver in ex1p 2026-04-08 15:05:50 +08:00
yuyangdai f478f687ff Update cuDSS solver integration:
- Use the full name of default communication library and threading library.
- Check the `CUDSS_COMM_LIB` and `CUDSS_THREADING_LIB` in environment first.
- Add the `cudss-solver` option in ex1p
2026-04-08 13:48:56 +08:00
daiyuyangandWill Pazner 0bf998510f Update config/defaults.mk
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2026-04-08 09:45:40 +08:00
Stowell, Mark L. 7330aca4e6 Fixing typoes in lorentz miniapp test runs 2026-04-07 21:04:13 -04:00
Veselin Dobrev 9ebfcf05af In the electrostatic PIC miniapp:
* Fix the out-of-source build.
* Use the same test options in CMake as in GNU make.
* Ensure the test is run from the GNU makefile.
2026-04-07 17:42:39 -07:00
Veselin Dobrev 10dbed9658 In GitHub CI, fix the parsing for the new formatting of 'brew info' 2026-04-07 17:23:31 -07:00
thatguynoe be1db1e4b7 initialize y, c, f_c 2026-04-07 20:07:28 -04:00
Noe Reyes 37fcdc1816 Merge branch 'master' into ex37 2026-04-07 19:50:41 -04:00
thatguynoe 0af98d7ff6 parameter c no longer present 2026-04-07 19:43:06 -04:00
thatguynoe cb6192167c correct assert message 2026-04-07 19:39:47 -04:00
thatguynoe bdf6aa6369 ensure root search interval is valid
We now look for a root of the function f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω) within the interval [a,b], where a := -‖αG‖_∞ and b := ‖αG‖_∞, and α and G are as in Step 4 and Step 5. It follows that f(a) ≤ ∫_Ω sigmoid(ψ + αG) dx - θ vol(Ω) ≤ f(b), and the inner quantity equals 0 since ψ_new := ψ_prev - αG in Step 5 and ∫_Ω sigmoid(ψ_new) dx - θ vol(Ω) = 0. This ensures f(a) ≤ 0 ≤ f(b), as required by the Illinois method.
2026-04-07 19:39:30 -04:00
Stowell, Mark L. da40ac4f2d Adding pic subdirectory to CMakeLists.txt as discussed in PR meeting 2026-04-07 16:51:07 -04:00
Will Pazner f09a062c04 Merge pull request #4800 from mfem/warn-gridfunc
Add VectorDim error checks for projecting coefficients onto a GridFunction
2026-04-07 11:13:31 -07:00
Alex Lindsay 0c97d6f375 Never allocate B 2026-04-06 20:21:41 -07:00
thatguynoe bff5d5e0cb correct comments 2026-04-06 22:14:46 -04:00
thatguynoe 26a152fb11 set y = 0.0 2026-04-06 22:11:14 -04:00
Veselin Dobrev aed9c8ef4a Merge pull request #5285 from mfem/uuid-fix
Fix signed char issue in Device::GetUUID
2026-04-06 15:52:32 -07:00
Veselin Dobrev e4e85e28ef Merge pull request #5289 from adam-sim-dev/fix_missing_parentheses
Fixed missing parenthesis in the comment
2026-04-06 15:43:19 -07:00
Veselin Dobrev fff973f192 Merge pull request #5246 from mfem/hughcars/simplex-quadrature-dev
Add positive-weight simplex quadrature rules for orders 0-20
2026-04-06 15:39:49 -07:00
John Camier dbaff07ae9 Merge branch 'master' into adapt-lim-pa 2026-04-04 07:10:04 -07:00
Veselin Dobrev 775f06c43b Adjust seed values in sample runs in ex12p to ensure LOBPCG convergence in
older hypre versions.
2026-04-03 14:55:08 -07:00
schnmich 18ff1d8289 linting 2026-04-03 11:21:06 -06:00
Nuno Nobre bca03a17af Avoid unneeded overrides of ProjectDiscCoefficient 2026-04-03 18:17:43 +01:00
schnmich 5a0962c674 add attributes to GetSerialMesh 2026-04-03 11:05:03 -06:00
Ketan Mittal 6479b2607d Merge branch 'master' into particles-pic-dev-pr 2026-04-03 08:57:26 -07:00
Nuno Nobre c1de6939f9 Remove unused ThresholdRefiner member current_sequence 2026-04-03 16:14:37 +01:00
yuyangdai d916299a49 Update INSTALL documentation for CUDSS requirements and backend support 2026-04-03 14:33:59 +08:00
yuyangdai 10840ac6b4 Refactor CuDSSSolver initialization. 2026-04-03 13:56:20 +08:00
yuyangdai b437016f6e Refactor cuDSS library path resolution in defaults.mk for improved handling of missing libraries 2026-04-02 15:01:00 +08:00
Veselin Dobrev 0d999709e6 Merge pull request #4917 from Sbozzolo/master
Improve error message for gmsh versions != 2.2
2026-04-01 20:20:15 -07:00
Veselin Dobrev 9300f47c83 Added review suggestions 2026-04-01 19:51:35 -07:00
Mittal, Ketan 75e49b217c modify top level makefile and run make style 2026-04-01 15:45:46 -07:00
Tzanio Kolev c2649eb998 Merge pull request #5245 from mfem/bugfix/chapman39/bilininteg-no-mod
bilininteg: eliminate usage of modulus to avoid llvm backend bug
2026-04-01 11:35:17 -07:00
yuyangdai 5170bbe010 Fix CUDSS library checks and improve status output in CMake and makefile 2026-04-01 16:16:01 +08:00
yuyangdai f8fcdf6a97 Update cuDSS configuration and library paths in CMake and makefiles; refactor cuDSS solver integration.
- Add `MFEM_CUDSS_COMM_LIB` for OpenMPI communication library
- Add 'MFEM_CUDSS_THREADING_LIB' for threading library
- Add SetMatrixCuDSS() to set matrix values for cudss
- Remove SetMatrixSortRow()
- Remove unused options and simplify conditions in ex1.cpp and ex1p.cpp.
2026-04-01 13:49:02 +08:00
Will Pazner a1ce49fb57 Merge pull request #5290 from mfem/ci-update-action-versions-2
Update the action `actions/cache/restore` to `v5`
2026-03-31 20:36:07 -07:00
Alex Lindsay f58cfc8170 Should not allocate B if b non-empty
Otherwise there will be an error in PlaceMemory
2026-03-31 20:25:34 -07:00
Alex Lindsay 6c837d2954 Add test of PetscNonlinearSolver with non-empty RHS 2026-03-31 20:24:22 -07:00
Veselin Dobrev 5b37c3b595 Update the action actions/cache/restore to v5 2026-03-31 16:45:21 -07:00
Andrew Ho b46baa5f5e Merge branch 'master' into fix-umpire-dep 2026-03-31 15:59:31 -07:00
Will Pazner e49f9f7988 Merge pull request #5288 from mfem/ci-update-action-versions
Update some GitHub actions to new versions
2026-03-31 11:03:48 -07:00
Will Pazner fdfc019cc1 Reviewer feedback 2026-03-31 09:40:16 -07:00
adam-sim-dev 7c36b55628 Fixed missing parenthesis in the comment 2026-03-31 14:29:39 +08:00
Veselin Dobrev faa73ef554 Updated the github/codeql-action/* actions to the latest, v4 2026-03-30 11:50:40 -07:00
Veselin Dobrev ecb6b06aa0 Updated actions/checkout to the latest major version, v6 2026-03-30 11:44:41 -07:00
Veselin Dobrev af4649a088 Update actions/{checkout,cache} to v5
Update github/codeql-action/* to v3
2026-03-30 10:30:31 -07:00
Andrew Ho a9f58f3982 check for null 2026-03-30 09:31:37 -07:00
Andrew Ho 6de6675783 fix compiler complaints 2026-03-30 09:04:18 -07:00
Andrew Ho 085ee02a29 compile error 2026-03-30 08:56:18 -07:00
Andrew Ho 9a124335a7 redundant checks 2026-03-30 08:53:36 -07:00
Andrew Ho 91d5e490aa Merge branch 'master' into warn-gridfunc 2026-03-30 08:53:08 -07:00
Veselin Dobrev 610196629e Merge branch 'mfem-v13-mesh-reader-fix-issue-4625' into pr-4626-tweaks 2026-03-30 00:50:29 -07:00
Veselin Dobrev 8453b4008d Merge branch 'master' into mfem-v13-mesh-reader-fix-issue-4625 2026-03-30 00:49:11 -07:00
Veselin Dobrev fab2afd8dc Proposed tweaks for PR 4626 2026-03-30 00:09:55 -07:00
John Camier d10c908b38 Merge branch 'master' into cudss-dev 2026-03-29 17:51:58 -07:00
Tzanio Kolev dd931b2584 Merge pull request #5219 from mfem/najlkin/project-bdr-coeff-rtnd
Projection of  scalar coefficients on RT grid functions
2026-03-29 12:53:11 -07:00
Tzanio Kolev 8a42ea2834 Merge pull request #5198 from mfem/najlkin/fix-ex22p-glvis
[BUG] Fixed visualization in example 22
2026-03-29 12:52:42 -07:00
Tzanio Kolev 24e5d5fc0a Merge pull request #4781 from mfem/najlkin/extrd-1d-vec
Extrusion of vector 1D grid functions
2026-03-29 12:51:54 -07:00
Tzanio Kolev 6722dd7a70 Merge pull request #5276 from mfem/bugfix/arrays-by-name-load
Adding bugfix and unit test which would have caught the bug
2026-03-29 12:51:15 -07:00
Tzanio Kolev cb862cbfa1 Merge pull request #5252 from mfem/assemble-face-integrator-fix
Added VDOFs transformation for boundary integration
2026-03-29 12:50:40 -07:00
Vladimir Z Tomov 73aceea741 EnableAdaptiveLimiting for ComboIntegrator 2026-03-27 17:46:48 -07:00
Will Pazner f7445844ba Fix signed char issue in Device::GetUUID 2026-03-26 15:23:21 -07:00
John Camier 89f85cee21 Merge branch 'master' into cudss-dev 2026-03-26 13:49:44 -07:00
John Camier 0e9a9d9f7c Merge branch 'master' into adapt-lim-pa 2026-03-26 13:49:24 -07:00
Hugh Carson 672e2a442b Address PR feedback
- Use [IntegrationRules] test tag instead of [PositiveWeightRules]
- Remove redundant case 21: (default branch handles it via the overwrite guard)
- Remove trailing blank line
2026-03-26 12:18:59 -04:00
rzhangbq 3e1f10daea incorporating PR #5282 2026-03-24 21:51:15 -07:00
Ketan Mittal c25be44dd6 Merge branch 'master' into hypre-init-bug 2026-03-24 21:29:12 -07:00
Veselin Dobrev 416536eb9d Merge pull request #4941 from mfem/globalvec_debug
GlobalVector bug fix
2026-03-24 12:04:48 -07:00
rzhangbq 35778347d0 resolve double-assigning b 2026-03-24 10:51:44 -07:00
camierjs 00da00b93a Add Copyright headers 2026-03-24 08:42:34 -07:00
camierjs 4eca673111 Cosmetic trailing spaces 2026-03-24 08:33:06 -07:00
John Camier 4507a02249 Merge branch 'master' into cudss-dev 2026-03-24 08:01:00 -07:00
Andrew Ho f557e348da removed comments 2026-03-23 14:02:52 -07:00
Andrew Ho 881598e5da also ensure boundary element is a scalar range type 2026-03-23 13:29:46 -07:00
Andrew Ho 564b7ab4ec fixed error message 2026-03-23 13:12:53 -07:00
Andrew HoandJan Nikl 3f2f925400 Update fem/gridfunc.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-23 13:09:37 -07:00
Jan Nikl 463e34dc7f Fixed spelling of transverse. 2026-03-23 10:33:51 -07:00
Tzanio Kolev 55e42eeefe Merge pull request #5238 from mfem/face-nbr-restr-vdim-bugfix
Fix bug in ParL2FaceRestriction with vdim > 1
2026-03-22 10:24:53 -07:00
Vladimir Z Tomov 3eac6fe764 copilot review suggestions. 2026-03-19 17:23:49 -07:00
Andrew Ho 077954d4b3 fix CMake umpire build and install 2026-03-19 13:43:48 -07:00
Stowell, Mark L. 9a456b908e Adding bugfix and unit test which would have caught the bug 2026-03-18 15:50:59 -07:00
Andrew HoandJan Nikl 616839388a Update tests/unit/fem/test_var_order.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-17 14:14:17 -07:00
Andrew HoandJan Nikl 2fda3db982 Update tests/unit/fem/test_var_order.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-17 14:13:57 -07:00
Andrew Ho 4823a33a6a fixed not checking the correct rangedim 2026-03-17 12:17:13 -07:00
Veselin Dobrev a96319e0be Small change in error message + formatting. 2026-03-17 12:16:46 -07:00
Andrew Ho 5f4283f512 fixed comment; vector coefficient is still spacedims 2026-03-17 11:37:02 -07:00
Andrew Ho 8735d28561 fixed GetPhysRangeDim and GetPhysCurlDim not being virtual 2026-03-17 11:33:55 -07:00
thatguynoe c9f7a90f81 store the diffusion global matrix 2026-03-17 13:01:43 -04:00
Noe Reyes 3b35d8210d Merge branch 'master' into ex37 2026-03-17 11:33:05 -04:00
Nuno Nobre 3babbe993b Clarify L2ZienkiewiczZhuEstimator only requires ComputeElementFlux() 2026-03-17 09:21:02 +00:00
thatguynoe 5f5421fde2 rename bool flag 2026-03-16 23:44:11 -04:00
Ketan Mittal f1ed582828 Merge branch 'master' into findpts-surface 2026-03-16 18:15:05 -07:00
Vladimir Z Tomov 07e0d7cd4f minor 2026-03-16 16:42:57 -07:00
Vladimir Z Tomov c68cc62143 Merge branch 'master' into adapt-lim-pa 2026-03-16 16:40:27 -07:00
Vladimir Z Tomov 2c02b41d71 style 2026-03-16 15:59:15 -07:00
Vladimir Z Tomov d71d1005f9 minor 2026-03-16 15:46:01 -07:00
thatguynoe 8644c8a8dd rename boundary dof extraction function 2026-03-16 18:27:26 -04:00
thatguynoe b863dd186f use MPITypeMap<real_t>::mpi_type 2026-03-16 18:15:57 -04:00
Noe ReyesandDohyun Kim 66702d831c correct return type in proj function
Co-authored-by: Dohyun Kim <dhkim.cse@gmail.com>
2026-03-16 18:12:22 -04:00
Vladimir Z Tomov b8f1071168 unused variable 2026-03-16 14:08:34 -07:00
Vladimir Z Tomov d88529d632 corrections 2026-03-16 13:57:46 -07:00
Vladimir Z Tomov 7bd028b7fe Corresponding edits in mesh-optimizer 2026-03-16 13:39:33 -07:00
Vladimir Z Tomov 339f20ea7f style 2026-03-16 13:19:13 -07:00
Vladimir Z Tomov 336d80e93a missing function call 2026-03-16 13:15:59 -07:00
Vladimir Z Tomov b64a189215 Unit test fixes. 2026-03-16 13:07:20 -07:00
Vladimir Z Tomov fc76ff8b2f unit test 2026-03-13 10:12:15 -07:00
Andrew Ho a10c7a943b Merge branch 'master' into warn-gridfunc 2026-03-13 09:47:50 -07:00
Nuno Nobre 60ab6ab8f5 New Is(Par)SubMesh methods to determine descendance 2026-03-13 15:37:28 +00:00
Tzanio Kolev fa89c5e98c Merge pull request #4856 from mfem/phys-range-dim
Range and curl dimension in physical space
2026-03-13 07:44:40 -07:00
Tzanio Kolev 0980bda63b Merge pull request #5215 from balay/barry/update-for-petsc-v3.25-PetscCtx
Update to change in PETSc API (in v3.25) for PetscCtx and PetscCtxRt
2026-03-13 07:44:06 -07:00
Nuno Nobre 878df1fef2 Revert "Allow evals of GridFunctionCoefficient on submeshes"
This reverts commit 8baa46babd.
2026-03-13 11:44:50 +00:00
Vladimir Z Tomov e0a65ffaaf Avoid reassembly of quad poitns grads and hessians. 2026-03-12 18:08:43 -07:00
Will Pazner a1758e51e5 Merge remote-tracking branch 'origin/master' into face-nbr-restr-vdim-bugfix 2026-03-12 18:02:19 -07:00
Will Pazner ccf84aab7c Use constexpr in unit test 2026-03-12 18:01:40 -07:00
Tzanio Kolev 61c7187a86 Review comments 2026-03-12 11:17:14 -07:00
Vladimir Z Tomov a6bad19b8f style 2026-03-12 10:07:59 -07:00
Vladimir Z Tomov 174d991451 Merge branch 'master' into adapt-lim-pa 2026-03-11 14:46:00 -07:00
Hugh Carson 96eff4684f Remove unused private helper methods from IntegrationRule
AddTriPoints3R, AddTetPoints4b, and AddTetPoints12bc are no longer
called after the legacy simplex rules were removed.
2026-03-11 16:58:36 -04:00
Hugh Carson b6255fc825 Use exact fractions for trivial quadrature weights and coordinates
For rules where the mathematical value is an exact simple fraction
(midpoint weights, equal-weight symmetric rules), use the fraction
directly rather than the Polyquad decimal expansion. Cleaner to read
and avoids any rounding from decimal-to-double conversion.
2026-03-11 16:55:28 -04:00
Hugh Carson 18d27f6ffb Restore original function order in intrules.cpp
Move TriangleIntegrationRule before SquareIntegrationRule to match
the original file layout, reducing diff noise against master.
2026-03-11 16:38:01 -04:00
Hugh Carson 7bfb57ef17 Remove legacy simplex rules; positive-weight rules are now the default
The positive-weight rules now cover the full tabulated range for both
triangles (0-25) and tetrahedra (0-20), so the old rules with negative
weights are no longer needed. Remove the SimplexQuadrature enum,
simplex_type member, and legacy rule functions — all simplex quadrature
now uses positive-weight rules by default, with Grundmann-Moller
fallback for higher orders.
2026-03-11 15:25:29 -04:00
Hugh Carson ab394d795e Add existing order 21-25 triangle rule to positive-weight rules
The 126-point degree-25 rule already has all positive weights.
Copy it into TrianglePositiveIntegrationRule so the positive-weight
path covers orders 0-25.
2026-03-11 15:25:25 -04:00
Tzanio KolevandCopilot 4fbefc6987 Update mesh/pmesh.cpp
Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
2026-03-11 12:20:13 -07:00
Tzanio Kolev 8dd75d2548 typo 2026-03-11 11:32:06 -07:00
Tzanio Kolev b0c2ec505f Added optional output of material interfaces in parallel 2026-03-11 11:20:59 -07:00
Will Pazner 82abd48bba Merge pull request #5080 from mfem/cmake-config
CMake config.mk for CUDA and HIP
2026-03-11 12:18:25 -04:00
Andrew Ho cad9cc4c82 fixed checks 2026-03-10 20:56:14 -07:00
Andrew Ho 4dc741ca48 fix merge with master
still need to update bdr project coefficient checks
2026-03-10 17:11:17 -07:00
Andrew Ho 918eb114d3 Merge branch 'phys-range-dim' into warn-gridfunc 2026-03-10 16:42:15 -07:00
chapman39 3341acf0f7 add comments showing each modulus replacement 2026-03-10 15:10:51 -07:00
Tucker Hartland 287cb24d0a Merge branch 'master' into globalvec_debug 2026-03-10 11:35:19 -07:00
Andrew Ho 70370b6241 Merge branch 'master' into warn-gridfunc 2026-03-10 11:21:36 -07:00
Tzanio Kolev d4374a9d5f Merge branch 'master' into cmake-config 2026-03-10 11:08:23 -07:00
Tzanio Kolev dcd3a25730 Merge branch 'master' into barry/update-for-petsc-v3.25-PetscCtx 2026-03-10 11:01:46 -07:00
Tzanio Kolev 9fb2327be9 Merge branch 'master' into ir 2026-03-10 10:59:26 -07:00
Wouter Tonnon ea291fb157 Merge branch 'master' into assemble-face-integrator-fix 2026-03-10 11:02:32 +01:00
Wouter Tonnon fce4ae7bb0 extended to MixedBilinearForm 2026-03-10 11:00:51 +01:00
Alex Tyler Chapman ef44f047aa Merge branch 'master' into bugfix/chapman39/bilininteg-no-mod 2026-03-09 16:41:43 -07:00
chapman39 ae002f7369 added comment 2026-03-09 14:36:47 -07:00
Andrew Ho a03095d84d Merge branch 'master' into hypre-init-bug 2026-03-09 09:45:36 -07:00
Vladimir Z Tomov 9ee63d6521 Initial 3D PA for the adaptive limiting + setup 3D problem. 2026-03-06 10:59:11 -08:00
Ketan Mittal d0324074c1 Merge branch 'master' into findpts-surface 2026-03-06 08:33:27 -08:00
Jan Nikl e4cd3f9e18 Merge branch 'master' into najlkin/project-bdr-coeff-rtnd 2026-03-05 14:48:38 -08:00
Mittal, Ketan 916e0b6acc Merge branch 'master' of https://github.com/mfem/mfem into particles-pic-dev-pr 2026-03-05 14:46:51 -08:00
Andrew Ho 0f99528c62 Merge branch 'master' into phys-range-dim 2026-03-05 12:45:42 -08:00
Veselin Dobrev ddfd74e899 Merge pull request #5255 from mfem/catch-tests
fix clang compiler warning for __COUNTER__
2026-03-05 12:39:29 -08:00
Vladimir Z Tomov 983d0f4361 Fixed a bug - missing assembly before the MultPA. 2026-03-05 11:32:47 -08:00
Mark L. Stowell 0248720eeb Merge branch 'master' into phys-range-dim 2026-03-05 09:35:14 -08:00
Andrew Ho 30016c83b8 Merge branch 'master' into hypre-init-bug 2026-03-05 09:28:46 -08:00
Andrew Ho feded39641 Merge branch 'master' into catch-tests 2026-03-05 09:28:31 -08:00
Andrew Ho cf5d93604e Merge branch 'catch-tests' into hypre-init-bug 2026-03-05 09:28:21 -08:00
Mittal, Ketan 65d36906c7 minor 2026-03-04 20:31:06 -08:00
Paul Hilscher 327f104c53 Merge branch 'master' into mfem-v13-mesh-reader-fix-issue-4625 2026-03-05 12:20:50 +09:00
Mittal, Ketan 4f01b485df update gitignore 2026-03-04 18:44:35 -08:00
Mittal, Ketan fc7f3fddfe merge with master, resolve conflicts, and move pic inside plasma 2026-03-04 18:43:34 -08:00
Rushan ZhangandJan Nikl 937651e509 Change test case
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 18:23:35 -05:00
Vladimir Z Tomov ada42c9fd8 Diagonal PA assembly 2D of the adaptive limiting. 2026-03-04 14:34:36 -08:00
Rushan ZhangandJan Nikl 16d9a2c311 Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:25:20 -05:00
Rushan ZhangandJan Nikl c652a269ca Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:25:07 -05:00
Rushan ZhangandJan Nikl 75bb2016a9 Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:24:35 -05:00
Rushan ZhangandJan Nikl 60d5a6cb77 Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:24:13 -05:00
Rushan ZhangandJan Nikl 3ee5f840ce Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:23:46 -05:00
Rushan ZhangandJan Nikl abbad56994 Update Energy computation
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-03-04 14:22:58 -05:00
Tzanio Kolev 09128b9a5d Merge pull request #5240 from mfem/bugfix/chapan39/use-mfem-abort-kernel-in-device
dfem integrate: use mfem abort kernel in device code
2026-03-04 09:55:23 -08:00
Tzanio Kolev 68383b462b Merge pull request #5231 from mfem/plasma-dir-dev
Plasma Miniapp Directory
2026-03-04 09:54:48 -08:00
Tzanio Kolev 24d5609585 Merge pull request #5212 from mfem/najlkin/fix-cmplx-assign
[BUG] Complex grid function copy assignment
2026-03-04 09:54:25 -08:00
Andrew Ho 0b802d8fce missing hypre init in parallel miniapps 2026-03-04 08:37:02 -08:00
Jan Nikl abdcf82d70 Added scalar unit test of ProjectBdrCoefficientNormal(). 2026-03-03 12:59:53 -08:00
Jan Nikl ad93d526b7 Added a unit test for vector ProjectBdrCoefficientNormal(). 2026-03-03 12:05:57 -08:00
Andrew Ho 670a3f9a45 comment on why TPL_LIBRARIES is reversed twice 2026-03-03 11:43:02 -08:00
Jan Nikl 87c1a5cb77 Made the ProjectBdrCoefficientNormal check non-debug. 2026-03-03 11:26:25 -08:00
Andrew Ho 7baae02d65 Merge remote-tracking branch 'base/cmake-config' into cmake-config 2026-03-02 16:31:18 -08:00
Andrew Ho 728a0f313b move cudart to MFEM_EXT_LIBS 2026-03-02 16:30:31 -08:00
Andrew Ho 1bb624e2a8 fix clang compiler warning for __COUNTER__ 2026-03-02 14:05:28 -08:00
Tzanio Kolev ee7ccd6464 Merge branch 'master' into plasma-dir-dev 2026-03-02 11:56:51 -08:00
Stowell, Mark L. a3ae5a6f01 Changing copyright date to pass CI checks 2026-03-02 09:22:29 -08:00
Tzanio Kolev 5ce6e90ceb Merge pull request #5253 from mfem/bugfix-nurbs-orientation
NURBS orientation bug fix
2026-03-01 11:46:34 -08:00
Tzanio Kolev 03ba184adb Merge pull request #4882 from mfem/debug-mem-silent
Add option to run debug memory backend without issuing errors
2026-03-01 11:46:15 -08:00
Tzanio Kolev cfa87477da Merge pull request #5248 from lindsayad/dont-do-math-with-enums
Don't do arithmetic with enums
2026-03-01 11:45:47 -08:00
Andrew HoandNuno Nobre 9243d00549 Update config/cmake/modules/MfemCmakeUtilities.cmake
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-28 14:53:22 -08:00
Andrew HoandNuno Nobre 4fe3db5a5f Update config/cmake/modules/MfemCmakeUtilities.cmake
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-28 14:53:15 -08:00
Andrew Ho 55bb710cba fixed wrong dir being marked as system 2026-02-27 14:02:24 -08:00
dylan-copeland 939932bc68 Merge branch 'master' of github.com:mfem/mfem into bugfix-nurbs-orientation 2026-02-27 13:59:37 -08:00
Tzanio Kolev d3ae34710c Merge pull request #5195 from mfem/flip-index
Index sign functions
2026-02-27 13:32:22 -08:00
Vladimir Z Tomov 9cdb604796 Update of PA.ALF after remap.
Fixed lex ordering of maps_nodes.
Improved the PA AdaptLim kernels.
Cleaned debug code.
2026-02-27 13:10:31 -08:00
Dylan Copeland 9b652996b2 Remove an unnecessary assertion. 2026-02-27 11:48:01 -08:00
Dylan Copeland 5cfbbe5fad Bug fix for NURBS orientation, when the same knotvector index is used on all edges of some patches. 2026-02-27 11:24:36 -08:00
Andrew HoandNuno Nobre 7ad6939454 Update config/cmake/modules/MfemCmakeUtilities.cmake
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2026-02-27 07:06:16 -08:00
Wouter Tonnon 89ad250940 Merge branch 'master' into assemble-face-integrator-fix 2026-02-27 14:44:45 +01:00
60cc94e5a1 Update to use PetscCtxRt from (3,25,0), and cleanup duplicate code
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
Co-authored-by: Satish Balay <balay@mcs.anl.gov>
2026-02-26 11:57:27 -06:00
Satish Balay 9122ac1839 update KSPMonitorFn usage for < (3,24,0) 2026-02-26 11:57:22 -06:00
Satish Balay 864186117d update PetscCtxDestroyFn usage for < (3,23,0) 2026-02-26 11:56:15 -06:00
Ketan Mittal 35de169fd0 Merge branch 'master' into plasma-dir-dev 2026-02-26 09:42:51 -08:00
Tzanio Kolev 71909cd5e3 Merge pull request #5203 from mfem/bugfix/unit-test-cout
Changing std::cout to mfem::out in unit test
2026-02-26 05:50:05 -08:00
Vladimir Z Tomov c55e3fa7d2 debug wip 2026-02-25 18:22:42 -08:00
Alex Lindsay a5a3169064 Don't do arithmetic with enums
Else with gcc 13.3 with `-std=c++20` I get warnings
2026-02-25 13:35:10 -07:00
Hugh Carson d5dec97d23 Fix memory leak 2026-02-25 11:46:21 -05:00
Hugh Carson 2d401bcb74 Add positive-weight simplex quadrature rules for orders 0-20
Triangle rules from Witherden & Vincent (2015), tet rules d=0-13
from Witherden & Vincent, tet rules d=14-20 from Chuluunbaatar et al.
(2022). All rules have strictly positive weights and interior points,
replacing the legacy rules which use negative weights at several
orders and fall back to Grundmann-Moller (negative weights, high
point counts) for tets at d>=9.
2026-02-24 20:21:49 -05:00
Vladimir Z Tomov 7b47ee4cf5 Merge branch 'master' into findpts-surface 2026-02-24 15:27:19 -08:00
Andrew Ho 0a3184ab31 MFEM_EXPORT_GPU_CONFIG should export CPU config.mk when set to off 2026-02-24 11:56:38 -08:00
Wouter Tonnon 4f383f4b19 added missing face orientation 2026-02-24 20:27:14 +01:00
Dylan Copeland afded067a7 Replace absdof. 2026-02-24 11:17:54 -08:00
Mark L. Stowell a438e09caf Merge branch 'master' into plasma-dir-dev 2026-02-24 11:02:04 -08:00
chapman39 7f35ecb8f5 eliminate usage of modulus to avoid llvm backend bug 2026-02-24 10:50:52 -08:00
Alex Tyler Chapman ea03a86df2 Merge branch 'master' into bugfix/chapan39/use-mfem-abort-kernel-in-device 2026-02-24 10:32:30 -08:00
chapman39 6ef7a9e6fb 80 chars/ line 2026-02-24 10:32:18 -08:00
John Camier 629e93afd9 Merge branch 'master' into debug-mem-silent 2026-02-24 09:20:59 -08:00
Tzanio Kolev 3e277808a9 Merge pull request #5210 from mfem/hughcars/array-move-assignment-bugfix
Fix Array non-owning move assignment
2026-02-24 08:41:48 -08:00
Tzanio Kolev 864fb1ce9e Merge pull request #5217 from mfem/quadrature-function-fix
Fix a bug in `QuadratureFunction::GetValues` that returns a `DenseMatrix` view
2026-02-24 08:40:53 -08:00
Tzanio Kolev ac7415cc69 Merge branch 'master' into bugfix/unit-test-cout 2026-02-24 08:38:11 -08:00
Veselin Dobrev 38030d4395 Merge pull request #5222 from mfem/bugfix/chapman39/rm-return-from-omp
Remove return from openmp section
2026-02-24 08:18:53 -08:00
Alex Tyler Chapman 2c96dc6a1f Merge branch 'master' into bugfix/chapman39/rm-return-from-omp 2026-02-23 11:53:58 -08:00
Alex Tyler Chapman db7dd30d32 Merge branch 'master' into bugfix/chapan39/use-mfem-abort-kernel-in-device 2026-02-23 09:33:23 -08:00
John Camier ecbc7bf8c2 Merge branch 'master' into bugfix/unit-test-cout 2026-02-23 08:00:52 -08:00
John Camier 5b5a21edac Merge branch 'master' into debug-mem-silent 2026-02-22 12:28:39 -08:00
Tzanio Kolev 1f84ba036e Merge pull request #5179 from izaid/pfes
Fixed NULL pointer segfault in GetSurfaceFittingErrors
2026-02-21 12:52:33 -08:00
Tzanio Kolev 76d0312309 Merge branch 'master' into hughcars/array-move-assignment-bugfix 2026-02-21 12:52:11 -08:00
Tzanio Kolev c7ed339260 Merge pull request #5228 from mfem/mesh-3d-part-fix
Improve 3d mesh partitions
2026-02-21 12:43:49 -08:00
Tzanio Kolev f1b3a33fb2 Merge pull request #5234 from mfem/few-small-fixes
A few small fixes
2026-02-21 12:43:02 -08:00
rzhangbq 9e261aeb36 format 2026-02-20 18:06:33 -05:00
rzhangbq 3fe3c00c72 format 2026-02-20 18:05:51 -05:00
rzhangbq 1d925e5b7b format 2026-02-20 16:24:17 -05:00
Rushan ZhangandJan Nikl e779a5d47e Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-20 15:15:03 -05:00
Nuno Nobre 7cd35f97f7 Add ProjectDiscCoefficient based on max attr for scalar coeffs 2026-02-20 10:26:42 +00:00
rzhangbq f69b6204df preconstruct RHS 2026-02-19 23:34:57 -05:00
chapman39 a1fe3a19b1 dfem integrate: use mfem abort kernel in device code 2026-02-19 17:45:11 -08:00
Nuno Nobre 8baa46babd Allow evals of GridFunctionCoefficient on submeshes 2026-02-19 22:19:36 +00:00
Nuno Nobre 494fc00d34 Change IsParSubMesh to take Mesh ptr instead 2026-02-19 20:35:52 +00:00
Nuno Nobre 4dd3fcf811 Fix Mesh::GetFaceElementType for 1d meshes 2026-02-19 20:35:52 +00:00
Nuno Nobre 33d7cd11a2 Add missing setters for IntegrationPoint 2026-02-19 20:35:52 +00:00
Nuno Nobre fbd80e7493 Allow custom IntegrationRule for DomainLFGradIntegrator 2026-02-19 20:35:52 +00:00
rzhangbq a4fb0daa8e Update comments 2026-02-19 12:29:16 -05:00
rzhangbq 0b36f2adaa limit to 80 2026-02-19 12:27:52 -05:00
Rushan ZhangandJan Nikl 0288a5f146 Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:26:14 -05:00
Rushan ZhangandJan Nikl a1efd7a514 Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:25:51 -05:00
Rushan ZhangandJan Nikl e0c69fb83d Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:25:34 -05:00
Rushan ZhangandJan Nikl 43e88dd04f Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:25:16 -05:00
rzhangbq 946d4dde84 update comment 2026-02-19 12:24:44 -05:00
rzhangbq e890e9e6a5 update 2026-02-19 12:23:57 -05:00
Rushan ZhangandJan Nikl 7930c675ea Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:23:21 -05:00
Rushan ZhangandJan Nikl 298b14c82d Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:23:08 -05:00
Rushan ZhangandJan Nikl abb68a80e6 Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:22:47 -05:00
rzhangbq aec0b75047 change -oci default 2026-02-19 12:22:18 -05:00
rzhangbq f4e7c56119 change domain length var 2026-02-19 12:20:30 -05:00
rzhangbq eb70410a54 change ic for sample case 2026-02-19 12:18:46 -05:00
rzhangbq 9bccf40eb2 time out every timestep 2026-02-19 12:11:35 -05:00
Rushan ZhangandJan Nikl 3cb7465ab7 Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-19 12:02:51 -05:00
Satish Balay 213ccd7a4e rework PetscContainerSetCtxDestroy() usage for < (3,23,0) 2026-02-18 15:51:42 -06:00
Veselin Dobrev e3dedbbd5b Increase a relative tolerance due to failures on some machines 2026-02-18 10:27:44 -08:00
Will Pazner 8e78471fdf Add comment about the shape of FaceNbrData 2026-02-18 09:13:16 -08:00
Will Pazner c0f8501950 Add unit test for parallel L2 face restriction with vdim > 1 2026-02-18 09:13:03 -08:00
Will Pazner c31510289f Fix bug in ParL2FaceRestriction with vdim > 1
The layout of the FaceNbrData vector was not handled properly
2026-02-17 21:07:17 -08:00
John Camier 261f3805b8 Merge branch 'master' into debug-mem-silent 2026-02-17 10:59:40 -08:00
John Camier b386b2d6b6 Merge branch 'master' into bugfix/chapman39/rm-return-from-omp 2026-02-17 08:28:34 -08:00
Tzanio Kolev 6e98055eb7 Merge branch 'master' into quadrature-function-fix 2026-02-17 08:28:12 -08:00
Tzanio Kolev 2b14134496 Merge branch 'master' into najlkin/fix-cmplx-assign 2026-02-17 08:27:01 -08:00
Tzanio Kolev 5abd44f212 Merge pull request #5227 from mfem/add-constexpr
Add if constexpr to Kernel Specializations
2026-02-17 08:25:11 -08:00
Vladimir Z Tomov 75be9250a9 fixed mesh-optimizer.cpp 2026-02-16 15:46:56 -08:00
Veselin Dobrev 75012728db Formatting: re-wrap comment to 80 chars/line. 2026-02-14 13:10:30 -08:00
Mittal, Ketan 211470966c remove undeclared function definition 2026-02-13 13:00:13 -08:00
rzhangbq 9b2bc9e57a update comment 2026-02-13 14:41:50 -05:00
rzhangbq 76d2f8fea9 changed verify input 2026-02-13 14:40:33 -05:00
rzhangbq 63f746b8dc changed some verify 2026-02-13 14:38:27 -05:00
rzhangbq 18d64b8b93 change abort to verify 2026-02-13 14:33:49 -05:00
rzhangbq a740225601 update comment 2026-02-13 14:25:28 -05:00
rzhangbq 0d5fc47a73 reduce para to pass 2026-02-13 14:22:43 -05:00
Mittal, Ketan be887d05a4 capture some const for lambda 2026-02-13 10:44:06 -08:00
Dylan Copeland 1f094244f8 Typo 2026-02-13 10:37:58 -08:00
Dylan Copeland cc16ddadbf Merge branch 'master' of github.com:mfem/mfem into flip-index 2026-02-13 10:35:32 -08:00
Veselin Dobrev 8acd5cd3a2 Fix a size bug in thread-safe mode in H1_TriangleElement::CalcHessian.
Fix use-after-delete bug in nurbs_ex10p.cpp.

Use relative tolerance in the "Collocated Derivative Kernels" unit test
to resolve failures in some setups with the original absolute tolerance.
2026-02-13 10:29:22 -08:00
chapman39 1f5bc1c3d8 wording 2026-02-13 10:03:00 -08:00
chapman39 3588d47ec1 Merge remote-tracking branch 'origin/master' into bugfix/chapman39/rm-return-from-omp 2026-02-13 09:59:31 -08:00
Andrew Ho e7f5996bdf fix warnings with some compilers 2026-02-13 09:28:16 -08:00
Tzanio Kolev 560ad1b5a3 Merge pull request #5157 from mfem/plbound-extremum
Estimate function minimum/maximum using recursion + piecewise linear bounds
2026-02-13 07:08:57 -08:00
Mittal, Ketan df386413a9 refactor host-device data movement 2026-02-12 15:30:01 -08:00
Mittal, Ketan fa7fbdf36b remove unusued variable to track newton iterations 2026-02-11 18:12:33 -08:00
Mittal, Ketan 30f1ad7c2c Merge branch 'findpts-surface' of https://github.com/mfem/mfem into findpts-surface 2026-02-11 18:05:21 -08:00
Mittal, Ketan 11debd6bf8 merge with master and resolve conflicts 2026-02-11 18:05:08 -08:00
Alex Tyler Chapman f0fe5b0ec0 Merge branch 'master' into bugfix/chapman39/rm-return-from-omp 2026-02-11 15:14:48 -08:00
chapman39 a92983051a style 2026-02-11 15:14:23 -08:00
Vladimir Z Tomov 44a783993e Working PA for 2D adaptive limiting. But there's still some diff with FA. 2026-02-11 13:43:48 -08:00
rzhangbq 89974e87b6 split funcs 2026-02-11 16:37:15 -05:00
rzhangbq ec071ad4ab refact 2026-02-11 16:04:51 -05:00
rzhangbq 22c873f097 refact 2026-02-11 16:04:33 -05:00
rzhangbq e57ffb8128 drop the flag neutralizing_const_computed and check if precomputed_neutralizing_lf is set (not null). 2026-02-11 16:04:00 -05:00
Mittal, Ketan 2d7c578033 new line before FindPointsGSLIB constructor, and set default redist interval to 5 2026-02-11 12:59:04 -08:00
rzhangbq b503939955 get rid of HyperParVec Pointer 2026-02-11 15:51:15 -05:00
rzhangbq 8a4a826248 remove redundant 2026-02-11 15:45:18 -05:00
rzhangbq 8011c106ae resolve line width 2026-02-11 15:44:21 -05:00
rzhangbq 11d0d6a7be update desc of rdi 2026-02-11 15:26:42 -05:00
rzhangbq 2cc4bd7285 abort 2026-02-11 15:25:38 -05:00
rzhangbq 7ff38189fb remove redundant codes 2026-02-11 15:04:42 -05:00
rzhangbq dc243c6f7c remove misputted comments 2026-02-11 15:02:55 -05:00
rzhangbq b3508002e1 delete redundant pointpos 2026-02-11 15:01:03 -05:00
rzhangbq 06177ea337 update comment 2026-02-11 14:58:07 -05:00
Stowell, Mark L. 794a5fbfc2 Adding miniapps/plasma subdirectory to build system 2026-02-11 11:56:06 -08:00
Stowell, Mark L. 746a62f017 Adding plasma miniapp directory 2026-02-11 11:53:01 -08:00
rzhangbq 526d86489a move reduce global ke to method 2026-02-11 14:52:23 -05:00
Rushan ZhangandJan Nikl e8872fa31f Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-11 14:44:34 -05:00
Rushan ZhangandJan Nikl d547dfc6bf Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-11 14:44:26 -05:00
Rushan ZhangandJan Nikl b68a35d611 Update miniapps/pic/electrostatic-pic.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2026-02-11 14:44:08 -05:00
rzhangbq d2e381183e remove t_init 2026-02-11 14:43:11 -05:00
rzhangbq d4c37a7c1b make E_gf a static 2026-02-11 14:40:07 -05:00
rzhangbq dee64c36e5 add options to not output csv 2026-02-11 14:34:47 -05:00
Will Pazner 37843b050c Add one more 'if constexpr' 2026-02-11 11:33:51 -08:00
rzhangbq 846147efc0 frequency -> interval 2026-02-11 14:29:39 -05:00
rzhangbq b621c9c4a2 rename size (num_ranks) 2026-02-11 14:19:05 -05:00
camierjs 436714f5ef Revert unused included header 2026-02-11 11:18:02 -08:00
rzhangbq 5b1295c955 rename fec and fes 2026-02-11 14:17:47 -05:00
rzhangbq 43609b5c35 restyle comments 2026-02-11 14:16:49 -05:00
camierjs 1ceef4f786 Add few missing Kernel() constexpr 2026-02-11 11:11:35 -08:00
rzhangbq 64b7fbdeb2 move vis to main() 2026-02-11 14:11:23 -05:00
rzhangbq 44ed485cf1 rename fes 2026-02-11 13:58:20 -05:00
rzhangbq 90d1ed5ae3 remove empty lines 2026-02-11 13:54:59 -05:00
rzhangbq d7614eeb7e adding a few comments 2026-02-11 13:52:37 -05:00
Vladimir Z Tomov 6bb6745c0e improve 3d partitions 2026-02-10 16:23:38 -08:00
Mittal, Ketan 60f47c287d CHANGELOG 2026-02-10 14:28:17 -08:00
Mittal, Ketan c441299f2b newline in gslib header, and some other minor change 2026-02-10 13:58:11 -08:00
Ketan Mittal 75526f58cc Merge branch 'master' into particles-pic-dev-pr 2026-02-10 13:10:10 -08:00
Mittal, Ketan 164ee942c8 Merge branch 'plbound-extremum' of https://github.com/mfem/mfem into plbound-extremum 2026-02-10 10:59:44 -08:00
Mittal, Ketan 16c4fbdd29 reviewer comments 2026-02-10 10:59:33 -08:00
Ketan Mittal e28093274b Merge branch 'master' into plbound-extremum 2026-02-09 19:06:18 -08:00
Andrew Ho 87dd19e6c0 more constexprs 2026-02-09 17:35:17 -08:00
Andrew Ho 678f53c306 use constexpr to prevent unintended kernel Pinstantiations 2026-02-09 17:17:26 -08:00
yuyangdai a3e229ef82 Make SetMatrix methods private and remove Init method
- SetMatrix methods are called inside SetOperator
- Init method has been removed as it is unnecessary
2026-02-09 16:40:39 +08:00
John Camier 00b6dcdd37 Merge branch 'master' into cudss-dev 2026-02-07 13:21:22 -08:00
John Camier d9913262df Merge branch 'master' into hypremat 2026-02-07 13:21:01 -08:00
Tzanio Kolev 1861b80627 Merge pull request #5132 from farscape-project/superlu
Add toggle for SuperLU_DIST device offload
2026-02-07 13:14:15 -08:00
Tzanio Kolev a8c0c856f6 Merge pull request #5190 from mfem/launch-bounds
[GPU] Launch bounds
2026-02-07 13:14:00 -08:00
Gabriele Bozzola 9e4d9799dc Improve error message for gmsh versions != 2.2
I am a new user of [palace](https://github.com/awslabs/palace). As I was
trying to set a simple mesh up (with gmsh), I kept getting indexing
errors that I could not decipher. I eventually
[learned](https://mfem.org/mesh-formats/) that supported version for
gmsh meshes is 2.2.

This commit catches this and adds an informative error.
2026-02-06 16:12:27 -08:00
chapman39 32fb4bf244 remove return from openmp section 2026-02-05 14:56:40 -08:00
Tzanio Kolev 19f444489f Merge branch 'master' into cudss-dev 2026-02-05 11:03:38 -08:00
camierjs dbb86c87c2 Revert last MFEM_VERIFY 2026-02-04 16:23:07 -08:00
camierjs e866ede0c4 Re-enable GPU Wrap2D runtime verifications and revert EAMassAssemble static qualifiers 2026-02-04 16:05:57 -08:00
Jan Nikl ac4e558164 Minor unification of docstrings. 2026-02-04 15:21:02 -08:00
Jan Nikl 691cd8a687 Generalized RT normal projection. 2026-02-04 15:20:03 -08:00
Jan Nikl cdc327a511 Removed unused code. 2026-02-04 13:10:45 -08:00
rzhangbq 422eb8710f change zero-stepping logic 2026-02-02 18:15:35 -05:00
rzhangbq 42c47e9225 apply doxygen style 2026-02-02 16:47:19 -05:00
Veselin Dobrev 2d021685de Fix a bug in QuadratureFunction::GetValues when getting a reference to
the data for one mesh element as a DenseMatrix.

Added new methods:
- Array<T>::MakeRef(Memory<T> &base, int offset, int size_)
- DenseMatrix::MakeRef(Memory<real_t> &base, int offset, int h, int w)
2026-02-02 08:42:58 -08:00
camierjs c498d9c759 Merge branch 'master' into launch-bounds 2026-02-01 20:41:26 -08:00
camierjs 2896b2fb02 Forcing EAMassAssemble internal linkage to avoid ODR issues with gcc + nvcc 2026-02-01 20:41:12 -08:00
rzhangbq f3dc010bda update cmake case 2026-02-01 18:08:24 -05:00
Tzanio Kolev 8efb9483bc Merge pull request #5183 from mfem/fix-5180
Improve HYPRE solver documentation
2026-02-01 11:57:26 -08:00
camierjs 5507772f37 HIP defaults to MFEM_HIP_BLOCKS 2026-01-31 12:52:17 -08:00
rzhangbq fa34b2dc63 update case 2026-01-30 23:28:51 -05:00
rzhangbq 23b4cc62e9 update -nx ny nz of 3d3v case 2026-01-30 19:37:30 -08:00
rzhangbq 08c332c1b0 add 3d3v case 2026-01-30 22:34:25 -05:00
rzhangbq b2ad517e03 rename files 2026-01-30 22:04:27 -05:00
rzhangbq 812a907abe add support for 3D 2026-01-30 21:51:54 -05:00
rzhangbq 3c73c50b29 format 2026-01-30 21:41:37 -05:00
Hugh Carson d6ea262498 Address PR feedback
- Hoist src deletion
- Remove unneeded explicit cast given src is named
2026-01-30 14:01:59 -05:00
Tzanio Kolev 1b20704b24 Merge branch 'master' into launch-bounds 2026-01-30 09:43:05 -08:00
Tzanio Kolev 55bafba69a Merge branch 'master' into fix-5180 2026-01-30 09:40:40 -08:00
Tzanio Kolev 3a29fda4cd Merge pull request #5209 from mfem/particlevec-findpts-interface
Remove FindPointsGSLIB::Interpolate overload
2026-01-30 09:40:24 -08:00
Andrew Ho 26e9057f02 revert change, updated comment to why libdl gets special treatment 2026-01-30 07:27:04 -08:00
Dylan Copeland 6b147fd9ff Merge branch 'master' of github.com:mfem/mfem into flip-index 2026-01-29 15:45:52 -08:00
Andrew Ho c53bce08f1 missed a few periods 2026-01-29 12:52:26 -08:00
Andrew Ho afadd435f5 review comments 2026-01-29 12:50:04 -08:00
Andrew Ho 36d7629b26 residual vector requires hypre >= 2.15.0 2026-01-29 12:41:53 -08:00
Andrew Ho 8f32e6620b updated changelog, removed unused declaration 2026-01-29 08:15:34 -08:00
Jan Nikl 0d2e8f93e6 Fixed vis of the initial exact solution. 2026-01-28 18:14:44 -08:00
Tzanio Kolev 766760659d Merge branch 'master' into fix-5180 2026-01-28 16:39:11 -08:00
Jan Nikl 16dfa11f27 Minor docstring correction. 2026-01-28 15:45:20 -08:00
Jan Nikl c7774e3c1c Fixed complex grid function copy assignment. 2026-01-28 15:34:55 -08:00
John Camier 2c0419a9a8 Merge branch 'master' into launch-bounds 2026-01-28 11:36:48 -08:00
Hugh Carson a0981cb363 Refactor test for showing equivalency of copy and move assignment 2026-01-27 14:15:08 -05:00
Mittal, Ketan 1fd8301d38 Merge branch 'particles-pic-dev-pr' of https://github.com/mfem/mfem into particles-pic-dev-pr 2026-01-27 11:03:35 -08:00
Mittal, Ketan a013a150c1 include ordering argument in Interpolate 2026-01-27 11:03:27 -08:00
Hugh Carson 894de992da Move assignment of non-owned arrays must fallback to copy assignment 2026-01-27 14:01:23 -05:00
John Camier ca1bbaa7ed Merge branch 'master' into launch-bounds 2026-01-27 09:46:35 -08:00
rzhangbq 0c9d63ba7f update comment 2026-01-26 23:47:14 -05:00
rzhangbq a367bcc30d add pre-compute grad-interpolator 2026-01-26 23:44:42 -05:00
Mittal, Ketan 2ce98bec12 documentation 2026-01-26 20:22:43 -08:00
Mittal, Ketan 3db24b1b40 remove interpolate overload with ParticleVector 2026-01-26 20:19:55 -08:00
Mittal, Ketan d1db3325f2 FindPointsGSLIB documentation for constructor 2026-01-26 19:35:24 -08:00
Mittal, Ketan 0a8b4ad9af use updated FindPointsGSLIB interface 2026-01-26 19:30:28 -08:00
rzhangbq 2283ea838a should not init vis-socket in the beginning 2026-01-26 20:15:05 -05:00
Mittal, Ketan dcc3ba856e Merge branch 'master' of https://github.com/mfem/mfem into particles-pic-dev-pr 2026-01-26 09:32:52 -08:00
rzhangbq 6a4d7db35b remove func call at particle step 2026-01-26 01:01:02 -05:00
rzhangbq e1567e2729 simplify particle step 2026-01-25 23:37:24 -05:00
rzhangbq 2ede430196 bind field solver to FESpace instead 2026-01-25 22:52:59 -05:00
rzhangbq 1e7b7403ff split total energy val 2026-01-25 20:37:23 -05:00
rzhangbq e33690db45 Change to use GradientInterpolator 2026-01-25 18:55:54 -05:00
Paul Hilscher cece1b642b Merge branch 'master' into mfem-v13-mesh-reader-fix-issue-4625 2026-01-26 08:31:46 +09:00
rzhangbq daac9192cc use stopwatch instead 2026-01-24 16:00:29 -05:00
rzhangbq 4699d9c9e1 get rid of fmod usage 2026-01-24 15:24:27 -05:00
rzhangbq 24abcaee7a Update descriptions of simulation paras 2026-01-24 15:16:04 -05:00
rzhangbq 14d59df037 rename class names 2026-01-24 15:12:57 -05:00
rzhangbq 5d23e37b83 remove SIZE 2026-01-24 15:08:25 -05:00
Tzanio Kolev 942249395b Merge branch 'master' into debug-mem-silent 2026-01-24 11:40:50 -08:00
rzhangbq 7f5b68dfbd remove redundant findpoints 2026-01-23 17:28:53 -05:00
Stowell, Mark L. a3f6d5b971 Changing std::cout to mfem::out 2026-01-22 15:37:51 -08:00
daiyuyang 10c637837c Add dependency check for CUDSS in MFEMConfig.cmake.in 2026-01-22 16:56:17 +08:00
daiyuyangandAndrew Ho 665ba30f65 Update linalg/cudss.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:38:05 +08:00
daiyuyangandAndrew Ho d6de2c1a1d Update examples/ex1p.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:35:32 +08:00
daiyuyangandAndrew Ho c4b4cfdc2f Update examples/ex1p.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:35:18 +08:00
daiyuyangandAndrew Ho 3435475ae0 Update examples/ex1.cpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:34:32 +08:00
daiyuyangandAndrew Ho e43b58fa02 Update config/cmake/modules/FindCUDSS.cmake
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-01-22 09:32:27 +08:00
Rushan Zhang ac0454f07f Merge branch 'master' into particles-pic-dev-pr 2026-01-21 15:00:08 -05:00
Andrew Ho 194f3d8140 suggestions from Veselin 2026-01-21 11:45:20 -08:00
rzhangbq 9e727d568c move func definition all to the bottom 2026-01-21 14:40:44 -05:00
rzhangbq 7fd9af27a5 move up comments 2026-01-21 14:40:27 -05:00
John Camier db10fd292a Merge branch 'master' into launch-bounds 2026-01-21 11:37:08 -08:00
rzhangbq 77646c87dd remove redundant comments 2026-01-21 14:28:22 -05:00
rzhangbq 8531a43aac get rid of ctx.L_x in member funcs 2026-01-21 14:27:09 -05:00
rzhangbq 2b7f4ca792 add descriptions 2026-01-21 14:24:16 -05:00
rzhangbq 8e41393e14 Remove RemoveLostParticles (we use periodic boundary, particles should never move out of the computation space) 2026-01-21 14:17:14 -05:00
rzhangbq 452531e22f avoid use of ctx. out of main() 2026-01-21 14:10:12 -05:00
Ketan Mittal 5546250963 Merge branch 'master' into plbound-extremum 2026-01-21 11:00:02 -08:00
rzhangbq 6b6e5bf4b8 not hardcoding visport 2026-01-21 13:51:26 -05:00
rzhangbq 274bd5b670 now we can safely remove redundant FindPoints 2026-01-21 13:50:13 -05:00
rzhangbq b8f3571ba1 remove old Init particle declaration 2026-01-21 13:49:43 -05:00
rzhangbq 7f8e9680a6 add member func declaration 2026-01-21 13:49:08 -05:00
rzhangbq 2bebdf7595 make particle init as pic member func, so find particle is called upon particle creation 2026-01-21 13:48:46 -05:00
rzhangbq 759dacf996 using new findpoints 2026-01-21 13:47:38 -05:00
rzhangbq 2e76b94e17 add findparticles func 2026-01-21 13:45:48 -05:00
Mark L. Stowell 3f9b44a9cd Merge branch 'master' into phys-range-dim 2026-01-21 10:38:56 -08:00
rzhangbq 128b7a092b add back FindPoints before Interpolate 2026-01-21 13:07:31 -05:00
rzhangbq 491c558a57 change testcase to do -rdf 2 2026-01-21 13:06:59 -05:00
John Camier 27c8412439 Merge branch 'master' into cudss-dev 2026-01-21 08:23:31 -08:00
rzhangbq 45bf80a62e format 2026-01-20 23:38:39 -05:00
rzhangbq fdc885ecd2 remove redundant FindPoints 2026-01-20 23:38:15 -05:00
rzhangbq e9b4630d58 changing -np def to total #particle 2026-01-20 23:03:21 -05:00
rzhangbq 5d8442c21c make sure there is only one finder 2026-01-20 22:56:53 -05:00
John Camier 5b065ad7f2 Merge branch 'master' into debug-mem-silent 2026-01-20 17:59:07 -08:00
Andrew Ho 8c2ffb9d26 added Set/GetUseTwoNorm 2026-01-20 13:25:08 -08:00
John Camier 2737feaa2a Merge branch 'master' into launch-bounds 2026-01-19 19:09:42 -08:00
Nuno Nobre 4b9299188a Ensure hypre_CSRMatrixSetRownnz() allocs on host if ownership set to -1 2026-01-19 11:10:50 +00:00
daiyuyang 5aaef22cc0 Add conditional support for cuDSS solver in ex1.cpp 2026-01-19 15:16:31 +08:00
daiyuyang f9df36a6de Update parameter names in documentation for clarity in cudss.hpp 2026-01-19 14:21:19 +08:00
daiyuyang b3a75a9295 Add cuDSS solver option in ex1.cpp 2026-01-19 13:42:00 +08:00
daiyuyang b8c90d24f1 Make MPI optional and add OpenMP support in CuDSSSolver
- Make the MPI optional: the CuDSSSolver now does not need MPI necessary
- Add the OpenMP supports by cudssSetThreadingLayer() API
2026-01-19 13:40:56 +08:00
John Camier 3dcba10659 Merge branch 'master' into launch-bounds 2026-01-17 07:58:32 -08:00
Andrew Ho 4c6be0bc6f default p=2 to match hypre 2026-01-16 14:21:14 -08:00
Ketan Mittal 7f4d7b8f4e Merge branch 'master' into plbound-extremum 2026-01-16 12:54:56 -08:00
Jan Nikl 47c9ad2e34 Fixed visulization in ex22p. 2026-01-16 11:20:13 -08:00
rzhangbq b31b0e04bd change suggested nx and ny s.t. Debye length is resolved 2026-01-15 23:41:26 -05:00
rzhangbq 838206e6a9 fix vis bug 2026-01-15 23:35:06 -05:00
rzhangbq c681a74f87 rm redundant continue 2026-01-15 21:26:38 -05:00
Mittal, Ketan 6529372830 reviewer comments 2026-01-15 13:10:47 -08:00
Rushan ZhangandKetan Mittal b45138e6d7 Use common::VisualizeField instead of my own vis
Co-authored-by: Ketan Mittal <ketan.mittal@gmail.com>
2026-01-15 11:46:24 -05:00
rzhangbq f692d94d08 make finder a member obj 2026-01-15 11:44:45 -05:00
Rushan ZhangandKetan Mittal 6a0e1a7a89 change ip set
Co-authored-by: Ketan Mittal <ketan.mittal@gmail.com>
2026-01-15 11:05:02 -05:00
Mittal, Ketan d66d799387 minor 2026-01-14 15:25:11 -08:00
camierjs ca43ab0c61 Fix SmemPAVectorDiffusionApply2D 2026-01-14 15:01:28 -08:00
camierjs 1c60d5946b Fix vector diffusion bounds 2026-01-14 14:57:07 -08:00
camierjs c9768e34bc Add missing specializations and vector bounds 2026-01-14 14:34:54 -08:00
Mittal, Ketan 51f205b273 reviewer comments 2026-01-14 14:14:31 -08:00
Dylan Copeland 0161ad9d92 Minor simplifications. 2026-01-14 12:22:47 -08:00
Dylan Copeland 3589479481 Merge branch 'master' of github.com:mfem/mfem into flip-index 2026-01-14 12:10:30 -08:00
Dylan Copeland e9acfeccda Using new sign functions. 2026-01-14 12:10:07 -08:00
Mittal, Ketan ec8cd31f32 build for make and cmake 2026-01-14 11:32:52 -08:00
rzhangbq ec1ba64dac use for loop instead of hardcoding dims 2026-01-13 19:34:12 -05:00
rzhangbq a9590b900a remove outdated comments 2026-01-13 19:23:41 -05:00
rzhangbq e7f2083f0b regulating line lengths 2026-01-13 19:20:53 -05:00
rzhangbq 1b93160f5d remove redundant code for b 2026-01-13 19:11:37 -05:00
rzhangbq 74476c8f89 put the sample run in 1 line 2026-01-13 19:05:23 -05:00
rzhangbq 934958771c update csv 2026-01-13 19:03:42 -05:00
rzhangbq 0f827820f6 remove outdated B_gf comments 2026-01-13 19:00:58 -05:00
Rushan Zhang 709a8ca7e4 move .gitignore 2026-01-13 18:58:14 -05:00
Rushan Zhang fea9d2c4ce move gitignore 2026-01-13 18:57:46 -05:00
Ketan Mittal ad7cf12cd5 Merge branch 'master' into plbound-extremum 2026-01-13 12:24:55 -08:00
Ketan Mittal ea9686bdc0 Merge branch 'master' into particles-pic-dev-pr 2026-01-13 12:22:04 -08:00
camierjs 2375135f9a Add the launch bounds for DGMassCGIteration 2026-01-13 11:46:46 -08:00
camierjs a4800f42dd Merge branch 'master' into launch-bounds 2026-01-13 11:45:11 -08:00
John Camier caa973d6a0 Merge branch 'master' into cmake-config 2026-01-13 08:13:47 -08:00
camierjs 6df83bd190 Merge branch 'gpu-bounds' into launch-bounds 2026-01-12 16:11:43 -08:00
camierjs 10d975d0e7 Remove unused code 2026-01-12 15:54:22 -08:00
camierjs bae772c6f1 CEED bench options 2026-01-12 10:25:01 -08:00
Ketan Mittal abdb023ae3 Merge branch 'master' into plbound-extremum 2026-01-12 10:12:14 -08:00
Rushan Zhang 9f03879386 Merge branch 'master' into particles-pic-dev-pr 2026-01-12 12:54:04 -05:00
rzhangbq 43b26e7a5b update description 2026-01-12 12:49:15 -05:00
rzhangbq 3a1fb995a4 fix argument description 2026-01-12 12:33:23 -05:00
rzhangbq 87cb7170b2 change description 2026-01-12 12:31:33 -05:00
camierjs 79e0bc1ab2 CEED bench options 2026-01-12 09:08:29 -08:00
camierjs c21c6cb00b Add CEED benches order 7 2026-01-11 19:50:28 -08:00
camierjs 1bd9dd2e5d CUDA CEED benches 2026-01-11 18:10:28 -08:00
rzhangbq 3165f09e0d update description 2026-01-11 19:01:06 -05:00
rzhangbq 03910bbe86 set vscode formatting 2026-01-11 17:26:55 -05:00
rzhangbq 9532220814 add a high-level summary 2026-01-11 16:44:04 -05:00
rzhangbq f5decb7c9e rename 2026-01-11 16:38:17 -05:00
rzhangbq 4e00bfb158 run astyle 2026-01-11 16:36:02 -05:00
camierjs 31409edb7f Add CUDA MAX_THREADS_PER_BLOCK logic 2026-01-11 11:50:44 -08:00
camierjs ed66371ebd Cleanup CEED bench 2026-01-11 11:34:00 -08:00
camierjs 5727331966 Cleanup CEED bench 2026-01-11 11:24:06 -08:00
rzhangbq 7b79732a28 make weather reproduce an option 2026-01-10 19:13:37 -05:00
rzhangbq bdf8f6d21b remove double-interpolate E and change redis 2026-01-10 19:03:13 -05:00
rzhangbq cbc63ad344 remove redundant 2026-01-10 18:54:59 -05:00
rzhangbq 844b655c76 add chrono 2026-01-10 17:46:03 -05:00
rzhangbq db6c8f5a9a change sample run 2026-01-10 17:30:36 -05:00
rzhangbq 06331492e5 update discription 2026-01-10 17:28:10 -05:00
rzhangbq dabb5652fe change para name 2026-01-10 16:54:57 -05:00
rzhangbq 4947faca83 update test case 2026-01-10 16:50:40 -05:00
rzhangbq 9d1cb51acc format and update banner 2026-01-10 16:46:16 -05:00
rzhangbq 1ff1f5777f update readme and update pic banner display 2026-01-10 16:39:59 -05:00
rzhangbq 7bc13bf237 finish migration 2026-01-10 16:32:36 -05:00
camierjs 40039a6897 Cleanup CEED benchmarks 2026-01-10 10:58:50 -08:00
camierjs a05d4e1852 Bounds for HIP 2026-01-09 14:49:40 -08:00
camierjs 29c2442e47 tests/benchmarks/bench_ceed 2026-01-09 12:06:35 -08:00
izaid 5cd3ec521b applied astyle 2026-01-08 23:19:16 +00:00
rzhangbq f65a0f093b remove B-stepping 2026-01-08 14:46:49 -05:00
rzhangbq e7058f6aca remove traj vis 2026-01-08 14:43:31 -05:00
rzhangbq 785afe66cd first commit 2026-01-08 14:42:27 -05:00
Mittal, Ketan d1a9c6e62d format miniapp output 2026-01-08 10:53:49 -08:00
Mittal, Ketan 8f0b57138b Merge branch 'master' of https://github.com/mfem/mfem into plbound-extremum 2026-01-08 10:37:40 -08:00
Mittal, Ketan 3167a1c98b remove default value from tol in pgridfunc.hpp 2026-01-08 10:31:19 -08:00
Andrew Ho a024fb10bc more initial solver configurations 2026-01-07 17:17:19 -08:00
Andrew Ho 531e8d7ad9 Added a way to get the residual vector r and ||r||_p 2026-01-07 17:02:25 -08:00
Andrew Ho 9008d5a050 Improved hypre's documentation and added getter methods (when possible) for solver parameters 2026-01-07 15:36:54 -08:00
Tzanio Kolev 9488637956 Merge branch 'master' into pfes 2026-01-07 14:20:49 -08:00
Ketan Mittal e0b2ba5e54 Merge branch 'master' into findpts-surface 2026-01-06 12:54:33 -08:00
Mittal, Ketan f429737c12 change 0.0 to 0_r 2026-01-06 12:22:27 -08:00
Andrew Ho ad40704e20 Merge branch 'master' into cmake-config 2026-01-06 11:55:40 -08:00
Mittal, Ketan 04fd683e9c doxygen fix 2026-01-06 11:38:42 -08:00
Mittal, Ketan 4b9f46a6b0 Merge branch 'plbound-extremum' of https://github.com/mfem/mfem into plbound-extremum 2026-01-06 11:30:04 -08:00
Mittal, Ketan 793a5b6d60 minor 2026-01-06 11:29:47 -08:00
Mittal, Ketan 4e6e9a13b6 Merge branch 'master' of https://github.com/mfem/mfem into plbound-extremum 2026-01-06 11:15:53 -08:00
Dylan Copeland 6f280d81b5 Introducing new functions for flipping index signs. 2026-01-05 18:46:25 -08:00
izaid 90353c437e Fixed bug in GetSurfaceFittingErrors 2026-01-01 23:09:09 +01:00
John Camier fbb50af208 Merge branch 'master' into debug-mem-silent 2025-12-30 12:50:33 -08:00
Vladimir Z Tomov ee859d044d cleanup 2025-12-22 18:16:52 -08:00
Vladimir Z Tomov cf8d1ddd10 working pa computation for Mult 2025-12-22 18:01:27 -08:00
Vladimir Z Tomov 987f1636aa energy adapt lim 2d 2025-12-17 10:12:40 -08:00
Vladimir Z Tomov 15faf0d225 Adaptive limiting - normalization, assemblePA, wip. 2025-12-17 09:58:28 -08:00
daiyuyang a8eba594d2 Update linalg/cudss.hpp and linalg/cudss.cpp; Revert general/device.cpp
- Move ` CUDA_REAL_T` to the cpp file.
- Remove `CuDSSHandle` singleton and replace it with a static cudssHandle_t variable.
- Delete `CuDSSHandle::Init()` from ex1p
2025-12-17 09:52:36 +08:00
Ketan Mittal 1783050f9a Merge branch 'master' into findpts-surface 2025-12-16 12:57:09 -08:00
Andrew Ho d3470c07c9 Merge branch 'master' into cmake-config 2025-12-16 12:05:27 -08:00
Ketan Mittal 5b917af59b Merge branch 'master' into plbound-extremum 2025-12-15 12:59:44 -08:00
yuyangdai 52291cadbf Revert code docs; add constructor comment; update the .gitignore 2025-12-15 13:09:45 +08:00
yuyangdai 1865b430b0 Revert makefile and example/CMakeLists.txt; delete examples/cudss 2025-12-15 11:01:42 +08:00
yuyangdai d8734b4b18 Update the examples/ex1p.cpp
- Add the global cudss handle before using the cudss solver.
2025-12-15 10:27:47 +08:00
yuyangdai eae1fa217b Add CuDSSHandle singleton and update CuDSSSolver
- Remove unused variable `myid`.
- Move `MFEM_CUDSS_CHECK` and `mfem_cudss_error` into linalg/cudss.cpp.
- Disable move copy constructor and move assignment.
- Introduce `CuDSSHandle` singleton to manage cuDSS handle lifetime.
- Rename `InitHandle` to `Init`
2025-12-15 10:27:22 +08:00
Mittal, Ketan f956c6b2de function for pargridfunction 2025-12-14 16:33:38 -08:00
Mittal, Ketan 62dbc570b2 add functions to compute min/max over all elements 2025-12-14 15:57:05 -08:00
Mittal, Ketan c221f5a29d initial commit 2025-12-13 15:20:43 -08:00
gengyan.zgy 7dc8b0d9fe Fix CuDSSSolver::ArrayMult() for single RHS 2025-12-11 14:32:20 +08:00
yuyangdai 3f236b406f Update examples/ex1p.cpp
- Add cudss solver into the ex1p.cpp
2025-12-11 13:55:16 +08:00
yuyangdai 47dcdd8159 Update config/cmake/modules/FindCUDSS.cmake and examples/cudss/CMakeLists.txt
- Add a newline at end of file
2025-12-11 13:53:49 +08:00
yuyangdai d93d36d6d3 Update linalg/cudss.cpp and linalg/cudss.hpp
- Define a cuDSS error check macro, MFEM_CUDSS_CHECK(x);
- Rename the method from InitRhsSol to SetNumRHS;
- Utilize CuMemAlloc/ CuMemcpyDtoD instead of cudaMalloc/cudaMemcpy;
- Update MPI_Comm usage.
2025-12-11 13:52:57 +08:00
Paul Hilscher af834012d0 Merge branch 'master' into mfem-v13-mesh-reader-fix-issue-4625 2025-12-10 06:42:34 +09:00
yuyangdai 6bcad07a5c Update the n_global variable name for global number of rows 2025-12-05 13:50:36 +08:00
yuyangdai d56493c2c5 Refactored enum classes MatType and MatViewType, updated variable names, and update some comments. 2025-12-05 11:01:29 +08:00
Will Pazner 42f2594430 Fix small issues 2025-12-04 14:00:33 -08:00
Will Pazner 5ba3e4de97 Fix sign comparison issues 2025-12-04 13:59:00 -08:00
Will Pazner 02938c9cce Small restructuring 2025-12-04 13:23:30 -08:00
Will Pazner 65313cd7d3 Edits to Doxygen comments 2025-12-04 13:23:30 -08:00
Will Pazner 5246f9dbc3 One more unification in Gmsh reader 2025-12-04 13:23:30 -08:00
Will Pazner 4d352bf726 Small optimizations and improvements for Gmsh mesh reader
- Don't need to finalize the topology twice.
- Use unordered_map instead of map for better performance.
    - Requires introducing hasher for pairs. A more general solution
      is provided in PR #4974. Once that is merged, the PairHasher
      introduced here can be removed.
- Simplify interface since 'curved' and 'read_gf' are not needed.
- Read the version number as string instead of floating point number.
2025-12-04 13:23:30 -08:00
Will Pazner eed3bc067f Combine common features in Gmsh 2.2 and 4.1 2025-12-04 13:23:30 -08:00
Will Pazner 7bd256e17c Refactor Gmsh 2.2 reader 2025-12-04 13:23:30 -08:00
Will Pazner 7dbad4da3e Add Gmsh v4 reader 2025-12-04 13:23:30 -08:00
Will Pazner fb85c34ca4 Simplify Gmsh mesh reader 2025-12-04 13:23:30 -08:00
Will Pazner 6401ca5847 Remove gmsh.hpp header 2025-12-04 13:23:30 -08:00
Will Pazner 496e240837 Move Mesh::ReadGmshMesh to separate file 2025-12-04 13:23:30 -08:00
Ketan Mittal 016ebe62cc Merge branch 'master' into findpts-surface 2025-12-03 11:32:09 -08:00
Ketan Mittal 28ab39cf96 Merge branch 'master' into findpts-surface 2025-12-02 09:41:11 -08:00
Nuno Nobre f724cf348a Add toggle for SuperLU_DIST device offload 2025-12-02 14:23:05 +00:00
daiyuyang ec67fe536e Merge branch 'master' into cudss-dev 2025-12-02 14:04:17 +08:00
Andrew Ho 06a15cb7a9 missed one old unsetting of shared_link_flag 2025-12-01 17:02:05 -08:00
Andrew Ho d19ff6c676 Merge branch 'master' into cmake-config 2025-12-01 12:33:24 -08:00
daiyuyang eff793d6fa Merge branch 'master' into cudss-dev 2025-11-28 13:53:59 +08:00
Andrew Ho d85fbc6504 review suggestions 2025-11-25 14:35:38 -08:00
Andrew Ho 29346a87b6 Merge branch 'master' into cmake-config 2025-11-25 14:31:21 -05:00
yuyangdai e5e7280a17 Ignore the solution files of cudss sample run 2025-11-24 15:18:53 +08:00
yuyangdai 773dc57712 Using the make style to format the class CuDSSSolver and example/cudss/ex1p codes 2025-11-24 15:16:48 +08:00
yuyangdai 98a043b195 Update the doc files for CuDSSSolver 2025-11-24 15:15:22 +08:00
yuyangdai c2a3c83099 Update the makefiles for CuDSSSolver 2025-11-24 15:14:51 +08:00
daiyuyang 3ea90aff27 Add cudss solver 2025-11-24 15:10:45 +08:00
gengyan.zgy 23a362d6ad Update cmake files for cuDSS 2025-11-24 15:10:30 +08:00
Mittal, Ketan 9b90a7980b minor 2025-11-21 15:36:26 -08:00
Mittal, Ketan cf2c43b5c7 clean up documentation 2025-11-21 15:35:36 -08:00
Mittal, Ketan 2cb6a6e899 minor 2025-11-21 15:25:56 -08:00
Mittal, Ketan 3ec6292520 document map 2025-11-21 15:24:14 -08:00
Mittal, Ketan 7011d623d3 fix shadow declaration 2025-11-21 15:09:21 -08:00
Mittal, Ketan d4326eddd3 add unit test for the grid maps 2025-11-21 14:47:19 -08:00
Mittal, Ketan 39dcdb18e2 remove some stuff from testing 2025-11-21 14:00:30 -08:00
Mittal, Ketan fb0abff1c2 clean up pfindpts 2025-11-21 12:59:55 -08:00
Mittal, Ketan 8a615b8742 make style 2025-11-21 12:56:10 -08:00
Mittal, Ketan f24d9d8c0d rename some methods 2025-11-21 12:55:51 -08:00
Mittal, Ketan 85cf7b41d5 Merge branch 'master' of https://github.com/mfem/mfem into findpts-surface 2025-11-21 12:36:56 -08:00
Mittal, Ketan dff07dd1e3 Merge branch 'findpts-surface' of https://github.com/mfem/mfem into findpts-surface 2025-11-21 12:36:48 -08:00
Mittal, Ketan f17c25caf0 documentation and minor refactor to reuse gridrange etc methods 2025-11-21 12:36:35 -08:00
Ketan Mittal 929c7baf16 Merge branch 'master' into findpts-surface 2025-11-13 13:07:38 -08:00
Ketan Mittal dfa845a91c Merge branch 'master' into findpts-surface 2025-11-12 14:51:51 -08:00
Mittal, Ketan 7fe9733e4d fix minor bug and add new file to mesh_headers 2025-11-11 12:11:33 -08:00
Mittal, Ketan 1b3c326784 document new class and add ifdef mpi guards 2025-11-11 11:59:48 -08:00
Mittal, Ketan 0fd42364ca fix lmap_nd usage 2025-11-11 10:59:50 -08:00
Mittal, Ketan 7e2c9641c2 merge with master and resolve conflicts 2025-11-11 09:38:10 -08:00
Mittal, Ketan a0d18d4d52 move some definitions to before they are used 2025-11-11 09:20:21 -08:00
Mittal, Ketan 7adea0556c get rid of some macros 2025-11-11 09:15:40 -08:00
Mittal, Ketan 1e8efef66b copyright 2025-11-10 13:13:41 -08:00
Mittal, Ketan f14747eead make style 2025-11-10 12:20:41 -08:00
Mittal, Ketan d1151c09a3 add release version checks 2025-11-10 12:20:32 -08:00
Mittal, Ketan a9501ed65f clean up, make style, and unify interpolate local since it depends on r-dim only 2025-11-09 15:46:48 -08:00
Mittal, Ketan d16deb42f5 Merge branch 'findpts-surface' of https://github.com/mfem/mfem into findpts-surface 2025-11-06 13:25:12 -08:00
Mittal, Ketan b8f88a6560 interpolation kernels 2025-11-06 13:24:56 -08:00
Ketan Mittal 1653781a9d minor 2025-11-03 13:16:07 -08:00
Mittal, Ketan fcff34045e 3D edges 2025-11-03 09:39:58 -08:00
Mittal, Ketan ae675a05ef clean up 2D edge meshes 2025-11-02 20:36:56 -08:00
Andrew Ho 3464f7a004 Merge branch 'master' into cmake-config 2025-10-28 11:08:38 -07:00
John Camier ce8cd01cfd Merge branch 'master' into debug-mem-silent 2025-10-28 10:20:51 -07:00
Ketan Mittal 7f370e8193 fix bug in nodal coordinate memory assignment for 3D surface 2025-10-27 16:09:59 -07:00
Mittal, Ketan 870732a5aa minor 2025-10-27 13:55:07 -07:00
Mittal, Ketan 2bf4de6db4 initial commit 2025-10-27 10:58:50 -07:00
John Camier a60baf8ce6 Merge branch 'master' into debug-mem-silent 2025-10-25 11:28:07 -07:00
Andrew Ho 7de48e47ad Merge branch 'master' into cmake-config 2025-10-24 09:24:50 -07:00
Andrew Ho 70814c640b fixes for hip 2025-10-20 14:01:50 -07:00
Andrew Ho e9d3ae80f7 remove debug printout 2025-10-20 13:30:20 -07:00
Andrew Ho c8efc23c12 seems to be building external laghos now 2025-10-20 13:26:04 -07:00
Andrew Ho f26eb33252 Merge remote-tracking branch 'base/cmake-gpu' into cmake-config 2025-10-20 10:25:58 -07:00
John Camier 96bba18449 Merge branch 'master' into debug-mem-silent 2025-10-20 09:36:28 -07:00
Andrew Ho 05e622f837 improving config.mk file generated by cmake to work with hip/cuda
Still need to export compiler flags
2025-10-20 08:33:17 -07:00
Tzanio Kolev de3f769f49 Merge branch 'master' into dc-ofstream-fix-minor 2025-10-16 06:51:38 -07:00
Mark L. Stowell e9f84b033f Merge branch 'master' into phys-range-dim 2025-10-15 06:49:33 -07:00
John Camier f9cce3ab62 Merge branch 'master' into debug-mem-silent 2025-09-29 09:25:14 -07:00
John Camier 292700bb52 Merge branch 'master' into debug-mem-silent 2025-09-25 15:58:22 -07:00
John Camier fd2f0df34f Merge branch 'master' into debug-mem-silent 2025-09-21 20:13:41 -07:00
John Camier 4ac41a6427 Merge branch 'master' into debug-mem-silent 2025-09-21 10:27:05 -07:00
Andrew Ho ed862050b2 Merge branch 'master' into warn-gridfunc 2025-09-17 10:51:29 -07:00
Tzanio Kolev 3c6c1eb634 Merge branch 'master' into najlkin/extrd-1d-vec 2025-09-17 03:30:23 -07:00
Jan Nikl 22851a9463 Merge branch 'master' into najlkin/extrd-1d-vec 2025-09-16 15:30:43 -07:00
Jan Nikl 38df8156b9 Added documentation and checks to the extrusion classes. 2025-09-16 01:46:22 -07:00
Tzanio Kolev 58826d64c9 Merge branch 'master' into debug-mem-silent 2025-08-31 15:32:36 -07:00
Victor DeCaria a126203ccd implement suggested change 2025-08-21 15:08:57 -06:00
victor-decaria-nnl 6656a7ef72 Merge branch 'master' into debug-mem-silent 2025-08-20 15:50:34 -04:00
Veselin Dobrev 542467fd6a Merge branch 'master' into globalvec_debug 2025-08-16 16:26:36 -07:00
thartland 5986542e3d VERIFY instead of ASSERT 2025-07-16 16:02:21 -07:00
Tucker Hartland 5163313285 style 2025-07-16 15:46:46 -07:00
thartland 2201f3354a adding a check to make sure that each process owns at least one entry of the HypreParVector prior to calling GlobalVector 2025-07-16 15:36:57 -07:00
Tzanio Kolev e60f43fff3 Merge branch 'master' into ex37 2025-07-01 15:02:18 -07:00
Mark L. Stowell 83fd119b95 Merge branch 'master' into phys-range-dim 2025-07-01 10:16:58 -07:00
John Camier 8804df317d Merge branch 'master' into debug-mem-silent 2025-07-01 09:51:06 -07:00
Noe Reyes 5e51751064 Merge branch 'master' into ex37 2025-06-27 19:10:28 -04:00
thatguynoe d87bc4d22c remove signum function 2025-06-27 19:09:47 -04:00
thatguynoe 29dd96acf3 use the Illinois method instead of bisection
Speeds up convergence for the Bregman projection.
2025-06-27 19:09:47 -04:00
John Camier 95408b0fae Merge branch 'master' into debug-mem-silent 2025-06-22 20:25:51 -07:00
John Camier 79819a5563 Merge branch 'master' into debug-mem-silent 2025-06-18 10:22:04 -07:00
Tzanio Kolev e30f5b9c96 Merge branch 'master' into ex37 2025-06-17 08:16:18 -07:00
Andrew Ho f5b03af9d6 Merge branch 'master' into warn-gridfunc 2025-06-16 12:20:24 -07:00
John Camier d57fc7c0d9 Merge branch 'master' into debug-mem-silent 2025-06-16 08:33:43 -07:00
John Camier 9fb590d79d Merge branch 'master' into debug-mem-silent 2025-06-13 08:48:05 -07:00
Victor DeCaria cb4ca9228f add option to disable protection with debug backend 2025-06-05 09:34:25 -06:00
Andrew Ho 80c7823ac7 Merge branch 'master' into warn-gridfunc 2025-06-03 11:26:09 -07:00
Andrew Ho a443f003bb Merge pull request #4851 from mfem/najlkin/warn-gridfunc
Vector dimension for bdr/face elements
2025-06-02 11:45:47 -07:00
thatguynoe f6979648e8 move boundary assembly into bilinear form assembly
Prevents the user from accidentally calling these methods in the wrong order.
2025-05-30 09:17:52 -07:00
thatguynoe 2a4decc635 use the bisection method instead of Newton 2025-05-29 12:33:49 -07:00
thatguynoe b9d19d3bb3 Merge branch 'master' into ex37 2025-05-29 11:30:21 -07:00
Brendan Keith d8da041edf Merge branch 'master' into ex37 2025-05-26 13:24:47 -04:00
Mark L. Stowell 4aecb86d71 Merge branch 'master' into phys-range-dim 2025-05-19 17:51:08 -07:00
Andrew Ho 1730b05078 Merge branch 'master' into warn-gridfunc 2025-05-12 14:04:42 -07:00
Stowell, Mark L. 776a4c1815 Updating unit tests 2025-05-11 14:40:05 -07:00
Stowell, Mark L. c870d7dc1c Using new MapType entries and implementing new GetPhys*Dim methods 2025-05-11 14:39:49 -07:00
Stowell, Mark L. 8519889074 Adding new MapType entries for R2D and R1D classes 2025-05-11 14:38:50 -07:00
Jan Nikl 8a522f5e7d Fixed submesh unit test. 2025-05-07 16:26:34 -07:00
Jan Nikl fcbd105b82 Fixed ParGridFunction projection checks. 2025-05-07 15:28:54 -07:00
Jan Nikl b82dcf1387 Fixed variable order vector element unit tests. 2025-05-07 15:16:36 -07:00
Jan Nikl d3471aef59 Cosmetic change in FiniteElementSpace::GetTypicalBE(). 2025-05-07 14:35:50 -07:00
Jan Nikl 822555df0b Fixed boundary projection checks in GridFunction. 2025-05-07 14:23:33 -07:00
Jan Nikl 4626d65ac1 Added *VectorDim shortcuts to GridFunction. 2025-05-07 13:37:35 -07:00
Jan Nikl 38a80ea0e4 Added methods for typical elements and vector dimension for bdr/face. 2025-05-07 13:28:46 -07:00
Jan Nikl 590f954d6f Replaced the special case for trace spaces in GetVectorDim() by a check. 2025-05-07 11:58:01 -07:00
Jan Nikl bc5fc2b0f3 Revert "fixed ProjectBdrCoefficient vdim verify check"
This reverts commit feecd75ff3.
2025-05-07 11:55:25 -07:00
Paul Hilscher 7994a3df8b fix unsigned signed comparison warning 2025-04-29 07:42:24 +09:00
Paul Hilscher 5bb0c458cd remove unused to address ci failure 2025-04-29 07:18:09 +09:00
Paul Hilscher c5b2f0945a update CMakeList to include correct unit test 2025-04-29 07:02:52 +09:00
Paul Hilscher 1b0425bfe9 add unit test for named mesh attributes 2025-04-29 06:58:39 +09:00
Tzanio Kolev ab52f334e2 Merge branch 'master' into mfem-v13-mesh-reader-fix-issue-4625 2025-04-26 12:25:30 -07:00
Paul Hilscher f8c494e59c fallback to tracking attributes as seek might not be available 2025-04-21 09:25:38 +09:00
Paul Hilscher f6d304864b fix parsing 2025-04-20 19:30:12 +09:00
Paul Hilscher 3593b4cd60 apply style 2025-04-20 09:17:18 +09:00
Paul Hilscher ef557b3fc1 add some more ws 2025-04-20 09:00:01 +09:00
Paul Hilscher 9a94a4b7b8 allow arbitrary white spaces 2025-04-20 08:59:04 +09:00
Paul Hilscher e18518d731 sort ex39 entry 2025-04-20 06:49:12 +09:00
Paul Hilscher e49bf21914 fix mfem v13 mesh format reader 2025-04-20 06:49:12 +09:00
Tzanio Kolev 0f78d8aa5c Merge branch 'master' into najlkin/extrd-1d-vec 2025-04-15 13:46:39 -07:00
Andrew Ho 3f98aa1cfb Merge branch 'master' into warn-gridfunc 2025-04-14 16:15:59 -07:00
Andrew Ho feecd75ff3 fixed ProjectBdrCoefficient vdim verify check 2025-04-11 22:39:33 -07:00
Andrew Ho 248bdcc149 revert change in GridFunction::GetGradient 2025-04-11 20:49:28 -07:00
Andrew HoandVeselin Dobrev e4e354834d Update fem/gridfunc.cpp
Co-authored-by: Veselin Dobrev <v-dobrev@users.noreply.github.com>
2025-04-11 20:47:15 -07:00
Andrew Ho d64a6d6255 Change GetVectorDim so it works for trace spaces.
Re-added boundary projection vector dim checks
2025-04-11 18:58:09 -07:00
Andrew Ho 510387a605 remove checks in Bdr projections
I think something more clever needs to be done here for projecting
onto trace elements
2025-04-11 17:23:44 -07:00
Andrew Ho e99b2a8410 Switched to VectorDim(), added checks for VectorCoefficient projection 2025-04-11 16:07:31 -07:00
Andrew Ho 6608111315 Added error checks for scalar coefficient projection onto a vector GridFunction 2025-04-11 14:39:06 -07:00
Jan Nikl b7253275fc Added extrusion of vector 1D grid functions. 2025-04-01 17:49:16 -07:00
thatguynoe 5808fc6966 use gradient descent step length in proj
See https://github.com/mfem/mfem/pull/4645#discussion_r1898110950.
2025-02-25 19:21:07 -05:00
thatguynoe b40bf6a64d Revert "add option to choose bisection method for roots"
This reverts commit 21778cd9335337a1419af482a7feaf6baac1057f.
2025-02-25 18:31:42 -05:00
thatguynoe 7a73e97922 Revert "rename newton arg for clarity"
This reverts commit d7681c26dc608bfab1ca4c891e22cbc1260340c6.
2025-02-25 18:31:42 -05:00
thatguynoe 3a97122e34 rename newton arg for clarity 2025-02-25 18:31:42 -05:00
thatguynoe 2f89a16314 update sample runs, fix stability 2025-02-25 18:31:42 -05:00
thatguynoe 0cd8c2e273 delete bilinear form when cleaning 2025-02-25 18:31:42 -05:00
thatguynoe b4992673b2 fix ex37 serial
If running ex37 serial with MFEM parallel, a segfault would occur when attempting to run MPI_Allreduce. To fix this, we use the associated FiniteElementSpace and check for MFEM parallel.
2025-02-25 18:31:42 -05:00
thatguynoe 46dce17970 fix ex37 serial
If running ex37 serial with MFEM parallel, a segfault would occur when attempting to run MPI_Allreduce. To fix this, we use the associated FiniteElementSpace and check for MFEM parallel.
2025-02-25 18:31:42 -05:00
thatguynoe f080627cba move ParGridFunction declaration 2025-02-25 18:31:42 -05:00
thatguynoe 85fb20a1d1 set smaller itol for better convergence 2025-02-25 18:31:42 -05:00
thatguynoe 428d203eac add option to choose bisection method for roots 2025-02-25 18:31:42 -05:00
thatguynoe c10ca25f62 assemble boundary, bilinear form outside of solve 2025-02-25 18:31:42 -05:00
thatguynoe 2e8f6f9c28 assemble boundary, bilinear form outside of solve 2025-02-25 18:31:42 -05:00
thatguynoe 11e4c46f25 add growth rate arg for grad descent step length 2025-02-25 18:31:42 -05:00
thatguynoe ff8d8752c7 move proj function into header file 2025-02-25 18:31:42 -05:00
stefanhenneking e3cfc28718 add ofstream.is_open() checks 2025-02-18 23:17:55 -06:00
212 changed files with 17369 additions and 3904 deletions
+2 -2
View File
@@ -25,7 +25,7 @@ runs:
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
- uses: actions/cache@v5
if: ${{env.DEBUG == 'true'}}
id: debug
with:
@@ -82,7 +82,7 @@ runs:
run: find . -type f -name '*.o' -delete
shell: bash
- uses: actions/upload-artifact@v4
- uses: actions/upload-artifact@v7
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
+2 -2
View File
@@ -36,7 +36,7 @@ runs:
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
- uses: actions/cache@v5
if: ${{env.DEBUG == 'true' && inputs.cache-skip != 'true'}}
id: debug
with:
@@ -49,7 +49,7 @@ runs:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- uses: actions/download-artifact@v4
- uses: actions/download-artifact@v8
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
+5 -5
View File
@@ -23,7 +23,7 @@ inputs:
runs:
using: 'composite'
steps:
- uses: actions/cache/restore@v4 # Cache for LLVM libcxx
- uses: actions/cache/restore@v5 # Cache for LLVM libcxx
with:
path: ${{env.LLVM_DIR}}
fail-on-cache-miss: true
@@ -32,14 +32,14 @@ runs:
- uses: ./.github/actions/sanitize/mpi
if: ${{inputs.par == 'true'}}
- uses: actions/cache/restore@v4 # Cache for Hypre
- uses: actions/cache/restore@v5 # Cache for Hypre
if: ${{inputs.par == 'true'}}
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- uses: actions/cache/restore@v4 # Cache for Metis
- uses: actions/cache/restore@v5 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
@@ -51,13 +51,13 @@ runs:
run: ln -s -f ${{env.HYPRE_DIR}} hypre && ln -s -f ${{env.METIS_DIR}} metis-4.0
shell: bash
- uses: actions/cache/restore@v4 # Cache for LSAN suppression file
- uses: actions/cache/restore@v5 # Cache for LSAN suppression file
with:
path: ${{env.LSAN_DIR}}
fail-on-cache-miss: true
key: build-lsan-suppression-file
- uses: actions/checkout@v4 # Checkout the repository
- uses: actions/checkout@v6 # Checkout the repository
with:
path: mfem
# ref: ${{env.BRANCH}}
+42
View File
@@ -0,0 +1,42 @@
# MFEM Pull Request Review Agent Guide
## Purpose and scope
Review MFEM PRs for correctness, maintainability, performance, portability, test coverage, and MFEM consistency. Use the diff and PR context; reference source files, tests, and CI results when available. Follow `CONTRIBUTING.md`, especially Developer Guidelines, PR rules, checklist, and testing.
## Critical review pillars
- Correctness and numerical behavior
- API and user-facing impact
- Performance implications
- Maintainability and portability
## Review workflow
1. Read the PR description, linked issues, and intended behavior.
2. Inspect the diff before commenting.
3. Identify affected MFEM components, examples, tests, build or docs changes, and downstream APIs.
4. Analyze the code against the critical review pillars.
5. Compare the change against nearby code and MFEM patterns; flag unmotivated deviations.
6. Check whether tests and documentation were updated appropriately.
7. Review CI results and suggest actions.
8. Produce a structured review with prioritized findings.
9. Always limit conclusions to available evidence.
## MFEM-specific review checklist
- Component-aware scope: identify the touched subsystem (FEM, solvers, preconditioners, linear algebra, mesh, examples, miniapps, build, or docs) and assess its impact against the review pillars.
- Numerical and algorithmic behavior: assess issues in convergence, stability, tolerances, precision, iteration limits, and failure handling. If clear opportunities exist to improve the algorithmic approach, call them out with expected impact.
- API and user-facing impact: assess backward compatibility, user-visible behavior and default changes, migration impact, deprecations, and whether documentation clearly explains user-facing API changes.
- Data structure and memory semantics: assess ownership, lifetime, aliasing, container behavior, and device-host synchronization.
- Parallel and serial behavior: assess whether the change preserves equivalent semantics in serial and parallel modes where applicable; if logic is currently mode-specific, check whether extension to the other mode is straightforward (clear abstractions, no hard-wired assumptions), document constraints, and call out expected behavior differences explicitly.
- Backend and portability impact: assess likely cross-backend risks in CPU, CUDA, HIP, OCCA, RAJA, partial assembly, fallback paths, compiler compatibility, and platform assumptions.
- Build, dependency, and configuration impact: assess CMake or make changes, optional dependency behavior, and feature-flag interactions.
- Tests and docs alignment: check available regression or unit coverage evidence for changed behavior, and ensure docs are updated for new flags, APIs, options, or behavior changes.
- MFEM developer-guideline fit: keep code lean, simple, general, logically separated, and portable; suggest C++17 improvements when they clearly improve safety, clarity, or maintainability.
- New source files, examples, or miniapps: if a PR adds source/header files, verify they are properly wired into the relevant `makefile` and `CMakeLists.txt`, referenced in docs where applicable (including `doc/CodeDocumentation.dox`), and added to top-level `.gitignore` only when generated artifacts require it.
- Changelog: verify `CHANGELOG` is updated if the PR introduces significant new features or user-facing changes.
- MFEM conventions: use `real_t`; use `mfem::out`/`mfem::err` instead of `std::cout`/`std::cerr` in library code; flag large/binary files; if AI assistance is apparent but undisclosed, suggest following `CONTRIBUTING.md`.
- Edge cases: if the PR touches complex or error-prone areas, suggest additional tests for edge cases, failure modes, and parallel behavior.
## Commenting guidelines
- Keep comments concise, actionable, and grounded in the diff.
- Focus on correctness, behavior changes, and user impact over style nits.
- Be professional, concise, collaborative, technically precise, and avoid unsupported assumptions.
+1 -1
View File
@@ -43,7 +43,7 @@ jobs:
remove-docker-images: 'true'
- name: Checkout
uses: actions/checkout@v4
uses: actions/checkout@v6
# It's easier to reference named variables than indexes of the matrix
- name: Set Environment
+6 -5
View File
@@ -153,7 +153,7 @@ jobs:
# /home/runner/work/mfem/mfem/mfem
# Note: Done now to access "install-hypre" and "install-metis" actions.
- name: checkout mfem
uses: actions/checkout@v4
uses: actions/checkout@v6
with:
path: ${{ env.MFEM_TOP_DIR }}
# Fetch the complete history for codecov to access commits ID
@@ -225,7 +225,7 @@ jobs:
- name: cache hypre
id: hypre-cache
if: matrix.mpi == 'par'
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
@@ -255,7 +255,7 @@ jobs:
- name: cache metis
id: metis-cache
if: matrix.mpi == 'par' && matrix.os != 'windows-latest'
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
@@ -270,7 +270,7 @@ jobs:
- name: cache vcpkg (Windows)
id: vcpkg-cache
if: matrix.os == 'windows-latest'
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: vcpkg_cache
key: ${{ runner.os }}-${{ matrix.mpi }}-vcpkg-v1
@@ -295,7 +295,8 @@ jobs:
export HOMEBREW_NO_INSTALL_CLEANUP=1
brew update
brew install enzyme
ENZYME_LLVM=$(brew info enzyme | sed -n 's/^Required:.*\(llvm[^ ]*\).*/\1/p')
ENZYME_LLVM=$(brew info enzyme | sed -n 's/^Required.*:.*\(llvm[^ ]*\).*/\1/p')
echo "ENZYME_LLVM=$ENZYME_LLVM"
LLVM_PREFIX=$(brew --prefix $ENZYME_LLVM)
echo "LLVM_PREFIX=$LLVM_PREFIX" >> $GITHUB_ENV
echo "OMPI_CC=$LLVM_PREFIX/bin/clang" >> $GITHUB_ENV
+4 -4
View File
@@ -40,11 +40,11 @@ jobs:
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@v6
# Initializes the CodeQL tools for scanning.
- name: Initialize CodeQL
uses: github/codeql-action/init@v2
uses: github/codeql-action/init@v4
with:
languages: ${{ matrix.language }}
# If you wish to specify custom queries, you can do so here or in a config file.
@@ -57,7 +57,7 @@ jobs:
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
# If this step fails, then you should remove it and run the build manually (see below)
- name: Autobuild
uses: github/codeql-action/autobuild@v2
uses: github/codeql-action/autobuild@v4
# ️ Command-line programs to run using the OS shell.
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
@@ -70,4 +70,4 @@ jobs:
# ./location_of_script_within_repo/buildscript.sh
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@v2
uses: github/codeql-action/analyze@v4
+3 -3
View File
@@ -39,7 +39,7 @@ jobs:
steps:
- name: checkout MFEM
uses: actions/checkout@v4
uses: actions/checkout@v6
with:
path: mfem
@@ -50,7 +50,7 @@ jobs:
- name: Cache Hypre Install
id: hypre-cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
@@ -65,7 +65,7 @@ jobs:
- name: Cache Metis Install
id: metis-cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
+4 -4
View File
@@ -38,7 +38,7 @@ jobs:
github.event.pull_request.head.repo.full_name != github.repository)
steps:
- name: checkout mfem
uses: actions/checkout@v4
uses: actions/checkout@v6
- name: copyright check
id: copyright
@@ -93,7 +93,7 @@ jobs:
github.event.pull_request.head.repo.full_name != github.repository)
steps:
- name: checkout mfem
uses: actions/checkout@v4
uses: actions/checkout@v6
- name: get astyle
run: |
@@ -110,7 +110,7 @@ jobs:
github.event.pull_request.head.repo.full_name != github.repository)
steps:
- name: checkout mfem
uses: actions/checkout@v4
uses: actions/checkout@v6
- name: get doxygen and graphviz
run: |
@@ -135,7 +135,7 @@ jobs:
runs-on: ubuntu-latest
steps:
- name: checkout mfem
uses: actions/checkout@v4
uses: actions/checkout@v6
with:
fetch-depth: 0
+2 -2
View File
@@ -17,11 +17,11 @@ jobs:
runs-on: ubuntu-latest
name: 2.19.0
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
+2 -2
View File
@@ -27,13 +27,13 @@ jobs:
llvm_use_sanitizer: "Undefined"
name: ${{matrix.sanitizer}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{env.LLVM_DIR}}
key: build-libcxx-${{env.LLVM_VER}}-${{matrix.sanitizer}}
+2 -2
View File
@@ -17,11 +17,11 @@ jobs:
runs-on: ubuntu-latest
name: lsan.supp
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{env.LSAN_DIR}}
key: build-lsan-suppression-file
+2 -2
View File
@@ -17,11 +17,11 @@ jobs:
runs-on: ubuntu-latest
name: 4.0.3
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
uses: actions/cache@v5
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
+9 -9
View File
@@ -28,7 +28,7 @@ jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/mfem
with:
par: ${{inputs.par}}
@@ -40,7 +40,7 @@ jobs:
env:
ex: ${{inputs.par && 'ex1p' || 'ex1'}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
@@ -58,7 +58,7 @@ jobs:
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
@@ -82,7 +82,7 @@ jobs:
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
@@ -107,7 +107,7 @@ jobs:
run: ${{inputs.par && '-R "_cpu_np"' || ''}}
exclude: ${{inputs.par && '"unit_tests|debug"' || '"^unit_tests$|debug"'}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
@@ -131,7 +131,7 @@ jobs:
env:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
@@ -146,7 +146,7 @@ jobs:
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: find . -type f -name '*.o' -delete
- uses: actions/upload-artifact@v4
- uses: actions/upload-artifact@v7
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit/${{env.unit_tests}}
@@ -165,14 +165,14 @@ jobs:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
np: ${{inputs.par && '_np=2' || ''}}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- uses: actions/download-artifact@v4
- uses: actions/download-artifact@v8
if: ${{steps.restore.outputs.cache-hit != 'true'}}
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
+4
View File
@@ -443,6 +443,10 @@ miniapps/diag-smoothers/mg-abs-l1-jacobi
miniapps/contact/contact
miniapps/contact/ParaView
miniapps/plasma/pic/electrostatic-*
!miniapps/plasma/pic/electrostatic-*.cpp
miniapps/plasma/pic/*.csv
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
+5
View File
@@ -85,3 +85,8 @@ opt_par_gcc_10_pumi:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +pumi"
opt_par_gcc_10_gslib:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +gslib"
+5
View File
@@ -63,3 +63,8 @@ opt_mpi_cuda_hypre_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
opt_mpi_cuda_gcc_gslib:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda +gslib cuda_arch=90 ^hypre+cuda"
+2 -2
View File
@@ -32,9 +32,9 @@ mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
# run
if [[ "${MACHINE_NAME}" == "dane" ]]; then
salloc --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
srun --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
srun --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
else
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
exit 1
+51
View File
@@ -11,22 +11,56 @@
Version 4.9.1 (development)
===========================
- Policy for AI-assisted contribution added to CONTRIBUTING.md
Discretization improvements
---------------------------
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Extend FindPointsGSLIB to support surface meshes.
- Replaced legacy simplex quadrature rules with symmetric positive-weight
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
rules guarantee all-positive weights and interior quadrature points,
improving numerical stability. Higher orders fall back to Grundmann-Moller.
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
2015.
Tet rules (d=1-13): Witherden & Vincent (ibid).
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
2022.
- Improved the gridfunction projection routines. Projections work for Scalar,
Vector and VectorFE, also NURBS versions. Optionally different types of
projections can be selected, default behaviour has not changed.
- Added methods to estimate function extremum using piecewise linear bounds +
recursive subdivision.
Meshing improvements
--------------------
- Improved support for 1D NURBS meshes with variable order, including using
the patches construct for 1D NURBS meshes.
- Added the option to include material interfaces (faces separating elements
with different element attributes) as additional boundary elements, for
parallel visualization, e.g. with GLVis. This is supported by both the Print
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
New and updated examples and miniapps
-------------------------------------
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
capability.
Miscellaneous
-------------
- Fixed signed DOF handling in parallel grid-function reading (read constructor)
and saving via ParGridFunction::SaveAsOne(). Simplified the process of
applying the DOF signs by using the new method ApplyDofSigns() in class
ParFiniteElementSpace -- the method will return immediately if no sign flips
are needed.
Version 4.9, released on Dec 11, 2025
=====================================
@@ -111,6 +145,23 @@ Linear and nonlinear solvers
Filtering (AMGF), providing robust preconditioning for linear systems arising
in constrained optimization problems such as frictionless contact.
Added 'GetResiduals' and 'GetFinalAbsResidualNorm' to 'HyprePCG',
'HypreGMRES', and 'HypreFGMRES' to get 'r' and '|r|_p'. Note that the latter
computes '|r|_p' from 'r' instead of returning a cached value like the
relative 'GetFinalResidualNorm'. These require Hypre >= 2.15.0.
Changed the default solver parameters for 'HyprePCG' to 'tol=1e-6' and
'max_iter=1000'. This matches the default parameters in Hypre 3.0.
Added various helper functions for querying/modifying Hypre solvers:
'HypreSmoother::GetType', 'HypreSmoother::GetSOROptions',
'HypreSmoother::GetPolyOptions', 'HypreSmoother::GetWindowParameters',
'HypreSmoother::IsOperatorSymmetric', 'HyprePCG::GetTol',
'HyprePCG::GetAbsTol', 'HyprePCG::GetMaxIter', 'HyprePCG::SetUseTwoNorm',
'HypreGMRES::GetTol', 'HypreGMRES::GetAbsTol', 'HypreGMRES::GetMaxIter',
'HypreGMRES::GetKDim', 'HypreFGMRES::GetTol', 'HypreFGMRES::GetMaxIter',
'HypreFGMRES::GetKDim', and 'HypreBoomerAMG::GetMaxIter'.
GPU computing
-------------
- Added the 'gpu', 'raja-gpu', and 'ceed-gpu' backend aliases/shortcuts which
+15 -2
View File
@@ -433,6 +433,15 @@ if (MFEM_USE_STRUMPACK)
endif()
endif()
# cuDSS can only be enabled in CUDA
if (MFEM_USE_CUDSS)
if (MFEM_USE_CUDA)
find_package(CUDSS REQUIRED)
else()
message(FATAL_ERROR " *** cuDSS requires that CUDA be enabled.")
endif()
endif()
# GnuTLS
if (MFEM_USE_GNUTLS)
find_package(_GnuTLS REQUIRED)
@@ -631,7 +640,7 @@ find_package(Threads REQUIRED)
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CUDSS CALIPER CODIPACK
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
ALGOIM ENZYME CUDA::cudart)
@@ -652,6 +661,8 @@ foreach(TPL IN LISTS MFEM_TPLS)
endif()
endforeach(TPL)
# reverse to remove the first instance of entries in TPL_LIBRARIES
# so later duplicates are kept (for dependency ordering)
list(REVERSE TPL_LIBRARIES)
list(REMOVE_DUPLICATES TPL_LIBRARIES)
list(REVERSE TPL_LIBRARIES)
@@ -1015,5 +1026,7 @@ install(DIRECTORY ${CMAKE_CURRENT_BINARY_DIR}/data
# Create 'config.mk' from 'config.mk.in' for the build and install locations and
# define install rules for 'config.mk' and 'test.mk'
#-------------------------------------------------------------------------------
if (MFEM_USE_CUDA OR MFEM_USE_HIP)
option(MFEM_EXPORT_GPU_CONFIG "Export config.mk for GPU-enabled downstream packages" ON)
endif()
mfem_export_mk_files()
+73 -65
View File
@@ -3,12 +3,13 @@
</p>
<p align="center">
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml?query=branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
</p>
@@ -24,6 +25,14 @@ must be made under this license.
Note also that MFEM has a [Code of Conduct](CODE_OF_CONDUCT.md). By participating
in the MFEM community, you agree to abide by its rules.
## AI Policy
- Use of AI code generation in MFEM is allowed but must be disclosed, e.g. by
selecting the `AI-assisted` label on the PR.
- By submitting a PR, the author acknowledges that they have reviewed and
understand the changes they are proposing.
- PR authors are still responsible for correctness, licensing, and attribution
of all changes.
If you plan on contributing to MFEM, consider reviewing the
[issue tracker](https://github.com/mfem/mfem/issues) first to check if a thread
already exists for your desired feature or the bug you ran into. Use a pull
@@ -76,7 +85,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
follow the [MFEM PR Rules](#mfem-pr-rules).
- When your contribution is fully working and ready to be reviewed, add
the `ready-for-review` label.
- PRs are treated similarly to journal submission with an "editor" assigning two
- PRs are treated similarly to journal submission, with an "editor" assigning two
reviewers to evaluate the changes.
- The reviewers have 3 weeks to evaluate the PR and work with the author to
fix issues and implement improvements.
@@ -117,7 +126,7 @@ The MFEM source code has the following structure:
│ ├── petsc
│ ├── pumi
│ ├── sundials
| └── superlu
└── superlu
├── fem
│ ├── ceed
│ ├── dfem
@@ -129,10 +138,6 @@ The MFEM source code has the following structure:
│ ├── moonolith
│ ├── qinterp
│ └── tmop
│ | ├── assemble
│ | ├── metrics
│ | ├── mult
│ | └── tools
├── general
├── linalg
│ ├── batched
@@ -145,11 +150,10 @@ The MFEM source code has the following structure:
│ ├── common
│ ├── contact
│ ├── dfem
│ ├── diag-smoothers
│ ├── dpg
│ ├── electromagnetics
│ ├── fluids
│ │ ├── navier
│ │ └── schrodinger-flow
│ ├── gslib
│ ├── hdiv-linear-solver
│ ├── hooke
@@ -159,6 +163,7 @@ The MFEM source code has the following structure:
│ ├── nurbs
│ ├── parelag
│ ├── performance
│ ├── plasma
│ ├── shifted
│ ├── solvers
│ ├── spde
@@ -189,15 +194,15 @@ respectively.
- The main finite element classes are:
+ [`FiniteElement`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElementCollection.html)
+ [`FiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1FiniteElementSpace.html)
+ [`GridFunction`](https://docs.mfem.org/html/classmfem_1_1GridFunction.html)
+ [`BilinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
- The main linear algebra classes and sources are
+ [`Operator`](https://docs.mfem.org/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1Vector.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`DenseMatrix`](https://docs.mfem.org/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://docs.mfem.org/html/classmfem_1_1SparseMatrix.html)
+ Sparse [smoothers](https://docs.mfem.org/html/sparsesmoothers_8hpp.html) and linear [solvers](https://docs.mfem.org/html/solvers_8hpp.html)
@@ -209,8 +214,8 @@ shared geometric entities between different tasks. The parallel source files
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
- The main parallel classes are
+ [`ParMesh`](https://docs.mfem.org/html/solvers_8hpp.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParNCMesh.html)
+ [`ParFiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1ParFiniteElementSpace.html)
+ [`ParGridFunction`](https://docs.mfem.org/html/classmfem_1_1ParGridFunction.html)
+ [`ParBilinearForm`](https://docs.mfem.org/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://docs.mfem.org/html/classmfem_1_1ParLinearForm.html)
@@ -220,14 +225,14 @@ have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
#### GPU and general device support
GPU and multi-core CPU support is based on device kernels supporting different
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
backends (CUDA, HIP, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
device/host memory manager.
- The main device-relevant classes and sources are:
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html), [`hip.hpp`](https://docs.mfem.org/html/hip_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
#### Utilities, building and documentation
- The `general/` directory contains C++ classes that serve as utilities for
@@ -241,8 +246,8 @@ device/host memory manager.
- `examples` and `miniapps` respectively gather simple and more fully-featured
demonstrations of the usage on MFEM. They both rely on `data/` for the
collection of meshes.
- The `tests/` directory contains a unit test suite and will later contain more
tests that run example codes.
- The `tests/` directory contains a unit test suite, additional tests, and
benchmarks.
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
website.
@@ -276,8 +281,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
the top of https://github.com/mfem.
- Consider making your membership public by going to https://github.com/orgs/mfem/people
and clicking on the organization visibility drop box next to your name.
- Project discussions and announcements will be posted at
https://github.com/orgs/mfem/teams/everyone.
- Project discussions and announcements will be posted at https://github.com/orgs/mfem/discussions,
tagging the `@mfem/everyone` team when appropriate.
#### Structure
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
@@ -337,11 +342,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Well-designed simple code is frequently more general and powerful.
- Lean code base is easier to understand by new collaborators.
- New features should be added only if they are necessary or generally useful.
- Introduction of language constructions not currently used in MFEM should be
- Introduction of language constructs not currently used in MFEM should be
justified and generally avoided (to maintain portability to various systems
and compilers, including early access hardware).
- We prefer basic C++ and the C++03 standard, to keep the code readable by
a large audience and to make sure it compiles anywhere.
- We prefer basic C++. Use C++17 features judiciously, prioritizing readability,
consistency with existing MFEM code, and portability to different systems,
compilers and device backends.
- *Keep the code general and reasonably efficient*
- The main goal is fast prototyping for research and application development.
@@ -384,7 +390,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- When your branch is ready for other developers to review / comment on
the code, create a pull request towards `mfem:master`.
- Pull request typically have titles like:
- Pull requests typically have titles like:
`Description [new-feature-dev]`
@@ -405,12 +411,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
team will add reviewers as appropriate.
- List outstanding TODO items in the description, see PR #222 for an example.
- List outstanding TODO items in the description.
- When your contribution is fully working and ready to be reviewed, add
the `ready-for-review` label.
or request the `ready-for-review` label.
- PRs are treated similarly to journal submission with an "editor" assigning
- PRs are treated similarly to journal submission, with an "editor" assigning
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
the PR and work with the author to implement improvements and fix issues.
@@ -436,7 +442,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
checks in GitHub Actions enforce MFEM-specific rules which are explained in
the error messages and the `tests/scripts` directory.
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
- Also note that the tests `branch-history` and `repo-check` found in GitHub
Actions can be triggered automatically before each push using git hooks. See
the [git hooks README](config/githooks/README.md) for a detailed explanation.
@@ -493,15 +499,15 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
3. To ensure the quality of the PR by making sure that the code adheres to the [Developer Guidelines](#developer-guidelines), e.g. all methods, data members, and functions have documentation, including data ownership and lifetime, new examples/miniapps have a corresponding PR in mfem/web, major features have `CHANGELOG` entries, etc.
3. To seek help from the editors in case of difficulties.
4. To seek help from the editors in case of difficulties.
4. To complete the review in a timely manner: 3 weeks from assignment.
5. To complete the review in a timely manner: 3 weeks from assignment.
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
6. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
7. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
8. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
#### Responsibilities of Authors
@@ -527,30 +533,30 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Code builds.
- [ ] Code passes `make style`.
- [ ] Update `CHANGELOG`:
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Update `INSTALL`:
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Has a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] Update `.gitignore`:
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] New examples:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
- [ ] Add any files generated by it to the `clean` target.
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new examples directory (e.g. `examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
@@ -567,13 +573,13 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] New capability:
- [ ] All new public, protected, and private classes, methods, data members, and functions have full Doxygen-style documentation in source comments. Documentation should include descriptions of member data, function arguments and return values, template parameters, and prerequisites for calling new functions.
- [ ] Pointer arguments and return values must specify whether ownership is being transferred or lent with the call.
@@ -675,7 +681,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
- [ ] Update URL shortlinks:
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
- [ ] Add the new shortlinks to the MFEM package in `spack`.
- [ ] Update website in `mfem/web` repo:
- Update version and shortlinks in `src/index.md` and `src/download.md`.
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
@@ -727,22 +733,24 @@ commit or push, see the [README](config/githooks/README.md) in the `config/githo
directory.
### Linux and Mac smoke tests
### GitHub Actions smoke tests
We use GitHub Actions to drive the default tests on the `master` and `next`
branches. See the `.github/workflows` files and the logs at
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
Testing using GitHub Actions should be kept lightweight, as there is a time
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
GitHub Actions testing should be kept lightweight, as there is a time
constraint on jobs. The current workflows cover Linux, macOS, and Windows
configurations.
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
- Tests on the `next` branch are currently scheduled to run each night.
### Additional Windows smoke test
### Windows smoke test
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor` file and the
build logs at
We also use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor.yml` file
and the build logs at
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
CMake is used to generate the MSVC Project files and drive the build. A release
+31 -16
View File
@@ -38,14 +38,13 @@ the option MFEM_USE_METIS.
MFEM also includes support for devices such as GPUs, and programming models such
as CUDA, HIP, OCCA, OpenMP and RAJA.
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
a newer compiler, e.g. GCC version 4.9 or higher.
- Starting with version 4.9, MFEM requires a C++17 compiler.
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
https://developer.nvidia.com/cuda-toolkit
- HIP support requires an AMD GPU and an installation of the ROCm software stack
https://rocmdocs.amd.com
https://rocm.docs.amd.com
- OCCA support requires the OCCA library
https://libocca.org
@@ -83,9 +82,9 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
make parallel -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
CUDA build:
make cuda -j 4
@@ -115,14 +114,14 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
make -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
Parallel build with fetching of hypre and METIS:
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
make -j 4
@@ -134,7 +133,8 @@ CUDA build:
HIP build:
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 \
-DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
make -j 4
Example codes (serial/parallel, depending on the build):
@@ -269,6 +269,7 @@ Compilers:
CXX - C++ compiler, serial build
MPICXX - MPI C++ compiler, parallel build
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
HIP_CXX - The HIP compiler, e.g. 'hipcc'
Compiler options:
OPTIM_FLAGS - Options for optimized build
@@ -395,6 +396,11 @@ MFEM_USE_STRUMPACK = YES/NO
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_CUDSS = YES/NO
Enable MFEM functionality based on the cuDSS library. When using cuDSS, CUDA
support must be also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
When enabled, this option uses the CUDSS_* library options, see below.
MFEM_USE_GINKGO = YES/NO
Enable MFEM functionality based on the Ginkgo library, which provides
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
@@ -554,13 +560,13 @@ MFEM_USE_RAJA = YES/NO
MFEM_USE_OCCA = YES/NO
Enables support for the OCCA library in MFEM. OCCA is an open-source library
which aims to make it easy to program different types of devices (e.g. CPU,
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
GPU, FPGA) by providing a unified API for interacting with JIT-compiled
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_GSLIB = YES/NO
Enables MFEM functionality based on the GSLIB library, and specifically its
FindPoints component, which provides a robust algorithms to evaluate finite
FindPoints component, which provides robust algorithms to evaluate finite
element functions in a collection of points in physical space. When enabled,
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
@@ -719,9 +725,18 @@ The specific libraries and their options are:
Options: STRUMPACK_OPT, STRUMPACK_LIB.
Versions: STRUMPACK >= 3.0.0.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
C++ compiler that supports the C++-17 standard. For additional requirements
and dependencies of specific modules, see the Ginkgo webpage below.
- CUDSS (optional), used when MFEM_USE_CUDSS = YES. Note that CUDSS requires
CUDA 12.x toolkit and the cuDSS libraries. The supported communication backend
is OpenMPI 4.x (default), and OpenMPI 4.x or a later version must be pre-built.
The source files in the cuDSS tarball provide guidance for developing custom
MPI implementations.
URL: https://developer.nvidia.com/cudss
https://docs.nvidia.com/cuda/cudss/advanced_features.html#communication-layer-library-in-cudss
Options: CUDSS_OPT, CUDSS_LIB.
Versions: cuDSS >= 0.6.0.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Ginkgo may have additional
requirements and module-specific dependencies; see the webpage below.
URL: https://ginkgo-project.github.io
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or
Debug).
@@ -793,7 +808,7 @@ The specific libraries and their options are:
Options: CONDUIT_OPT, CONDUIT_LIB.
Versions: Conduit >= 0.3.1.
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
- ADIOS2 (optional), used when MFEM_USE_ADIOS2 = YES.
URL: https://adios2.readthedocs.io/
Versions: ADIOS >= 2.5.0.
@@ -869,7 +884,7 @@ The specific libraries and their options are:
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
Versions: RAJA >= 2022.10.3.
- Moonolith (optional), use when MFEM_USE_MOONOLITH = YES.
- Moonolith (optional), used when MFEM_USE_MOONOLITH = YES.
URL: https://bitbucket.org/zulianp/par_moonolith
Options: MOONOLITH_DIR
Versions: MOONOLITH >= 1.1.0.
@@ -957,7 +972,7 @@ CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
To use a specific generator use the "-G <generator>" option of cmake:
cmake <mfem-source-dir> -G "Xcode"
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
cmake <mfem-source-dir> -G "Visual Studio 17 2022"
cmake <mfem-source-dir> -G "MinGW Makefiles"
With CMake it is possible to build MFEM as a shared library using the standard
@@ -1202,7 +1217,7 @@ larger problems, there are two options:
Specific options for HIP
========================
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
ROCm install, as well as having `$ROCM_PATH/bin` in `PATH`.
Specific options for RAJA+HIP+MPI
=================================
+1
View File
@@ -28,6 +28,7 @@ license files. These software products and their licenses are as follows:
* AmgXWrapper (linalg/amgxsolver.{hpp,cpp}) -- MIT license
* Catch++ (tests/unit/catch.hpp) -- Boost 1.0 license
* Gecko (general/gecko.{cpp,hpp}) -- BSD 3-clause license
* gslib (fem/gslib.{cpp,hpp}, mesh/bb_grid_map.{cpp,hpp}) -- BSD 3-clause license
* Picojson (fem/picojson.h) -- Custom 2-clause license
* TinyXML2 (general/tinyxml2.{cpp,h}) -- zlib license
* Zstr (general/zstr.hpp) -- MIT license
+9
View File
@@ -35,6 +35,7 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_CUDSS @MFEM_USE_CUDSS@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
@@ -109,6 +110,14 @@ if (MFEM_USE_RAJA)
find_dependency(RAJA)
endif()
if (MFEM_USE_CUDSS)
find_dependency(cudss)
endif (MFEM_USE_CUDSS)
if (MFEM_USE_UMPIRE)
find_dependency(umpire)
endif()
if (NOT TARGET mfem)
include(${CMAKE_CURRENT_LIST_DIR}/MFEMTargets.cmake)
endif (NOT TARGET mfem)
+9
View File
@@ -108,6 +108,15 @@
// Enable MFEM functionality based on the STRUMPACK library.
#cmakedefine MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
#cmakedefine MFEM_USE_CUDSS
// CUDSS communication layer library path
#cmakedefine MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
#cmakedefine MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable functionality based on the Ginkgo library.
#cmakedefine MFEM_USE_GINKGO
+68
View File
@@ -0,0 +1,68 @@
if (NOT cudss_DIR AND CUDSS_DIR)
set(cudss_DIR ${CUDSS_DIR}/lib/cmake/cudss)
endif()
message(STATUS "Looking for CUDSS ...")
message(STATUS " in CUDSS_DIR = ${CUDSS_DIR}")
message(STATUS " cudss_DIR = ${cudss_DIR}")
find_package(cudss)
set(CUDSS_FOUND ${cudss_FOUND})
set(CUDSS_LIBRARIES "cudss")
if (CUDSS_FOUND)
message(STATUS
"Found CUDSS target: ${CUDSS_LIBRARIES} (version: ${cudss_VERSION})")
else()
set(msg STATUS)
if (CUDSS_FIND_REQUIRED)
set(msg FATAL_ERROR)
endif()
message(${msg}
"CUDSS not found. Please set CUDSS_DIR to the install prefix.")
endif()
if(CUDSS_FOUND AND TARGET cudss)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION)
if(NOT CUDSS_LIBRARY_LOCATION)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION_RELEASE)
endif()
if(CUDSS_LIBRARY_LOCATION)
get_filename_component(CUDSS_LIBRARY_DIR "${CUDSS_LIBRARY_LOCATION}" DIRECTORY)
else()
message(WARNING "Could not determine the location of the cuDSS library.")
endif()
else()
message(WARNING "cuDSS target not available; cannot determine library directory.")
endif()
# Set the full name of the cuDSS threading library if OpenMP is enabled.
# The threading layer library (libcudss_mtlayer_gomp.so) is located under the
# cuDSS library directory by default.
if (MFEM_USE_OPENMP)
find_file(
CUDSS_THREADING_LIB
NAMES libcudss_mtlayer_gomp.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_THREADING_LIB AND CUDSS_THREADING_LIB)
set(MFEM_CUDSS_THREADING_LIB "${CUDSS_THREADING_LIB}")
endif()
message(STATUS "CUDSS threading layer library: ${MFEM_CUDSS_THREADING_LIB}")
endif()
# Set the full name of the cuDSS communication library if MFEM use OpenMPI.
# The communication layer library (libcudss_commlayer_mpi.so) is located under the
# cuDSS library directory by default.
# The communication layer library is used pre-built communication layers for OpenMPI
# by default.
if (MFEM_USE_MPI)
find_file(
CUDSS_COMM_LIB
NAMES libcudss_commlayer_openmpi.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_COMM_LIB AND CUDSS_COMM_LIB)
set(MFEM_CUDSS_COMM_LIB "${CUDSS_COMM_LIB}")
endif()
message(STATUS "CUDSS communication layer library: ${MFEM_CUDSS_COMM_LIB}")
endif()
+3 -3
View File
@@ -14,12 +14,12 @@
# - UMPIRE_LIBRARIES
# - UMPIRE_INCLUDE_DIRS
if (NOT umpire_DIR AND UMPIRE_DIR)
set(umpire_DIR ${UMPIRE_DIR}/lib/cmake/umpire)
if (NOT umpire_ROOT AND UMPIRE_DIR)
set(umpire_ROOT ${UMPIRE_DIR})
endif()
message(STATUS "Looking for UMPIRE ...")
message(STATUS " in UMPIRE_DIR = ${UMPIRE_DIR}")
message(STATUS " umpire_DIR = ${umpire_DIR}")
message(STATUS " umpire_ROOT = ${umpire_ROOT}")
find_package(umpire CONFIG)
set(UMPIRE_FOUND ${umpire_FOUND})
set(UMPIRE_LIBRARIES "umpire")
+89 -17
View File
@@ -701,7 +701,6 @@ endfunction(mfem_find_library)
# Extract compile and link options needed by the given target.
#
function(mfem_get_target_options Target CompileOptsVar LinkOptsVar)
if (NOT TARGET ${Target})
return()
endif()
@@ -799,7 +798,12 @@ function(mfem_get_target_options Target CompileOptsVar LinkOptsVar)
# message(STATUS "Lib = ${Lib}")
# Filter-out generator expressions
if (NOT ("${Lib}" MATCHES "^\\$"))
list(APPEND LinkOpts "${Lib}")
if(NOT ("${Lib}" STREQUAL "dl"))
list(APPEND LinkOpts "${Lib}")
else()
# for some reason libdl doesn't include the "-l"
list(APPEND LinkOpts "-ldl")
endif()
endif()
else()
mfem_get_target_options(${Lib} COpts LOpts)
@@ -888,9 +892,18 @@ function(mfem_export_mk_files)
set(${var} NO)
endif()
endforeach()
# TODO: Add support for MFEM_USE_CUDA=YES
set(MFEM_CXX ${CMAKE_CXX_COMPILER})
set(MFEM_HOST_CXX ${MFEM_CXX})
if (MFEM_USE_CUDA AND MFEM_EXPORT_GPU_CONFIG)
set(MFEM_CXX ${CMAKE_CUDA_COMPILER})
if(MFEM_CUDA_COMPILER_IS_NVCC)
set(MFEM_HOST_CXX ${CMAKE_CUDA_HOST_COMPILER})
else()
set(MFEM_HOST_CXX ${CMAKE_CXX_COMPILER})
endif()
else()
# mfem doesn't use enable_language(HIP)
set(MFEM_CXX ${CMAKE_CXX_COMPILER})
set(MFEM_HOST_CXX ${CMAKE_CXX_COMPILER})
endif()
set(MFEM_CPPFLAGS "")
get_target_property(cxx_std mfem CXX_STANDARD)
# For now, we ignore the setting of the CXX_EXTENSIONS property. If this
@@ -900,6 +913,50 @@ function(mfem_export_mk_files)
string(STRIP
"${cxx_std_flag} ${CMAKE_CXX_FLAGS_${BUILD_TYPE}} ${CMAKE_CXX_FLAGS}"
MFEM_CXXFLAGS)
if(MFEM_EXPORT_GPU_CONFIG)
if (MFEM_USE_CUDA)
set(MFEM_CXXFLAGS "${MFEM_CXXFLAGS} ${CMAKE_CUDA_FLAGS}")
if (MFEM_CUDA_COMPILER_IS_NVCC)
set(MFEM_CXXFLAGS "-x=cu ${MFEM_CXXFLAGS} -ccbin ${CMAKE_CXX_COMPILER} --forward-unknown-to-host-compiler")
# The following intentionally hides CUDA deprecation warnings
foreach(ENTRY IN LISTS CUDAToolkit_INCLUDE_DIRS)
set(MFEM_CXXFLAGS "${MFEM_CXXFLAGS} -isystem ${ENTRY}")
endforeach()
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.18.0)
# architecture flags not part of CMAKE_CUDA_FLAGS
if ("all" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "native" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "all-major" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}")
set(MFEM_CXXFLAGS "${MFEM_CXXFLAGS} -arch=${CMAKE_CUDA_ARCHITECTURES}")
else()
foreach (ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(MFEM_CXXFLAGS
"${MFEM_CXXFLAGS} -gencode arch=compute_${ENTRY},code=sm_${ENTRY}")
endforeach()
endif()
endif()
else()
set(MFEM_CXXFLAGS "${MFEM_CXXFLAGS} -xcuda --cuda-path=${CUDAToolkit_LIBRARY_ROOT}")
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.18.0)
# architecture flags not part of CMAKE_CUDA_FLAGS
if ("all" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "native" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "all-major" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}")
# TODO: not supported
else()
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(MFEM_CXXFLAGS "-cuda-gpu-arch=sm_${ENTRY} ${MFEM_CXXFLAGS}")
endforeach()
endif()
endif()
endif()
elseif (MFEM_USE_HIP)
set(MFEM_CXXFLAGS "${MFEM_CXXFLAGS} -xhip")
foreach(ENTRY IN LISTS CMAKE_HIP_ARCHITECTURES)
set(MFEM_CXXFLAGS "--offload-arch=${ENTRY} ${MFEM_CXXFLAGS}")
endforeach()
endif()
endif()
set(MFEM_TPLFLAGS "")
foreach(dir ${TPL_INCLUDE_DIRS})
set(MFEM_TPLFLAGS "${MFEM_TPLFLAGS} -I${dir}")
@@ -930,6 +987,9 @@ function(mfem_export_mk_files)
set(MFEM_SHARED NO)
set(MFEM_STATIC YES)
endif()
if (MFEM_USE_CUDA)
set(MFEM_EXT_LIBS "${MFEM_EXT_LIBS} -lcudart")
endif()
set(MFEM_BUILD_TAG "${CMAKE_SYSTEM}")
set(MFEM_PREFIX "${CMAKE_INSTALL_PREFIX}")
# For the next 4 variables, these are the values for the build-tree version of
@@ -938,8 +998,15 @@ function(mfem_export_mk_files)
set(MFEM_LIB_DIR "${PROJECT_BINARY_DIR}")
set(MFEM_TEST_MK "${PROJECT_SOURCE_DIR}/config/test.mk")
set(MFEM_CONFIG_EXTRA "MFEM_BUILD_DIR ?= ${PROJECT_BINARY_DIR}")
# TODO: CUDA/HIP support:
set(MFEM_XLINKER "${CMAKE_CXX_LINKER_WRAPPER_FLAG}")
if (MFEM_USE_CUDA AND MFEM_EXPORT_GPU_CONFIG)
if (MFEM_CUDA_COMPILER_IS_NVCC)
set(MFEM_XLINKER "-Xlinker=")
else()
set(MFEM_XLINKER "${CMAKE_CUDA_LINKER_WRAPPER_FLAG}")
endif()
else()
set(MFEM_XLINKER "${CMAKE_CXX_LINKER_WRAPPER_FLAG}")
endif()
set(MFEM_MPIEXEC ${MPIEXEC})
if (NOT MFEM_MPIEXEC)
set(MFEM_MPIEXEC "mpirun")
@@ -987,16 +1054,21 @@ function(mfem_export_mk_files)
# handle interfaces (e.g., SCOREC::apf)
if ("${lib}" MATCHES "SCOREC::.*" OR "${lib}" MATCHES "Ginkgo::.*" OR "${lib}" MATCHES "ParMoonolith::.*")
elseif (TARGET "${lib}")
mfem_get_target_options(${lib} CompileOpts LinkOpts)
mfem_get_target_options(${lib} CompileOpts2 LinkOpts2)
# remove generator expressions
string(GENEX_STRIP "${CompileOpts2}" CompileOpts)
string(GENEX_STRIP "${LinkOpts2}" LinkOpts)
# Removing duplicates may lead to issues:
# list(REMOVE_DUPLICATES CompileOpts)
# list(REMOVE_DUPLICATES LinkOpts)
string(REPLACE ";" " " COpts "${CompileOpts}")
string(REPLACE ";" " " LOpts "${LinkOpts}")
# message(STATUS "${lib}[COpts]: '${COpts}'")
# message(STATUS "${lib}[LOpts]: '${LOpts}'")
set(MFEM_TPLFLAGS "${MFEM_TPLFLAGS} ${COpts}")
set(MFEM_EXT_LIBS "${MFEM_EXT_LIBS} ${LOpts}")
# message(WARNING "${lib}[LinkOpts]: ${LinkOpts}")
# message(WARNING "${lib}[CompileOpts]: ${CompileOpts}")
foreach(LOpt IN LISTS LinkOpts)
set(MFEM_EXT_LIBS "${MFEM_EXT_LIBS} ${LOpt}")
endforeach()
foreach(COpt IN LISTS CompileOpts)
set(MFEM_TPLFLAGS "${MFEM_TPLFLAGS} ${COpt}")
endforeach()
# message(FATAL_ERROR "***** interface lib found ... exiting *****")
# handle static and shared libs
elseif ("${suffix}" STREQUAL "${CMAKE_SHARED_LIBRARY_SUFFIX}")
@@ -1004,7 +1076,7 @@ function(mfem_export_mk_files)
get_filename_component(fullLibName ${lib} NAME_WE)
string(REGEX REPLACE "^lib" "" libname ${fullLibName})
set(MFEM_EXT_LIBS
"${MFEM_EXT_LIBS} ${shared_link_flag}${dir} -L${dir} -l${libname}")
"${MFEM_EXT_LIBS} ${shared_link_flag}${dir} -L${dir} -l${libname}")
else()
set(MFEM_EXT_LIBS "${MFEM_EXT_LIBS} ${lib}")
endif()
@@ -1013,7 +1085,7 @@ function(mfem_export_mk_files)
# Create the build-tree version of 'config.mk'
configure_file(
"${PROJECT_SOURCE_DIR}/config/config.mk.in"
"${PROJECT_BINARY_DIR}/config/config.mk")
"${PROJECT_BINARY_DIR}/config/config.mk" @ONLY)
# Copy 'test.mk' from the source-tree to the build-tree
configure_file(
"${PROJECT_SOURCE_DIR}/config/test.mk"
@@ -1031,7 +1103,7 @@ function(mfem_export_mk_files)
# Create the install-tree version of 'config.mk'
configure_file(
"${PROJECT_SOURCE_DIR}/config/config.mk.in"
"${PROJECT_BINARY_DIR}/config/config-install.mk")
"${PROJECT_BINARY_DIR}/config/config-install.mk" @ONLY)
# Install rules for 'config.mk' and 'test.mk'
install(FILES ${PROJECT_SOURCE_DIR}/config/test.mk
+6
View File
@@ -157,4 +157,10 @@ constexpr real_t operator""_r(unsigned long long v)
#endif
#endif // MFEM_USE_MPI not defined
#ifndef MFEM_USE_CUDA
#ifdef MFEM_USE_CUDSS
#error Building with cuDSS (MFEM_USE_CUDSS=YES) requires CUDA (MFEM_USE_CUDA=YES)
#endif
#endif // MFEM_USE_CUDSS not defined
#endif // MFEM_CONFIG_HPP
+9
View File
@@ -108,6 +108,15 @@
// Enable MFEM functionality based on the STRUMPACK library.
// #define MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
// #define MFEM_USE_CUDSS
// CUDSS communication layer library path
// #define MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
// #define MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable MFEM features based on the Ginkgo library.
// #define MFEM_USE_GINKGO
+3
View File
@@ -36,6 +36,9 @@ MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_SUPERLU5 = @MFEM_USE_SUPERLU5@
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_CUDSS = @MFEM_USE_CUDSS@
MFEM_CUDSS_COMM_LIB = @MFEM_CUDSS_COMM_LIB@
MFEM_CUDSS_THREADING_LIB = @MFEM_CUDSS_THREADING_LIB@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_AMGX = @MFEM_USE_AMGX@
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
+1
View File
@@ -38,6 +38,7 @@ option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_SUPERLU5 "Use the old SuperLU_DIST 5.1 version" OFF)
option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_CUDSS "Enable cuDSS usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
+15 -1
View File
@@ -153,6 +153,7 @@ MFEM_USE_SUPERLU = NO
MFEM_USE_SUPERLU5 = NO
MFEM_USE_MUMPS = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_CUDSS = NO
MFEM_USE_GINKGO = NO
MFEM_USE_AMGX = NO
MFEM_USE_MAGMA = NO
@@ -368,6 +369,19 @@ STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
$(SCOTCH_LIB) $(SCALAPACK_LIB)
# CUDSS library configuration
CUDSS_DIR = @MFEM_DIR@/../cudss
CUDSS_INCLUDE_DIR = $(CUDSS_DIR)/include
CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
CUDSS_LIB = \
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
# The cuDSS communication and threading libraries.
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR),$(CUDSS_LIBRARY_DIR)/libcudss_mtlayer_gomp.so))))
# Ginkgo library configuration
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
GINKGO_SEARCH_DIR = $(subst @MFEM_DIR@,$(MFEM_DIR),$(GINKGO_DIR))
@@ -621,7 +635,7 @@ PARELAG_LIB = -L$(PARELAG_DIR)/build/src -lParELAG
AXOM_DIR = @MFEM_DIR@/../axom
TRIBOL_DIR = @MFEM_DIR@/../tribol
TRIBOL_OPT = -I$(TRIBOL_DIR)/include -I$(AXOM_DIR)/include
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -ltribol_shared -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
-laxom_slam -laxom_slic -laxom_core
# Enzyme configuration
+1 -1
View File
@@ -215,7 +215,7 @@ if (MFEM_ENABLE_TESTING)
add_test(NAME ex1p_ceed_np=${MFEM_MPI_NP}
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:ex1p> "-no-vis" "-d ceed-cpu" "-pa" "-a"
$<TARGET_FILE:ex1p> "-no-vis" "-d" "ceed-cpu" "-pa" "-a"
${MPIEXEC_POSTFLAGS})
endif()
endif()
+1 -1
View File
@@ -64,7 +64,7 @@ PARALLEL_NAME := Parallel AMGX example
$(MFEM_LIB_FILE):
$(error The MFEM library is not build)
clean: clean-build
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
+3 -3
View File
@@ -64,12 +64,12 @@ ex1p-test-par: ex1p
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec $(SUBDIRS_CLEAN)
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
@rm -f refined.mesh mesh.*
@rm -f sol.*
+23 -11
View File
@@ -224,17 +224,29 @@ int main(int argc, char *argv[])
// 11. Solve the linear system A X = B.
if (!pa)
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#ifdef MFEM_USE_CUDSS
if (Device::Allows(Backend::CUDA_MASK))
{
// Use cuDSS to solve the system.
CuDSSSolver cudss_solver;
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
}
}
else
{
@@ -273,7 +285,7 @@ int main(int argc, char *argv[])
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << x << flush;
+2 -2
View File
@@ -5,9 +5,9 @@
// Sample runs:
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 462 -n 10 -o 2 -elast
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 464 -n 10 -o 2 -elast
// mpirun -np 4 ex12p -m ../data/beam-hex.mesh -s 3878
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 81
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 82
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3877 -o 2 -sys
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh -s 4544 -n 6 -o 3 -elast
// mpirun -np 4 ex12p -m ../data/beam-quad-nurbs.mesh
+47 -22
View File
@@ -83,6 +83,9 @@ int main(int argc, char *argv[])
const char *device_config = "cpu";
bool visualization = true;
bool algebraic_ceed = false;
#ifdef MFEM_USE_CUDSS
bool cudss_solver = false;
#endif
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -102,6 +105,10 @@ int main(int argc, char *argv[])
args.AddOption(&algebraic_ceed, "-a", "--algebraic",
"-no-a", "--no-algebraic",
"Use algebraic Ceed solver");
#endif
#ifdef MFEM_USE_CUDSS
args.AddOption(&cudss_solver, "-cudss", "--cudss-solver", "-no-cudss",
"--no-cudss-solver", "Use the cuDSS Solver.");
#endif
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
@@ -248,33 +255,51 @@ int main(int argc, char *argv[])
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use Jacobi smoothing, for now.
Solver *prec = NULL;
if (pa)
#ifdef MFEM_USE_CUDSS
if (!pa && (Device::Allows(Backend::CUDA_MASK) && cudss_solver))
{
if (UsesTensorBasis(fespace))
{
if (algebraic_ceed)
{
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
// Solve using a direct solver with cuDSS
CuDSSSolver cudss_solver(MPI_COMM_WORLD);
cudss_solver.SetMatrixSymType(
CuDSSSolver::SYMMETRIC_POSITIVE_DEFINITE);
cudss_solver.SetMatrixViewType(CuDSSSolver::UPPER);
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
prec = new HypreBoomerAMG;
Solver *prec = NULL;
if (pa)
{
if (UsesTensorBasis(fespace))
{
if (algebraic_ceed)
{
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
}
else
{
prec = new HypreBoomerAMG;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec)
{
cg.SetPreconditioner(*prec);
}
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec) { cg.SetPreconditioner(*prec); }
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
// 14. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
+27 -9
View File
@@ -302,15 +302,21 @@ int main(int argc, char *argv[])
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Exact: Real Part'" << flush;
// Make sure all ranks have sent their real solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
socketstream sol_sock_i(vishost, visport);
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i.precision(8);
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Exact: Imaginary Part'" << flush;
// Make sure all ranks have sent their imaginary solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
}
// 11. Set up the parallel sesquilinear form a(.,.) on the finite element
@@ -534,15 +540,21 @@ int main(int argc, char *argv[])
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u.real()
<< "window_title 'Solution: Real Part'" << flush;
// Make sure all ranks have sent their real solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
socketstream sol_sock_i(vishost, visport);
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i.precision(8);
sol_sock_i << "solution\n" << *pmesh << u.imag()
<< "window_title 'Solution: Imaginary Part'" << flush;
// Make sure all ranks have sent their imaginary solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
}
if (visualization && exact_sol)
{
@@ -551,15 +563,21 @@ int main(int argc, char *argv[])
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Error: Real Part'" << flush;
// Make sure all ranks have sent their real solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
socketstream sol_sock_i(vishost, visport);
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i.precision(8);
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Error: Imaginary Part'" << flush;
// Make sure all ranks have sent their imaginary solution before initiating
// another set of GLVis connections (one from each rank):
MPI_Barrier(pmesh->GetComm());
}
if (visualization)
{
+11 -52
View File
@@ -5,8 +5,8 @@
// Sample runs:
// ex37 -alpha 10
// ex37 -alpha 10 -pv
// ex37 -lambda 0.1 -mu 0.1
// ex37 -o 2 -alpha 5.0 -mi 50 -vf 0.4 -ntol 1e-5
// ex37 -lambda 0.1 -mu 0.1 -growth 1
// ex37 -o 2 -alpha 10.0 -mi 50 -vf 0.4 -ntol 1e-5 -growth 1.5
// ex37 -r 6 -o 1 -alpha 25.0 -epsilon 0.02 -mi 50 -ntol 1e-5
//
// Description: This example code demonstrates the use of MFEM to solve a
@@ -55,53 +55,6 @@
using namespace std;
using namespace mfem;
/**
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
* ∫_Ω ρ dx = θ vol(Ω) as follows:
*
* 1. Compute the root of the R → R function
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
* 2. Set ψ ← ψ + c.
*
* @param psi a GridFunction to be updated
* @param target_volume θ vol(Ω)
* @param tol Newton iteration tolerance
* @param max_its Newton maximum iteration number
* @return real_t Final volume, ∫_Ω sigmoid(ψ)
*/
real_t proj(GridFunction &psi, real_t target_volume, real_t tol=1e-12,
int max_its=10)
{
MappedGridFunctionCoefficient sigmoid_psi(&psi, sigmoid);
MappedGridFunctionCoefficient der_sigmoid_psi(&psi, der_sigmoid);
LinearForm int_sigmoid_psi(psi.FESpace());
int_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
LinearForm int_der_sigmoid_psi(psi.FESpace());
int_der_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(
der_sigmoid_psi));
bool done = false;
for (int k=0; k<max_its; k++) // Newton iteration
{
int_sigmoid_psi.Assemble(); // Recompute f(c) with updated ψ
const real_t f = int_sigmoid_psi.Sum() - target_volume;
int_der_sigmoid_psi.Assemble(); // Recompute df(c) with updated ψ
const real_t df = int_der_sigmoid_psi.Sum();
const real_t dc = -f/df;
psi += dc;
if (abs(dc) < tol) { done = true; break; }
}
if (!done)
{
mfem_warning("Projection reached maximum iteration without converging. "
"Result may not be accurate.");
}
int_sigmoid_psi.Assemble();
return int_sigmoid_psi.Sum();
}
/*
* ---------------------------------------------------------------
* ALGORITHM PREAMBLE
@@ -180,10 +133,11 @@ int main(int argc, char *argv[])
int ref_levels = 5;
int order = 2;
real_t alpha = 1.0;
real_t growth = 2;
real_t epsilon = 0.01;
real_t vol_fraction = 0.5;
int max_it = 1e3;
real_t itol = 1e-1;
real_t itol = 1e-2;
real_t ntol = 1e-4;
real_t rho_min = 1e-6;
real_t lambda = 1.0;
@@ -198,6 +152,8 @@ int main(int argc, char *argv[])
"Order (degree) of the finite elements.");
args.AddOption(&alpha, "-alpha", "--alpha-step-length",
"Step length for gradient descent.");
args.AddOption(&growth, "-growth", "--alpha-growth-rate",
"Growth rate of step length for gradient descent.");
args.AddOption(&epsilon, "-epsilon", "--epsilon-thickness",
"Length scale for ρ.");
args.AddOption(&max_it, "-mi", "--max-it",
@@ -332,6 +288,7 @@ int main(int argc, char *argv[])
}
FilterSolver->SetEssentialBoundary(ess_bdr_filter);
FilterSolver->SetupFEM();
FilterSolver->AssembleDiffusionBilinear();
BilinearForm mass(&control_fes);
mass.AddDomainIntegrator(new InverseIntegrator(new MassIntegrator(one)));
@@ -385,7 +342,7 @@ int main(int argc, char *argv[])
// 11. Iterate:
for (int k = 1; k <= max_it; k++)
{
if (k > 1) { alpha *= ((real_t) k) / ((real_t) k-1); }
if (k > 1) { alpha = std::pow((real_t) k,growth); }
mfem::out << "\nStep = " << k << std::endl;
@@ -422,7 +379,9 @@ int main(int argc, char *argv[])
// Step 5 - Update design variable ψ ← proj(ψ - αG)
psi.Add(-alpha, grad);
const real_t material_volume = proj(psi, target_volume);
GridFunction alpha_grad(grad);
alpha_grad *= alpha;
const real_t material_volume = proj(psi, alpha_grad, target_volume);
// Compute ||ρ - ρ_old|| in control fes.
real_t norm_increment = zerogf.ComputeL1Error(succ_diff_rho);
+189 -29
View File
@@ -137,7 +137,7 @@ public:
exponent(exponent_), rho_min(rho_min_)
{
MFEM_ASSERT(rho_min_ >= 0.0, "rho_min must be >= 0");
MFEM_ASSERT(rho_min_ < 1.0, "rho_min must be > 1");
MFEM_ASSERT(rho_min_ < 1.0, "rho_min must be < 1");
MFEM_ASSERT(u, "displacement field is not set");
MFEM_ASSERT(rho_filter, "density field is not set");
}
@@ -231,9 +231,12 @@ private:
FiniteElementCollection * fec = nullptr;
FiniteElementSpace * fes = nullptr;
Array<int> ess_bdr;
Array<int> ess_tdof_list;
Array<int> neumann_bdr;
GridFunction * u = nullptr;
LinearForm * b = nullptr;
BilinearForm * a = nullptr;
OperatorPtr A;
bool parallel;
#ifdef MFEM_USE_MPI
ParMesh * pmesh = nullptr;
@@ -267,6 +270,8 @@ public:
void ResetFEM();
void SetupFEM();
void UpdateEssentialTDofs();
void AssembleDiffusionBilinear(bool update_ess_tdofs=true);
void Solve();
GridFunction * GetFEMSolution();
LinearForm * GetLinearForm() {return b;}
@@ -371,6 +376,130 @@ public:
};
/**
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
* ∫_Ω ρ dx = θ vol(Ω) as follows:
*
* 1. Compute the root of the R → R function
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
* using the Illinois method
* 2. Set ψ ← ψ + c.
*
* @param psi a GridFunction to be updated
* @param alpha_grad alpha multiplied by gradient
* @param target_volume θ vol(Ω)
* @param tol Illinois iteration tolerance
* @param max_its Illinois maximum iteration number
* @return real_t Final volume (∫_Ω sigmoid(ψ) dx)
*/
real_t proj(GridFunction &psi, GridFunction &alpha_grad, real_t target_volume,
real_t tol = 1e-12, int max_its = 100)
{
#ifdef MFEM_USE_MPI
FiniteElementSpace *fes = psi.FESpace();
ParFiniteElementSpace *pfes = dynamic_cast<ParFiniteElementSpace*>(fes);
#endif
ConstantCoefficient zero_cf(0.0);
real_t a = -alpha_grad.ComputeMaxError(zero_cf);
real_t b = -a;
real_t y = 0.0;
MappedGridFunctionCoefficient sigmoid_psi(
&psi, [&y](const real_t x) { return sigmoid(x + y); });
std::unique_ptr<LinearForm> int_sigmoid_psi;
#ifdef MFEM_USE_MPI
ParGridFunction *par_psi = dynamic_cast<ParGridFunction *>(&psi);
if (par_psi)
{
int_sigmoid_psi.reset(new ParLinearForm(par_psi->ParFESpace()));
}
else
{
int_sigmoid_psi.reset(new LinearForm(psi.FESpace()));
}
#else
int_sigmoid_psi.reset(new LinearForm(psi.FESpace()));
#endif
int_sigmoid_psi->AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
y = a;
int_sigmoid_psi->Assemble();
real_t f_a = int_sigmoid_psi->Sum(); // f_a := f(a) + θ vol(Ω)
y = b;
int_sigmoid_psi->Assemble();
real_t f_b = int_sigmoid_psi->Sum(); // f_b := f(b) + θ vol(Ω)
#ifdef MFEM_USE_MPI
if (pfes)
{
MPI_Allreduce(MPI_IN_PLACE, &f_a, 1, MPITypeMap<real_t>::mpi_type,
MPI_SUM, MPI_COMM_WORLD);
MPI_Allreduce(MPI_IN_PLACE, &f_b, 1, MPITypeMap<real_t>::mpi_type,
MPI_SUM, MPI_COMM_WORLD);
}
#endif
f_a -= target_volume; // f_a := f(a)
f_b -= target_volume; // f_b := f(b)
real_t c = 0.0;
real_t f_c = 0.0;
int side = 0;
bool done = false;
for (int k=0; k < max_its; k++)
{
c = (f_a * b - f_b * a) / (f_a - f_b);
if (abs(b - a) < tol * abs(b + a)) { done = true; break; }
y = c;
int_sigmoid_psi->Assemble();
f_c = int_sigmoid_psi->Sum(); // f_c := f(c) + θ vol(Ω)
#ifdef MFEM_USE_MPI
if (pfes)
{
MPI_Allreduce(MPI_IN_PLACE, &f_c, 1, MPITypeMap<real_t>::mpi_type,
MPI_SUM, MPI_COMM_WORLD);
}
#endif
f_c -= target_volume; // f_c := f(c)
if (f_c * f_b > 0)
{
b = c;
f_b = f_c;
if (side == -1) { f_a /= 2.0; }
side = -1;
}
else if (f_c * f_a > 0)
{
a = c;
f_a = f_c;
if (side == 1) { f_b /= 2.0; }
side = 1;
}
else
{
done = true; break;
}
}
if (!done)
{
mfem_warning("Projection reached maximum iteration without converging. "
"Result may not be accurate.");
}
y = 0.0;
psi += c;
int_sigmoid_psi->Assemble();
real_t material_volume = int_sigmoid_psi->Sum();
#ifdef MFEM_USE_MPI
if (pfes)
{
MPI_Allreduce(MPI_IN_PLACE, &material_volume, 1,
MPITypeMap<real_t>::mpi_type, MPI_SUM, MPI_COMM_WORLD);
}
#endif
return material_volume;
}
// Poisson solver
@@ -422,12 +551,8 @@ void DiffusionSolver::SetupFEM()
}
}
void DiffusionSolver::Solve()
void DiffusionSolver::UpdateEssentialTDofs()
{
OperatorPtr A;
Vector B, X;
Array<int> ess_tdof_list;
#ifdef MFEM_USE_MPI
if (parallel)
{
@@ -440,7 +565,39 @@ void DiffusionSolver::Solve()
#else
fes->GetEssentialTrueDofs(ess_bdr,ess_tdof_list);
#endif
*u=0.0;
}
void DiffusionSolver::AssembleDiffusionBilinear(bool update_ess_tdofs)
{
if (update_ess_tdofs)
{
UpdateEssentialTDofs();
}
#ifdef MFEM_USE_MPI
if (parallel)
{
a = new ParBilinearForm(pfes);
}
else
{
a = new BilinearForm(fes);
}
#else
a = new BilinearForm(fes);
#endif
a->AddDomainIntegrator(new DiffusionIntegrator(*diffcf));
if (masscf)
{
a->AddDomainIntegrator(new MassIntegrator(*masscf));
}
a->Assemble();
a->FormSystemMatrix(ess_tdof_list, A);
}
void DiffusionSolver::Solve()
{
Vector B, X;
if (b)
{
delete b;
@@ -475,31 +632,33 @@ void DiffusionSolver::Solve()
b->Assemble();
BilinearForm * a = nullptr;
#ifdef MFEM_USE_MPI
if (parallel)
{
a = new ParBilinearForm(pfes);
}
else
{
a = new BilinearForm(fes);
}
#else
a = new BilinearForm(fes);
#endif
a->AddDomainIntegrator(new DiffusionIntegrator(*diffcf));
if (masscf)
{
a->AddDomainIntegrator(new MassIntegrator(*masscf));
}
a->Assemble();
*u=0.0;
if (essbdr_cf)
{
u->ProjectBdrCoefficient(*essbdr_cf,ess_bdr);
}
a->FormLinearSystem(ess_tdof_list, *u, *b, A, X, B);
#ifdef MFEM_USE_MPI
if (parallel)
{
X.SetSize(pfes->TrueVSize());
B.SetSize(pfes->TrueVSize());
dynamic_cast<ParGridFunction*>(u)->ParallelAssemble(X);
dynamic_cast<ParLinearForm*>(b)->ParallelAssemble(B);
dynamic_cast<ParBilinearForm*>(a)->ParallelEliminateTDofsInRHS(
ess_tdof_list, X, B);
}
else
{
X.NewDataAndSize(u->GetData(), u->Size());
B.NewDataAndSize(b->GetData(), b->Size());
a->EliminateVDofsInRHS(ess_tdof_list, X, B);
}
#else
X.NewDataAndSize(u->GetData(), u->Size());
B.NewDataAndSize(b->GetData(), b->Size());
a->EliminateVDofsInRHS(ess_tdof_list, X, B);
#endif
CGSolver * cg = nullptr;
Solver * M = nullptr;
@@ -528,7 +687,6 @@ void DiffusionSolver::Solve()
delete M;
delete cg;
a->RecoverFEMSolution(X, *b, *u);
delete a;
}
GridFunction * DiffusionSolver::GetFEMSolution()
@@ -560,6 +718,8 @@ DiffusionSolver::~DiffusionSolver()
#endif
delete fec; fec = nullptr;
delete b;
A.Clear();
delete a;
}
+11 -60
View File
@@ -4,8 +4,8 @@
//
// Sample runs:
// mpirun -np 4 ex37p -alpha 10 -pv
// mpirun -np 4 ex37p -lambda 0.1 -mu 0.1
// mpirun -np 4 ex37p -o 2 -alpha 5.0 -mi 50 -vf 0.4 -ntol 1e-5
// mpirun -np 4 ex37p -lambda 0.1 -mu 0.1 -growth 1
// mpirun -np 4 ex37p -o 2 -alpha 10.0 -mi 50 -vf 0.4 -ntol 1e-5 -growth 1.5
// mpirun -np 4 ex37p -r 6 -o 2 -alpha 10.0 -epsilon 0.02 -mi 50 -ntol 1e-5
//
// Description: This example code demonstrates the use of MFEM to solve a
@@ -54,61 +54,6 @@
using namespace std;
using namespace mfem;
/**
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
* ∫_Ω ρ dx = θ vol(Ω) as follows:
*
* 1. Compute the root of the R → R function
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
* 2. Set ψ ← ψ + c.
*
* @param psi a GridFunction to be updated
* @param target_volume θ vol(Ω)
* @param tol Newton iteration tolerance
* @param max_its Newton maximum iteration number
* @return real_t Final volume, ∫_Ω sigmoid(ψ)
*/
real_t proj(ParGridFunction &psi, real_t target_volume, real_t tol=1e-12,
int max_its=10)
{
MappedGridFunctionCoefficient sigmoid_psi(&psi, sigmoid);
MappedGridFunctionCoefficient der_sigmoid_psi(&psi, der_sigmoid);
ParLinearForm int_sigmoid_psi(psi.ParFESpace());
int_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
ParLinearForm int_der_sigmoid_psi(psi.ParFESpace());
int_der_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(
der_sigmoid_psi));
bool done = false;
for (int k=0; k<max_its; k++) // Newton iteration
{
int_sigmoid_psi.Assemble(); // Recompute f(c) with updated ψ
real_t f = int_sigmoid_psi.Sum();
MPI_Allreduce(MPI_IN_PLACE, &f, 1, MPITypeMap<real_t>::mpi_type,
MPI_SUM, MPI_COMM_WORLD);
f -= target_volume;
int_der_sigmoid_psi.Assemble(); // Recompute df(c) with updated ψ
real_t df = int_der_sigmoid_psi.Sum();
MPI_Allreduce(MPI_IN_PLACE, &df, 1, MPITypeMap<real_t>::mpi_type,
MPI_SUM, MPI_COMM_WORLD);
const real_t dc = -f/df;
psi += dc;
if (abs(dc) < tol) { done = true; break; }
}
if (!done)
{
mfem_warning("Projection reached maximum iteration without converging. "
"Result may not be accurate.");
}
int_sigmoid_psi.Assemble();
real_t material_volume = int_sigmoid_psi.Sum();
MPI_Allreduce(MPI_IN_PLACE, &material_volume, 1,
MPITypeMap<real_t>::mpi_type, MPI_SUM, MPI_COMM_WORLD);
return material_volume;
}
/*
* ---------------------------------------------------------------
* ALGORITHM PREAMBLE
@@ -193,10 +138,11 @@ int main(int argc, char *argv[])
int ref_levels = 5;
int order = 2;
real_t alpha = 1.0;
real_t growth = 2;
real_t epsilon = 0.01;
real_t vol_fraction = 0.5;
int max_it = 1e3;
real_t itol = 1e-1;
real_t itol = 1e-2;
real_t ntol = 1e-4;
real_t rho_min = 1e-6;
real_t lambda = 1.0;
@@ -211,6 +157,8 @@ int main(int argc, char *argv[])
"Order (degree) of the finite elements.");
args.AddOption(&alpha, "-alpha", "--alpha-step-length",
"Step length for gradient descent.");
args.AddOption(&growth, "-growth", "--alpha-growth-rate",
"Growth rate of step length for gradient descent.");
args.AddOption(&epsilon, "-epsilon", "--epsilon-thickness",
"Length scale for ρ.");
args.AddOption(&max_it, "-mi", "--max-it",
@@ -359,6 +307,7 @@ int main(int argc, char *argv[])
}
FilterSolver->SetEssentialBoundary(ess_bdr_filter);
FilterSolver->SetupFEM();
FilterSolver->AssembleDiffusionBilinear();
ParBilinearForm mass(&control_fes);
mass.AddDomainIntegrator(new InverseIntegrator(new MassIntegrator(one)));
@@ -412,7 +361,7 @@ int main(int argc, char *argv[])
// 11. Iterate:
for (int k = 1; k <= max_it; k++)
{
if (k > 1) { alpha *= ((real_t) k) / ((real_t) k-1); }
if (k > 1) { alpha = std::pow((real_t) k,growth); }
if (myid == 0)
{
@@ -452,7 +401,9 @@ int main(int argc, char *argv[])
// Step 5 - Update design variable ψ ← proj(ψ - αG)
psi.Add(-alpha, grad);
const real_t material_volume = proj(psi, target_volume);
ParGridFunction alpha_grad(grad);
alpha_grad *= alpha;
const real_t material_volume = proj(psi, alpha_grad, target_volume);
// Compute ||ρ - ρ_old|| in control fes.
real_t norm_increment = zerogf.ComputeL1Error(succ_diff_rho);
+1 -1
View File
@@ -76,4 +76,4 @@ clean-build:
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh sol.gf
@rm -f refined.mesh sol.gf mesh.* sol.*
+7 -2
View File
@@ -71,6 +71,7 @@ endif
SUBDIRS_ALL = $(addsuffix /all,$(SUBDIRS))
SUBDIRS_TEST = $(addsuffix /test,$(SUBDIRS))
SUBDIRS_TEST_NOCLEAN = $(addsuffix /test-noclean,$(SUBDIRS))
SUBDIRS_CLEAN = $(addsuffix /clean,$(SUBDIRS))
SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
@@ -87,8 +88,9 @@ SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
all: $(EXAMPLES) $(SUBDIRS_ALL)
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN):
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) \
$(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) $(SUBDIRS_CLEAN):
$(MAKE) -C $(@D) $(@F)
$(SUBDIRS_TPRINT):
@$(MAKE) -C $(@D) $(@F)
@@ -107,6 +109,7 @@ endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
test: $(SUBDIRS_TEST)
test-noclean: $(SUBDIRS_TEST_NOCLEAN)
test-print: $(SUBDIRS_TPRINT)
# Testing: Parallel vs. serial runs
@@ -157,6 +160,8 @@ ex37-test-seq: ex37
@$(call mfem-test,$<,, Serial example,-mi 3)
ex37p-test-par: ex37p
@$(call mfem-test,$<, $(RUN_MPI), Parallel example,-mi 3)
ex39-test-seq: ex39
@$(call mfem-test,$<,, Serial example,-m ../data/compass.mesh)
ex41-test-seq: ex41
@$(call mfem-test,$<,, Serial example,-tf 1.0)
ex41p-test-par: ex41p
+4
View File
@@ -171,8 +171,12 @@ set(SRCS
tmop_tools.cpp
tmop_amr.cpp
gslib.cpp
gslib/findptsedge_local_2.cpp
gslib/findptsedge_local_3.cpp
gslib/findptssurf_local_3.cpp
gslib/findpts_local_2.cpp
gslib/findpts_local_3.cpp
gslib/interpolate_local_1.cpp
gslib/interpolate_local_2.cpp
gslib/interpolate_local_3.cpp
transfer.cpp
+7 -3
View File
@@ -729,7 +729,8 @@ void BilinearForm::Assemble(int skip_zeros)
tr = mesh -> GetBdrFaceTransformations (i);
if (tr != NULL)
{
fes -> GetElementVDofs (tr -> Elem1No, vdofs);
mfem::DofTransformation doftrans;
fes -> GetElementVDofs (tr -> Elem1No, vdofs, doftrans);
fe1 = fes -> GetFE (tr -> Elem1No);
// The fe2 object is really a dummy and not used on the boundaries,
// but we can't dereference a NULL pointer, and we don't want to
@@ -743,6 +744,7 @@ void BilinearForm::Assemble(int skip_zeros)
boundary_face_integs[k] -> AssembleFaceMatrix (*fe1, *fe2, *tr,
elemmat);
doftrans.TransformDual(elemmat);
mat -> AddSubMatrix (vdofs, vdofs, elemmat, skip_zeros);
}
}
@@ -1723,6 +1725,7 @@ void MixedBilinearForm::Assemble(int skip_zeros)
}
}
DofTransformation dom_dof_trans, ran_dof_trans;
for (int i = 0; i < trial_fes -> GetNBE(); i++)
{
const int bdr_attr = mesh->GetBdrAttribute(i);
@@ -1731,8 +1734,8 @@ void MixedBilinearForm::Assemble(int skip_zeros)
ftr = mesh -> GetBdrFaceTransformations (i);
if (ftr != NULL)
{
trial_fes->GetElementVDofs(ftr->Elem1No, trial_vdofs);
test_fes->GetElementVDofs(ftr->Elem1No, test_vdofs);
trial_fes->GetElementVDofs(ftr->Elem1No, trial_vdofs, dom_dof_trans);
test_fes->GetElementVDofs(ftr->Elem1No, test_vdofs, ran_dof_trans);
trial_fe1 = trial_fes->GetFE(ftr->Elem1No);
test_fe1 = test_fes->GetFE(ftr->Elem1No);
// The test_fe2 object is really a dummy and not used on the
@@ -1748,6 +1751,7 @@ void MixedBilinearForm::Assemble(int skip_zeros)
boundary_face_integs[k]->AssembleFaceMatrix(*trial_fe1, *test_fe1, *trial_fe2,
*test_fe2,
*ftr, elemmat);
TransformDual(ran_dof_trans, dom_dof_trans, elemmat);
mat->AddSubMatrix(test_vdofs, trial_vdofs, elemmat, skip_zeros);
}
}
+1 -1
View File
@@ -2710,7 +2710,7 @@ public:
/** Integrator for $(-Q u, \nabla v)$ for Nedelec ($u$) and $H^1$ ($v$) elements.
This is equivalent to a weak divergence of the $H(curl$ basis functions. */
This is equivalent to a weak divergence of the $H(curl)$ basis functions. */
class VectorFEWeakDivergenceIntegrator: public BilinearFormIntegrator
{
protected:
+67 -28
View File
@@ -39,8 +39,8 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
b_type = b_type_i;
cp_type = cp_type_i;
tol = tol_i;
lbound.SetSize(nb, ncp);
ubound.SetSize(nb, ncp);
lbound.SetSize(ncp, nb);
ubound.SetSize(ncp, nb);
nodes.SetSize(nb);
weights.SetSize(nb);
control_points.SetSize(ncp);
@@ -125,21 +125,25 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
{
if (j == 0)
{
lbound(i, j) = bv(i);
ubound(i, j) = bv(i);
lbound(j,i) = bv(i);
ubound(j,i) = bv(i);
}
else if (j == ncp-1)
{
lbound(i, j) = bv(i);
ubound(i, j) = bv(i);
lbound(j,i) = bv(i);
ubound(j,i) = bv(i);
}
else
{
vals(0) = bv(i);
vals(1) = bmv(i) + dm*bdmv(i);
vals(2) = bpv(i) + dp*bdpv(i);
lbound(i, j) = vals.Min()-tol; // tolerance for good measure
ubound(i, j) = vals.Max()+tol; // tolerance for good measure
lbound(j,i) = vals.Min()-tol; // tolerance for good measure
ubound(j,i) = vals.Max()+tol; // tolerance for good measure
if (b_type == 2)
{
lbound(j,i) = std::max(lbound(j,i),0_r);
}
}
}
}
@@ -273,8 +277,7 @@ void PLBound::Get1DBounds(const Vector &coeff, Vector &intmin,
intmax.SetSize(ncp);
intmin = 0.0;
intmax = 0.0;
Vector coeffm(nb);
coeffm = 0.0;
Vector coeffm;
real_t a0 = 0.0;
real_t a1 = 0.0;
@@ -302,6 +305,8 @@ void PLBound::Get1DBounds(const Vector &coeff, Vector &intmin,
// compute L2 projection for linear bases: a0 + a1*x
if (proj)
{
coeffm.SetSize(nb);
coeffm = 0.0;
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes_int(i)-1;
@@ -342,8 +347,8 @@ void PLBound::Get1DBounds(const Vector &coeff, Vector &intmin,
real_t c = coeffm(i);
for (int j = 0; j < ncp; j++)
{
intmin(j) += min(lbound(i,j)*c, ubound(i,j)*c);
intmax(j) += max(lbound(i,j)*c, ubound(i,j)*c);
intmin(j) += min(lbound(j,i)*c, ubound(j,i)*c);
intmax(j) += max(lbound(j,i)*c, ubound(j,i)*c);
}
}
}
@@ -474,10 +479,10 @@ void PLBound::Get2DBounds(const Vector &coeff, Vector &intmin,
real_t w1 = intmaxT(id2++);
for (int k = 0; k < ncp; k++) // kth row
{
vals(0) = w0*lbound(j,k);
vals(1) = w0*ubound(j,k);
vals(2) = w1*lbound(j,k);
vals(3) = w1*ubound(j,k);
vals(0) = w0*lbound(k,j);
vals(1) = w0*ubound(k,j);
vals(2) = w1*lbound(k,j);
vals(3) = w1*ubound(k,j);
intmin(k*ncp+i) += vals.Min();
intmax(k*ncp+i) += vals.Max();
}
@@ -553,17 +558,17 @@ void PLBound::Get3DBounds(const Vector &coeff, Vector &intmin,
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes(i)-1; // x-coordinate
minBounds(i) -= a0V(j) + a1V(j)*x;
maxBounds(i) -= a0V(j) + a1V(j)*x;
minNodalVals(i) -= a0V(j) + a1V(j)*x;
maxNodalVals(i) -= a0V(j) + a1V(j)*x;
}
// Compute Bernstein coefficients
LUFactors lu(basisMatLU.GetData(), lu_ip.GetData());
lu.Solve(nb, 1, minBounds.GetData());
lu.Solve(nb, 1, maxBounds.GetData());
lu.Solve(nb, 1, minNodalVals.GetData());
lu.Solve(nb, 1, maxNodalVals.GetData());
for (int i = 0; i < nb; i++)
{
intminT(i*ncp2+j) = minBounds(i);
intmaxT(i*ncp2+j) = maxBounds(i);
intminT(i*ncp2+j) = minNodalVals(i);
intmaxT(i*ncp2+j) = maxNodalVals(i);
}
}
}
@@ -617,10 +622,10 @@ void PLBound::Get3DBounds(const Vector &coeff, Vector &intmin,
real_t w1 = intmaxT(id2++);
for (int k = 0; k < ncp; k++) // kth slice
{
vals(0) = w0*lbound(j,k);
vals(1) = w0*ubound(j,k);
vals(2) = w1*lbound(j,k);
vals(3) = w1*ubound(j,k);
vals(0) = w0*lbound(k,j);
vals(1) = w0*ubound(k,j);
vals(2) = w1*lbound(k,j);
vals(3) = w1*ubound(k,j);
intmin(k*ncp2+i) += vals.Min();
intmax(k*ncp2+i) += vals.Max();
}
@@ -653,7 +658,8 @@ void PLBound::SetupBernsteinBasisMat(DenseMatrix &basisMat,
Vector &nodesBern) const
{
const int nbern = nodesBern.Size();
L2_SegmentElement el(nbern-1, 2); // we use L2 to leverage lexicographic order
L2_SegmentElement el(nbern-1, 2);
// we use L2 to leverage lexicographic order
Array<int> ordering = el.GetLexicographicOrdering();
basisMat.SetSize(nbern, nbern);
Vector shape(nbern);
@@ -666,6 +672,39 @@ void PLBound::SetupBernsteinBasisMat(DenseMatrix &basisMat,
}
}
DenseMatrix PLBound::GetBoundingMatrix(int dim, bool is_lower) const
{
if (dim > 1)
{
const int ncpd = static_cast<int>(std::pow(ncp, dim));
const int nbd = static_cast<int>(std::pow(nb, dim));
DenseMatrix boundND(ncpd, nbd);
Vector phimin, phimax, col;
Vector coeffs(nbd);
coeffs = 0.0;
for (int j = 0; j < nbd; j++)
{
coeffs(j) = 1.0;
boundND.GetColumnReference(j, col);
GetNDBounds(dim, coeffs, phimin, phimax);
col = is_lower ? phimin : phimax;
coeffs(j) = 0.0;
}
return boundND;
}
return is_lower ? lbound : ubound;
}
DenseMatrix PLBound::GetLowerBoundMatrix(int dim) const
{
return GetBoundingMatrix(dim, true);
}
DenseMatrix PLBound::GetUpperBoundMatrix(int dim) const
{
return GetBoundingMatrix(dim, false);
}
constexpr int PLBound::min_ncp_gl_x[2][11];
constexpr int PLBound::min_ncp_gll_x[2][11];
constexpr int PLBound::min_ncp_pos_x[2][11];
@@ -716,4 +755,4 @@ void PLBound::Print(std::ostream &outp) const
ubound.Print(outp);
}
}
}
+71 -20
View File
@@ -19,14 +19,18 @@ namespace mfem
{
/** @name Piecewise linear bounds of bases
\brief Piecewise linear bounds of bases can be used to compute bounds on the grid function in each element. The bounds for the bases are constructed based on the following parameters:
\brief Piecewise linear bounds of bases can be used to compute bounds on
the grid function in each element. The bounds for the bases are constructed
based on the following parameters:
(i) @b nb: number of bases/nodes in 1D (i.e. polynomial order+1),
(ii) @b b_type: bases type, 0 - Lagrange interpolants on Gauss-Legendre nodes, 1 - Lagrange interpolants on Gauss-Lobatto-Legendre nodes, and
(ii) @b b_type: bases type, 0 - Lagrange interpolants on Gauss-Legendre
nodes, 1 - Lagrange interpolants on Gauss-Lobatto-Legendre nodes, and
2 - Positive/Bernstein bases on uniformly distributed nodes,
(iii) @b ncp: number of control points used to construct the piecewise linear bounds
(iii) @b ncp: number of control points used to construct the piecewise
linear bounds
(iv) @b cp_type: control point distribution. 0 - GL + end-points,
1 - Chebyshev.
@@ -35,7 +39,9 @@ namespace mfem
If the user does not specify @b ncp and @b cp_type, the minimum value of
@b ncp is used that would bound the bases for the @b cp_type. We default
to @b cp_type = 0 as it requires fewer number of points to bound the bases. Typically, @b ncp = 2 @b nb is sufficient to get fairly compact bounds, and increasing @b ncp results in tighter bounds.
to @b cp_type = 0 as it requires fewer number of points to bound the bases.
Typically, @b ncp = 2 @b nb is sufficient to get fairly compact bounds, and
increasing @b ncp results in tighter bounds.
Finally, only tensor-product elements are currently supported.
@@ -54,7 +60,7 @@ private:
bool proj = true; // Use linear projection to compute bounds.
real_t tol = 0.0; // offset bounds to avoid round-off errors
Vector nodes, weights, control_points;
DenseMatrix lbound, ubound; // nb x ncp matrices with bounds of all bases
DenseMatrix lbound, ubound; // ncp x nb matrices with bounds of all bases
// Some auxillary storage for computing the bounds with Bernstein
DenseMatrix basisMatNodes; // Bernstein bases at equispaced nodes
DenseMatrix basisMatInt; // Bernstein bases at GLL nodes
@@ -80,6 +86,9 @@ private:
{3,5,8,9,11,12,13,13,14,15,16}
};
/// Helper function to extract lower or upper bounding matrix
DenseMatrix GetBoundingMatrix(int dim, bool is_lower) const;
public:
// Constructor
PLBound(const int nb_i, const int ncp_i, const int b_type_i,
@@ -92,40 +101,82 @@ public:
PLBound(const FiniteElementSpace *fes,
const int ncp_i = -1, const int cp_type_i = 0);
// Get minimum number of control points needed to bound the given bases
/// Get minimum number of control points needed to bound the given bases
int GetMinimumPointsForGivenBases(int nb_i, int b_type_i,
int cp_type_i) const;
// Print information about the bounds
/// Print information about the bounds
void Print(std::ostream &outp = mfem::out) const;
// Enable (default) or disable linear projection before bounding.
// This projection increases the computational cost but results in tighter
// bounds.
/** @brief Enable (default) or disable linear projection before bounding.
*
* @details This projection increases the computational cost but results in
* tighter bounds.
*/
void SetProjectionFlagForBounding(bool proj_) { proj = proj_; }
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D/2D/3D.
/** @brief Compute piecewise linear bounds for the lexicographically-ordered
* nodal coefficients in @a coeff in 1D/2D/3D.
*
* @param[in] rdim The spatial dimension of the element (1, 2, or 3).
* @param[in] coeff The vector of lexicographically-ordered coefficients.
* Should be of size nb^rdim, where nb is the number of
* bases/nodes in 1D. These coefficients must correspond
* to the bases type and number of bases, used in the
* constructor of PLBound.
*
* @param[out] intmin The vector of minimum bound for all control points.
* @param[out] intmax The vector of maximum bound for all control points.
* Both intmin and intmax are of size ncp^rdim, where
* ncp is the number of control points in 1D, and are
* ordered lexicographically.
*/
void GetNDBounds(const int rdim, const Vector &coeff,
Vector &intmin, Vector &intmax) const;
/// Get number of control points used to compute the bounds.
int GetNControlPoints() const { return ncp; }
/// Get 1D control point locations (lexicographic order) in [0,1].
const Vector &GetControlPoints() const { return control_points; }
/** @brief Get lower and upper bounding matrix (ncp^dim x nb^dim)
*
* @details The matrices can be used to compute the bounds at control points
* by a simple matrix-vector product with the
* lexicographically-ordered nodal coefficients.
* The resulting output is also lexicographically-ordered.
*
* @note These matrices do not account for the linear projection step that
* is optionally done in GetNDBounds before bounding the function.
*/
///@{
DenseMatrix GetLowerBoundMatrix(int dim = 1) const;
DenseMatrix GetUpperBoundMatrix(int dim = 1) const;
///@}
private:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D.
/** @brief Compute piecewise linear bounds for the lexicographically-ordered
* nodal coefficients in @a coeff in 1D.
* See GetNDBounds for details of the input and output parameters.
*/
void Get1DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 2D.
/** @brief Compute piecewise linear bounds for the lexicographically-ordered
* nodal coefficients in @a coeff in 2D.
* See GetNDBounds for details of the input and output parameters.
*/
void Get2DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 3D.
/** @brief Compute piecewise linear bounds for the lexicographically-ordered
* nodal coefficients in @a coeff in 3D.
* See GetNDBounds for details of the input and output parameters.
*/
void Get3DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Setup matrix used to compute values at given 1D locations in [0,1]
/// for Bernstein bases.
/** @brief Setup matrix used to compute values at given 1D locations in [0,1]
* for Bernstein bases.
*/
void SetupBernsteinBasisMat(DenseMatrix &basisMat, Vector &nodesBern) const;
void Setup(const int nb_i, const int ncp_i, const int b_type_i,
+3
View File
@@ -52,6 +52,9 @@ public:
/// Get the time for time dependent coefficients
real_t GetTime() { return time; }
/// Returns dimension of the vector.
int GetVDim() { return 1; }
/** @brief Evaluate the coefficient in the element described by @a T at the
point @a ip. */
/** @note When this method is called, the caller must make sure that the
+13 -26
View File
@@ -830,15 +830,9 @@ ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
int vsize = pfes->GetVSize();
Vector::Load(input, 2*vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
real_t *h_data = HostReadWrite();
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
@@ -1051,15 +1045,14 @@ void ParComplexGridFunction::Save(std::ostream &os) const
os << '\n';
int vsize = pfes->GetVSize();
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t *h_data = const_cast<real_t*>(HostRead());
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
if (pfes->GetOrdering() == Ordering::byNODES)
{
@@ -1070,14 +1063,8 @@ void ParComplexGridFunction::Save(std::ostream &os) const
Vector::Print(os, pfes->GetVDim());
}
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
os.flush();
}
+19
View File
@@ -82,6 +82,25 @@ public:
/// underlying #fes
int VectorDim() const;
/// Copy assignment. Only the data of the base class Vector is copied.
/** It is assumed that this object and @a rhs use FiniteElementSpace%s that
have the same size.
@note Defining this method overwrites the implicitly defined copy
assignment operator. */
ComplexGridFunction &operator=(const ComplexGridFunction &rhs)
{ return operator=((const Vector &)rhs); }
/// Copy the data from @a v.
/** The size of @a v must be equal to double of the size of the associated
FiniteElementSpace #fes. */
ComplexGridFunction &operator=(const Vector &v)
{
MFEM_ASSERT(fes && v.Size() == 2*fes->GetVSize(), "");
Vector::operator=(v);
return *this;
}
/// Assign constant values to the ComplexGridFunction data.
ComplexGridFunction &operator=(const std::complex<real_t> & value)
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
+4
View File
@@ -114,6 +114,10 @@ void ConduitDataCollection::Save()
n_mesh["fields"][name]);
}
// TODO: in parallel, we need to call ParFiniteElementSpace::ApplyDofSigns
// for all ParGridFunction objects before and after saving, see
// ParGridFunction::Save.
// save mesh data
SaveMeshAndFields(myid,
n_mesh,
+18 -5
View File
@@ -492,6 +492,8 @@ void VisItDataCollection::SaveRootFile()
to_padded_string(cycle, pad_digits_cycle) +
".mfem_root";
std::ofstream root_file(root_name);
MFEM_VERIFY(root_file.is_open(),
"Failed to open ofstream " << root_name);
root_file << GetVisItRootString();
if (!root_file)
{
@@ -977,7 +979,10 @@ void ParaViewDataCollection::Save()
// Save the local part of the mesh and grid functions fields to the local
// VTU file. Also save coefficient fields.
{
std::ofstream os(vtu_prefix + GenerateVTUFileName("proc", myid));
std::string os_str = vtu_prefix + GenerateVTUFileName("proc", myid);
std::ofstream os(os_str);
MFEM_VERIFY(os.is_open(),
"Failed to open ofstream " << os_str);
os.precision(precision);
SaveDataVTU(os, levels_of_detail);
}
@@ -989,7 +994,10 @@ void ParaViewDataCollection::Save()
"QuadratureFunction output is not supported for "
"ParaViewDataCollection on domain boundary!");
const std::string &field_name = qfield.first;
std::ofstream os(vtu_prefix + GenerateVTUFileName(field_name, myid));
std::string os_str = vtu_prefix + GenerateVTUFileName(field_name, myid);
std::ofstream os(os_str);
MFEM_VERIFY(os.is_open(),
"Failed to open ofstream " << os_str);
qfield.second->SaveVTU(os, pv_data_format, GetCompressionLevel(), field_name);
}
@@ -1000,7 +1008,10 @@ void ParaViewDataCollection::Save()
{
// Create the main PVTU file
{
std::ofstream pvtu_out(vtu_prefix + GeneratePVTUFileName("data"));
std::string os_str = vtu_prefix + GeneratePVTUFileName("data");
std::ofstream pvtu_out(os_str);
MFEM_VERIFY(pvtu_out.is_open(),
"Failed to open ofstream " << os_str);
WritePVTUHeader(pvtu_out);
// Grid function fields and coefficient fields
@@ -1055,8 +1066,10 @@ void ParaViewDataCollection::Save()
const std::string &q_field_name = q_field.first;
std::string q_fname = GeneratePVTUPath() + "/"
+ GeneratePVTUFileName(q_field_name);
std::ofstream pvtu_out(col_path + "/" + q_fname);
std::string os_str = col_path + "/" + q_fname;
std::ofstream pvtu_out(os_str);
MFEM_VERIFY(pvtu_out.is_open(),
"Failed to open ofstream " << os_str);
WritePVTUHeader(pvtu_out);
int vec_dim = q_field.second->GetVDim();
pvtu_out << "<PPointData>\n";
+11 -8
View File
@@ -90,8 +90,8 @@ void map_quadrature_data_to_fields_impl(
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor");
MFEM_ABORT_KERNEL("quadrature data mapping to field is not implemented"
" for this field descriptor");
}
}
@@ -169,8 +169,9 @@ void map_quadrature_data_to_fields_tensor_impl_1d(
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
MFEM_ABORT_KERNEL("quadrature data mapping to field is not implemented"
"for this field descriptor with sum factorization on"
" tensor product elements");
}
}
@@ -306,8 +307,9 @@ void map_quadrature_data_to_fields_tensor_impl_2d(
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
MFEM_ABORT_KERNEL("quadrature data mapping to field is not implemented"
" for this field descriptor with sum factorization on"
" tensor product elements");
}
}
@@ -492,8 +494,9 @@ void map_quadrature_data_to_fields_tensor_impl_3d(
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
MFEM_ABORT_KERNEL("quadrature data mapping to field is not implemented"
" for this field descriptor with sum factorization on"
" tensor product elements");
}
}
+2 -2
View File
@@ -57,7 +57,7 @@ void DGMassApply(const int e,
}
else if (DIM == 3)
{
SmemPAMassApply3D_Element<TD1D,TQ1D,ACCUM>(e, NE, B, pa_data, x, y);
SmemPAMassApply3D_Element<TD1D,TQ1D,NBZ,ACCUM>(e, NE, B, pa_data, x, y);
}
else
{
@@ -387,7 +387,7 @@ void DGMassInverse::DGMassCGIteration(const Vector &b_, Vector &u_) const
static constexpr int NB = Q1D ? Q1D : 1; // block size
mfem::forall_2D(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_2D<NB*NB>(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
{
// Perform change of basis if needed
if (CHANGE_BASIS)
+6 -6
View File
@@ -320,8 +320,8 @@ public:
error estimation procedure where the flux averaging is replaced by a global
L2 projection (requiring a mass matrix solve).
The required BilinearFormIntegrator must implement the methods
ComputeElementFlux() and ComputeFluxEnergy().
The required BilinearFormIntegrator must implement the method
ComputeElementFlux().
Implemented for the parallel case only.
*/
@@ -357,8 +357,8 @@ protected:
public:
/** @brief Construct a new L2ZienkiewiczZhuEstimator object.
@param integ This BilinearFormIntegrator must implement the methods
ComputeElementFlux() and ComputeFluxEnergy().
@param integ This BilinearFormIntegrator must implement the method
ComputeElementFlux().
@param sol The solution field whose error is to be estimated.
@param flux_fes The L2ZienkiewiczZhuEstimator assumes ownership of this
FiniteElementSpace and will call its Update() method when
@@ -382,8 +382,8 @@ public:
{ }
/** @brief Construct a new L2ZienkiewiczZhuEstimator object.
@param integ This BilinearFormIntegrator must implement the methods
ComputeElementFlux() and ComputeFluxEnergy().
@param integ This BilinearFormIntegrator must implement the method
ComputeElementFlux().
@param sol The solution field whose error is to be estimated.
@param flux_fes The L2ZienkiewiczZhuEstimator does NOT assume ownership
of this FiniteElementSpace; will call its Update() method
+122 -44
View File
@@ -663,58 +663,59 @@ const
#pragma omp critical (DofToQuad)
#endif
{
// If the new Dof2Quad is already present, e.g. added in a previous call
// or added by another omp thread, return.
// Do not run if the new Dof2Quad is already present, e.g. added in a
// previous call or added by another omp thread.
if (DofToQuad::SearchArray(dof2quad_array, ir,
DofToQuad::LEXICOGRAPHIC_FULL))
{ return; }
// Undo the native ordering which is what FiniteElement::GetDofToQuad
// returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
DofToQuad::LEXICOGRAPHIC_FULL) == nullptr)
{
for (int d = 0; d < b_dim; d++)
// Undo the native ordering which is what FiniteElement::GetDofToQuad
// returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
{
for (int j = 0; j < dof; j++)
for (int d = 0; d < b_dim; d++)
{
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
for (int j = 0; j < dof; j++)
{
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
}
}
}
}
const int g_dim = [this]()
{
switch (deriv_type)
const int g_dim = [this]()
{
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
switch (deriv_type)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
}
}
}
dof2quad_array.Append(d2q_new);
}
dof2quad_array.Append(d2q_new);
}
}
@@ -1043,9 +1044,50 @@ void VectorFiniteElement::SetDerivMembers()
switch (map_type)
{
case H_DIV:
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
switch (dim)
{
case 3: // div: 3D H_DIV -> 3D INTEGRAL
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
break;
case 2: // div: 2D H_DIV -> 2D INTEGRAL
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
break;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
case H_DIV_R2D:
switch (dim)
{
case 2: // div: 2D H_DIV_R2D -> 2D INTEGRAL
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
break;
case 1: // div: 1D H_DIV_R2D -> 1D INTEGRAL
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
break;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
case H_DIV_R1D:
switch (dim)
{
case 1: // div: 1D H_DIV_R1D -> 1D INTEGRAL
deriv_type = DIV;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
break;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
case H_CURL:
switch (dim)
@@ -1063,13 +1105,49 @@ void VectorFiniteElement::SetDerivMembers()
break;
case 1:
deriv_type = NONE;
deriv_range_type = SCALAR;
deriv_map_type = INTEGRAL;
deriv_range_type = UNKNOWN_RANGE_TYPE;
deriv_map_type = UNKNOWN_MAP_TYPE;
break;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
case H_CURL_R2D:
switch (dim)
{
case 2:
// curl: 2D H_CURL_R2D -> H_DIV_R2D
deriv_type = CURL;
deriv_range_type = VECTOR;
deriv_map_type = H_DIV_R2D;
break;
case 1:
// curl: 1D H_CURL_R2D -> H_DIV_R2D
deriv_type = CURL;
deriv_range_type = VECTOR;
deriv_map_type = H_DIV_R2D;
break;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
case H_CURL_R1D:
switch (dim)
{
case 1:
// curl: 1D H_CURL_R1D -> H_DIV_R1D
deriv_type = CURL;
deriv_range_type = VECTOR;
deriv_map_type = H_DIV_R1D;
break;
case 0:
deriv_type = NONE;
deriv_range_type = UNKNOWN_RANGE_TYPE;
deriv_map_type = UNKNOWN_MAP_TYPE;
default:
MFEM_ABORT("Invalid dimension, Dim = " << dim);
}
break;
default:
MFEM_ABORT("Invalid MapType = " << map_type);
}
+31 -3
View File
@@ -295,10 +295,20 @@ public:
$ u(x) = (1/w) \hat u(\hat x) $ */
H_DIV, /**< For vector fields; preserves surface integrals of the
normal component $ u(x) = (J/w) \hat u(\hat x) $ */
H_CURL /**< For vector fields; preserves line integrals of the
H_CURL, /**< For vector fields; preserves line integrals of the
tangential component
$ u(x) = J^{-t} \hat u(\hat x) $ (square J),
$ u(x) = J(J^t J)^{-1} \hat u(\hat x) $ (general J) */
H_DIV_R2D, /**< For 3-component vector fields in 2D; equivalent to a
direct sum of an H_DIV basis and an INTEGRAL basis */
H_CURL_R2D,/**< For 3-component vector fields in 2D; equivalent to a
direct sum of an H_CURL basis and a VALUE basis */
H_DIV_R1D, /**< For 3-component vector fields in 1D; equivalent to a
direct sum of a VALUE basis and a pair of INTEGRAL
bases */
H_CURL_R1D /**< For 3-component vector fields in 1D; equivalent to a
direct sum of an INTEGRAL basis and a pair of VALUE
bases */
};
/** @brief Enumeration for DerivType: defines which derivative method
@@ -330,12 +340,28 @@ public:
int GetDim() const { return dim; }
/** @brief Returns the vector dimension for vector-valued finite elements,
which is also the dimension of the interpolation operation. */
which is also the dimension of the interpolation operation and the
width of the DenseMatrix argument in
CalcVShape(const IntegrationPoint &ip, DenseMatrix &shape). */
int GetRangeDim() const { return vdim; }
/// Returns the dimension of the curl for vector-valued finite elements.
/** @brief Returns the vector dimension, in physical space, for
vector-valued finite elements, which is also the width of the
DenseMatrix argument in
CalcPhysVShape(ElementTransformation &Trans, DenseMatrix &shape). */
virtual int GetPhysRangeDim(int /* space_dim */) const { return vdim; }
/** Returns the dimension of the curl for vector-valued finite elements,
which is also the width of the DenseMatrix argument in
CalcCurlShape(const IntegrationPoint &ip, DenseMatrix &curl_shape). */
int GetCurlDim() const { return cdim; }
/** Returns the dimension, in physical space, of the curl for vector-valued
finite elements, which is also the width of the DenseMatrix argument in
CalcPhysCurlShape(ElementTransformation &Trans, DenseMatrix &curl_shape).
*/
virtual int GetPhysCurlDim(int /* space_dim */) const { return cdim; }
/// Returns the Geometry::Type of the reference element.
Geometry::Type GetGeomType() const { return geom_type; }
@@ -990,6 +1016,8 @@ protected:
public:
VectorFiniteElement(int D, Geometry::Type G, int Do, int O, int M,
int F = FunctionSpace::Pk);
int GetPhysRangeDim(int space_dim) const override { return space_dim; }
};
/// @brief Class for computing 1D special polynomials and their associated basis
+1 -1
View File
@@ -589,7 +589,7 @@ void H1_TriangleElement::CalcHessian(const IntegrationPoint &ip,
Vector shape_x(p + 1), shape_y(p + 1), shape_l(p + 1);
Vector dshape_x(p + 1), dshape_y(p + 1), dshape_l(p + 1);
Vector ddshape_x(p + 1), ddshape_y(p + 1), ddshape_l(p + 1);
DenseMatrix ddu(dof, dim);
DenseMatrix ddu(dof, (dim*(dim+1))/2);
#endif
poly1d.CalcBasis(p, ip.x, shape_x, dshape_x, ddshape_x);
+4 -4
View File
@@ -2531,7 +2531,7 @@ void ND_FuentesPyramidElement::calcCurlBasis(const int p,
ND_R1D_PointElement::ND_R1D_PointElement(int p)
: VectorFiniteElement(1, Geometry::POINT, 2, p,
H_CURL, FunctionSpace::Pk)
H_CURL_R1D, FunctionSpace::Pk)
{
// VectorFiniteElement::SetDerivMembers doesn't support 0D H_CURL elements
// so we mimic a 1D element and then correct the dimension here.
@@ -2562,7 +2562,7 @@ ND_R1D_SegmentElement::ND_R1D_SegmentElement(const int p,
const int cb_type,
const int ob_type)
: VectorFiniteElement(1, Geometry::SEGMENT, 3 * p + 2, p,
H_CURL, FunctionSpace::Pk),
H_CURL_R1D, FunctionSpace::Pk),
dof2tk(dof),
cbasis1d(poly1d.GetBasis(p, VerifyClosed(cb_type))),
obasis1d(poly1d.GetBasis(p - 1, VerifyOpen(ob_type)))
@@ -2839,7 +2839,7 @@ ND_R2D_SegmentElement::ND_R2D_SegmentElement(const int p,
const int cb_type,
const int ob_type)
: VectorFiniteElement(1, Geometry::SEGMENT, 2 * p + 1, p,
H_CURL, FunctionSpace::Pk),
H_CURL_R2D, FunctionSpace::Pk),
dof2tk(dof),
cbasis1d(poly1d.GetBasis(p, VerifyClosed(cb_type))),
obasis1d(poly1d.GetBasis(p - 1, VerifyOpen(ob_type)))
@@ -3023,7 +3023,7 @@ void ND_R2D_SegmentElement::Project(VectorCoefficient &vc,
ND_R2D_FiniteElement::ND_R2D_FiniteElement(int p, Geometry::Type G, int Do,
const real_t *tk_fe)
: VectorFiniteElement(2, G, Do, p,
H_CURL, FunctionSpace::Pk),
H_CURL_R2D, FunctionSpace::Pk),
tk(tk_fe),
dof_map(dof),
dof2tk(dof)
+6
View File
@@ -663,6 +663,9 @@ public:
const int cb_type = BasisType::GaussLobatto,
const int ob_type = BasisType::GaussLegendre);
int GetPhysRangeDim(int space_dim) const override { return 2; }
int GetPhysCurlDim(int space_dim) const override { return 1; }
void CalcVShape(const IntegrationPoint &ip,
DenseMatrix &shape) const override;
@@ -705,6 +708,9 @@ private:
DenseMatrix &I) const;
public:
int GetPhysRangeDim(int space_dim) const override { return 3; }
int GetPhysCurlDim(int space_dim) const override { return 3; }
using FiniteElement::CalcVShape;
using FiniteElement::CalcPhysCurlShape;
+3 -3
View File
@@ -2006,7 +2006,7 @@ RT_R1D_SegmentElement::RT_R1D_SegmentElement(const int p,
const int cb_type,
const int ob_type)
: VectorFiniteElement(1, Geometry::SEGMENT, 3 * p + 4, p + 1,
H_DIV, FunctionSpace::Pk),
H_DIV_R1D, FunctionSpace::Pk),
dof2nk(dof),
cbasis1d(poly1d.GetBasis(p + 1, VerifyClosed(cb_type))),
obasis1d(poly1d.GetBasis(p, VerifyOpen(ob_type)))
@@ -2281,7 +2281,7 @@ const real_t RT_R2D_SegmentElement::nk[2] = { 0.,1.};
RT_R2D_SegmentElement::RT_R2D_SegmentElement(const int p,
const int ob_type)
: VectorFiniteElement(1, Geometry::SEGMENT, p + 1, p + 1,
H_DIV, FunctionSpace::Pk),
H_DIV_R2D, FunctionSpace::Pk),
dof2nk(dof),
obasis1d(poly1d.GetBasis(p, VerifyOpen(ob_type)))
{
@@ -2392,7 +2392,7 @@ void RT_R2D_SegmentElement::LocalInterpolation(const VectorFiniteElement &cfe,
RT_R2D_FiniteElement::RT_R2D_FiniteElement(int p, Geometry::Type G, int Do,
const real_t *nk_fe)
: VectorFiniteElement(2, G, Do, p + 1,
H_DIV, FunctionSpace::Pk),
H_DIV_R2D, FunctionSpace::Pk),
nk(nk_fe),
dof_map(dof),
dof2nk(dof)
+6
View File
@@ -510,6 +510,9 @@ public:
RT_R2D_SegmentElement(const int p,
const int ob_type = BasisType::GaussLegendre);
int GetPhysRangeDim(int space_dim) const override { return 2; }
int GetPhysCurlDim(int space_dim) const override { return 0; }
void CalcVShape(const IntegrationPoint &ip,
DenseMatrix &shape) const override;
@@ -547,6 +550,9 @@ private:
DenseMatrix &I) const;
public:
int GetPhysRangeDim(int space_dim) const override { return 3; }
int GetPhysCurlDim(int space_dim) const override { return 0; }
using FiniteElement::CalcVShape;
void CalcVShape(ElementTransformation &Trans,
+28 -19
View File
@@ -282,14 +282,7 @@ int FiniteElementSpace::DofToVDof(int dof, int vd, int ndofs_) const
void FiniteElementSpace::AdjustVDofs(Array<int> &vdofs)
{
int n = vdofs.Size(), *vdof = vdofs;
for (int i = 0; i < n; i++)
{
int j;
if ((j = vdof[i]) < 0)
{
vdof[i] = -1-j;
}
}
for (int i = 0; i < n; i++) { vdof[i] = UnsignIndex(vdof[i]); }
}
void FiniteElementSpace::GetElementVDofs(int i, Array<int> &vdofs,
@@ -483,13 +476,14 @@ void FiniteElementSpace::ReorderElementToDofTable()
for (int k = 0, dof_counter = 0; k < nnz; k++)
{
const int sdof = J[k]; // signed dof
const int dof = (sdof < 0) ? -1-sdof : sdof;
const int dof = UnsignIndex(sdof);
int new_dof = dof_marker[dof];
if (new_dof < 0)
{
dof_marker[dof] = new_dof = dof_counter++;
}
J[k] = (sdof < 0) ? -1-new_dof : new_dof; // preserve the sign of sdof
// Preserve the sign of sdof
J[k] = (sdof < 0) ? FlipIndexSign(new_dof) : new_dof;
}
}
@@ -547,7 +541,7 @@ void MarkDofs(const Array<int> &dofs, Array<int> &mark_array)
{
for (auto d : dofs)
{
mark_array[d >= 0 ? d : -1 - d] = -1;
mark_array[UnsignIndex(d)] = -1;
}
}
@@ -931,7 +925,7 @@ void FiniteElementSpace::AddDependencies(
if (std::abs(coef) > 1e-12)
{
const int mdof = master_dofs[j];
if (mdof != sdof && mdof != (-1-sdof))
if (mdof != sdof && mdof != FlipIndexSign(sdof))
{
deps.Add(sdof, mdof, coef);
}
@@ -1024,7 +1018,7 @@ int FiniteElementSpace::GetDegenerateFaceDofs(int index, Array<int> &dofs,
// FiniteElementSpace::AddDependencies.
Array<int> edof;
int order = GetEdgeDofs(-1 - index, edof, variant);
int order = GetEdgeDofs(FlipIndexSign(index), edof, variant);
int nv = fec->DofForGeometry(Geometry::POINT);
int ne = fec->DofForGeometry(Geometry::SEGMENT);
@@ -1710,8 +1704,8 @@ SparseMatrix *FiniteElementSpace::RefinementMatrix_main(
for (int i = 0; i < fine_ldof; i++)
{
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
const int r = DofToVDof(dofs[i], vd);
const int m = UnsignIndex(r);
if (!mark[m])
{
@@ -1772,7 +1766,7 @@ SparseMatrix *FiniteElementSpace::VariableOrderRefinementMatrix(
for (int i = 0; i < fine_ldof; i++)
{
const int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
const int m = UnsignIndex(r);
if (!mark[m])
{
@@ -2482,8 +2476,8 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
{
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
const int r = DofToVDof(dofs[i], vd);
const int m = UnsignIndex(r);
if (is_dg || !mark[m])
{
@@ -3201,7 +3195,7 @@ void FiniteElementSpace::CalcEdgeFaceVarOrders(
else
{
// degenerate face (i.e., edge-face constraint)
slave_orders |= edge_orders[-1 - slave.index];
slave_orders |= edge_orders[FlipIndexSign(slave.index)];
}
}
@@ -3940,6 +3934,16 @@ const FiniteElement *FiniteElementSpace::GetBE(int i) const
return BE;
}
const FiniteElement *FiniteElementSpace::GetTypicalBE() const
{
if (mesh->GetNBE() > 0) { return GetBE(0); }
Geometry::Type geom = mesh->GetTypicalFaceGeometry();
const FiniteElement *be = fec->FiniteElementForGeometry(geom);
MFEM_VERIFY(be != nullptr, "Could not determine a typical BE!");
return be;
}
const FiniteElement *FiniteElementSpace::GetFaceElement(int i) const
{
MFEM_VERIFY(!IsVariableOrder(), "not implemented");
@@ -3970,6 +3974,11 @@ const FiniteElement *FiniteElementSpace::GetFaceElement(int i) const
return fe;
}
const FiniteElement *FiniteElementSpace::GetTypicalFaceElement() const
{
return fec->FiniteElementForGeometry(mesh->GetTypicalFaceGeometry());
}
const FiniteElement *FiniteElementSpace::GetEdgeElement(int i,
int variant) const
{
+14 -2
View File
@@ -839,7 +839,7 @@ public:
Note: For vector-valued elements, the results pads up the range dimension
to the spatial dimension. E.g., consider a stack of 5 vector-valued
elements each representing 2D vectors, living in a 3 dimensional space.
Then this fucntion would give 15, not 10.
Then this function would give 15, not 10.
*/
int GetVectorDim() const;
@@ -1150,7 +1150,7 @@ public:
/// Helper to return the DOF associated with a sign encoded DOF
static inline int DecodeDof(int dof)
{ return (dof >= 0) ? dof : (-1 - dof); }
{ return UnsignIndex(dof); }
/// Helper to determine the DOF and sign of a sign encoded DOF
static inline int DecodeDof(int dof, real_t& sign)
@@ -1323,12 +1323,24 @@ public:
associated with i'th boundary face in the mesh object. */
const FiniteElement *GetBE(int i) const;
/// @brief Return a "typical" boundary element.
///
/// This can be used in situations where the local mesh partition may be
/// empty.
const FiniteElement *GetTypicalBE() const;
/** @brief Returns pointer to the FiniteElement in the FiniteElementCollection
associated with i'th face in the mesh object. Faces in this case refer
to the MESHDIM-1 primitive so in 2D they are segments and in 1D they are
points.*/
const FiniteElement *GetFaceElement(int i) const;
/// @brief Return a "typical" face element.
///
/// This can be used in situations where the local mesh partition may be
/// empty.
const FiniteElement *GetTypicalFaceElement() const;
/** @brief Returns pointer to the FiniteElement in the FiniteElementCollection
associated with i'th edge in the mesh object. */
const FiniteElement *GetEdgeElement(int i, int variant = 0) const;
+642 -73
View File
@@ -30,6 +30,7 @@
#include <cmath>
#include <iostream>
#include <algorithm>
#include <queue>
namespace mfem
{
@@ -344,27 +345,6 @@ void GridFunction::ComputeFlux(BilinearFormIntegrator &blfi,
}
}
int GridFunction::VectorDim() const
{
const FiniteElement *fe = fes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return fes->GetVDim();
}
return fes->GetVDim()*std::max(fes->GetMesh()->SpaceDimension(),
fe->GetRangeDim());
}
int GridFunction::CurlDim() const
{
const FiniteElement *fe = fes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return 2 * fes->GetMesh()->SpaceDimension() - 3;
}
return fes->GetVDim()*fe->GetCurlDim();
}
void GridFunction::GetTrueDofs(Vector &tv) const
{
const SparseMatrix *R = fes->GetRestrictionMatrix();
@@ -2049,6 +2029,18 @@ void GridFunction::AccumulateAndCountBdrValues(
Coefficient *coeff[], VectorCoefficient *vcoeff, const Array<int> &attr,
Array<int> &values_counter)
{
if (vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
"vcoeff vdim != fes VDim");
MFEM_VERIFY(fes->GetTypicalBE()->GetMapType() == FiniteElement::VALUE &&
fes->GetTypicalBE()->GetRangeType() ==
FiniteElement::SCALAR,
"Can only call ProjectBdrCoefficient on scalar value-type "
"boundary elements. "
"Did you intended to call ProjectBdrCoefficientNormal or "
"ProjectBdrCoefficientTangent for vector finite elements?");
}
Array<int> vdofs;
Vector vc;
@@ -2201,6 +2193,9 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
VectorCoefficient &vcoeff, const Array<int> &bdr_attr,
Array<int> &values_counter)
{
MFEM_VERIFY(fes->GetTypicalBE()->GetPhysRangeDim(
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
"vcoeff vdim != PhysRangeDim");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
@@ -2354,6 +2349,9 @@ void GridFunction::ProjectDeltaCoefficient(DeltaCoefficient &delta_coeff,
void GridFunction::ProjectCoefficient(Coefficient &coeff, ProjectType type)
{
MFEM_VERIFY(
VectorDim() == 1,
"Cannot project scalar Coefficient onto vector GridFunction");
DeltaCoefficient *delta_c = dynamic_cast<DeltaCoefficient *>(&coeff);
DofTransformation doftrans;
Array<int> vdofs;
@@ -2629,6 +2627,7 @@ void GridFunction::ProjectCoefficient(
void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
ProjectType type)
{
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
Array<int> vdofs;
Vector vals;
DofTransformation doftrans;
@@ -2944,6 +2943,7 @@ void GridFunction::ProjectCoefficientElementL2(VectorCoefficient &vcoeff)
void GridFunction::ProjectCoefficient(
VectorCoefficient &vcoeff, Array<int> &dofs)
{
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
int el = -1;
ElementTransformation *T = NULL;
const FiniteElement *fe = NULL;
@@ -2973,6 +2973,7 @@ void GridFunction::ProjectCoefficient(
void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff, int attribute)
{
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
int i;
Array<int> vdofs;
Vector vals;
@@ -3029,9 +3030,14 @@ void GridFunction::ProjectCoefficient(Coefficient *coeff[])
}
}
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
Array<int> &dof_attr)
void GridFunction::ProjectDiscCoefficient(
std::variant<Coefficient*, VectorCoefficient*> coeff, Array<int> &dof_attr)
{
std::visit([&](auto* c)
{
MFEM_VERIFY(VectorDim() == c->GetVDim(), "coeff vdim != VectorDim()");
}, coeff);
Array<int> vdofs;
Vector vals;
@@ -3045,7 +3051,10 @@ void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
{
fes->GetElementVDofs(i, vdofs);
vals.SetSize(vdofs.Size());
fes->GetFE(i)->Project(coeff, *fes->GetElementTransformation(i), vals);
std::visit([&](auto* c)
{
fes->GetFE(i)->Project(*c, *fes->GetElementTransformation(i), vals);
}, coeff);
// the values in shared dofs are determined from the element with maximal
// attribute
@@ -3061,17 +3070,15 @@ void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
}
}
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
{
Array<int> dof_attr;
ProjectDiscCoefficient(coeff, dof_attr);
}
void GridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
{
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
MFEM_VERIFY(
VectorDim() == 1,
"Cannot project a scalar coefficient onto a vector GridFunction");
Array<int> zones_per_vdof;
AccumulateAndCountZones(coeff, type, zones_per_vdof);
@@ -3081,6 +3088,7 @@ void GridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
AvgType type)
{
MFEM_VERIFY(VectorDim() == coeff.GetVDim(), "coeff vdim != VectorDim()");
Array<int> zones_per_vdof;
AccumulateAndCountZones(coeff, type, zones_per_vdof);
@@ -3136,52 +3144,33 @@ void GridFunction::ProjectBdrCoefficient(Coefficient *coeff[],
}
void GridFunction::ProjectBdrCoefficientNormal(
VectorCoefficient &vcoeff, const Array<int> &bdr_attr)
Coefficient *coeff, VectorCoefficient *vcoeff, const Array<int> &bdr_attr)
{
#if 0
// implementation for the case when the face dofs are integrals of the
// normal component.
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
int dim = vcoeff.GetVDim();
Vector vc(dim), nor(dim), lvec, shape;
for (int i = 0; i < fes->GetNBE(); i++)
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalBE()->GetRangeType() == FiniteElement::SCALAR &&
fes->GetTypicalBE()->GetMapType() == FiniteElement::INTEGRAL,
"Not an RT FE space!");
if (vcoeff)
{
if (bdr_attr[fes->GetBdrAttribute(i)-1] == 0)
{
continue;
}
fe = fes->GetBE(i);
T = fes->GetBdrElementTransformation(i);
int intorder = 2*fe->GetOrder(); // !!!
const IntegrationRule &ir = IntRules.Get(fe->GetGeomType(), intorder);
int nd = fe->GetDof();
lvec.SetSize(nd);
shape.SetSize(nd);
lvec = 0.0;
for (int j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
vcoeff.Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
fe->CalcShape(ip, shape);
lvec.Add(ip.weight * (vc * nor), shape);
}
fes->GetBdrElementDofs(i, dofs);
SetSubVector(dofs, lvec);
MFEM_VERIFY(vcoeff->GetVDim() == fes->GetMesh()->SpaceDimension(),
"vcoeff vdim (" << vcoeff->GetVDim()
<< ") != SpaceDimension ("
<< fes->GetMesh()->SpaceDimension() << ")");
}
#else
// implementation for the case when the face dofs are scaled point
// values of the normal component.
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
int dim = vcoeff.GetVDim();
Vector vc(dim), nor(dim), lvec;
Vector vc, nor, lvec;
DofTransformation doftrans;
if (vcoeff)
{
const int dim = vcoeff->GetVDim();
vc.SetSize(dim);
nor.SetSize(dim);
}
for (int i = 0; i < fes->GetNBE(); i++)
{
@@ -3197,15 +3186,22 @@ void GridFunction::ProjectBdrCoefficientNormal(
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
vcoeff.Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
lvec(j) = (vc * nor);
if (coeff)
{
const real_t c = coeff->Eval(*T, ip);
lvec(j) = c * T->Weight();
}
else if (vcoeff)
{
vcoeff->Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
lvec(j) = (vc * nor);
}
}
fes->GetBdrElementDofs(i, dofs, doftrans);
doftrans.TransformPrimal(lvec);
SetSubVector(dofs, lvec);
}
#endif
}
void GridFunction::ProjectBdrCoefficientTangent(
@@ -5006,6 +5002,14 @@ real_t ExtrudeCoefficient::Eval(ElementTransformation &T,
return sol_in.Eval(*T_in, ip);
}
void VectorExtrudeCoefficient::Eval(Vector &v, ElementTransformation &T,
const IntegrationPoint &ip)
{
ElementTransformation *T_in =
mesh_in->GetElementTransformation(T.ElementNo / n);
T_in->SetIntPoint(&ip);
sol_in.Eval(v, *T_in, ip);
}
GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
GridFunction *sol, const int ny)
@@ -5056,10 +5060,17 @@ GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
return NULL;
}
FiniteElementSpace *solfes2d;
// assuming sol is scalar
solfes2d = new FiniteElementSpace(mesh2d, solfec2d);
const int vdim = sol->FESpace()->GetVDim();
solfes2d = new FiniteElementSpace(mesh2d, solfec2d, vdim);
sol2d = new GridFunction(solfes2d);
sol2d->MakeOwner(solfec2d);
if (vdim > 1)
{
VectorGridFunctionCoefficient vcsol(sol);
VectorExtrudeCoefficient vc2d(mesh, vcsol, ny);
sol2d->ProjectCoefficient(vc2d);
}
else
{
GridFunctionCoefficient csol(sol);
ExtrudeCoefficient c2d(mesh, csol, ny);
@@ -5117,6 +5128,103 @@ void GridFunction::GetElementBoundsAtControlPoints(const int elem,
}
}
void GridFunction::GetElementBoundsAtControlPoints(const int elem,
const PLBound &plb,
const Vector &ref_range,
const int vdim,
Vector &lower, Vector &upper,
Vector &control_pos) const
{
const FiniteElement *fe = fes->GetFE(elem);
const IntegrationRule ir_in = fe->GetNodes();
IntegrationRule ir_new(ir_in.GetNPoints());
const int dim = fes->GetMesh()->Dimension();
const L2_FECollection *l2fec = dynamic_cast<const L2_FECollection *>
(fes->FEColl());
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fe);
MFEM_VERIFY(tbe != NULL, "TensorBasis FiniteElement expected.");
const Array<int> &dof_map = tbe->GetDofMap();
bool lexico = (dof_map.Size() == 0);
bool bern = (tbe->GetBasisType() == BasisType::Positive);
bool h1 = (l2fec == nullptr);
Vector loc_data; // gridfunction values
// Construct an integration rule to evaluate the gridfunction in
// subinterval.
for (int i = 0; i < ir_in.GetNPoints(); i++)
{
IntegrationPoint &ip_new = ir_new.IntPoint(i);
const IntegrationPoint &ip_old =
ir_in.IntPoint((lexico || bern) ? i : dof_map[i]);
Vector ip_coord(dim);
ip_old.Get(ip_coord.GetData(), dim);
for (int d = 0; d < dim; d++)
{
ip_coord(d) = ref_range(d) +
(ref_range(dim+d) - ref_range(d)) * ip_coord(d);
}
ip_new.Set(ip_coord.GetData(), dim);
}
GetValues(elem, ir_new, loc_data, vdim);
// At this point, the loc_data contains function values ordered
// lexicographically, unless we are using Bernstein bases.
// For Bernstein, we need to project and get coefficients first.
// For bernstein, we get coefficients corresponding to these function values
if (bern)
{
int bt = 4; // BasisType::ClosedUniform
int o = fe->GetOrder();
DenseMatrix projmat;
NodalTensorFiniteElement *ntfe = nullptr;
if (dim == 1)
{
if (h1) { ntfe = new H1_SegmentElement(o, bt); }
else { ntfe = new L2_SegmentElement(o, bt); }
}
else if (dim == 2)
{
if (h1) { ntfe = new H1_QuadrilateralElement(o, bt); }
else { ntfe = new L2_QuadrilateralElement(o, bt); }
}
else if (dim == 3)
{
if (h1) { ntfe = new H1_HexahedronElement(o, bt); }
else { ntfe = new L2_HexahedronElement(o, bt); }
}
// projection matrix from H1 to Positive
ElementTransformation *eltran = fes->GetElementTransformation(elem);
fe->Project(*ntfe, *eltran, projmat);
Vector loc_data_temp(loc_data.Size());
projmat.Mult(loc_data, loc_data_temp);
for (int i = 0; i < dof_map.Size(); i++)
{
loc_data(i) = loc_data_temp(dof_map[i]);
}
if (dof_map.Size() == 0) { loc_data = loc_data_temp; }
delete ntfe;
}
// Get bounds at control points
plb.GetNDBounds(dim, loc_data, lower, upper);
// Save control point positions
int ncp = plb.GetNControlPoints();
control_pos.SetSize(dim * ncp);
const Vector control_pos_1D = plb.GetControlPoints();
for (int i = 0; i < ncp; i++)
{
for (int d = 0; d < dim; d++)
{
control_pos(i + d*ncp) =
ref_range(d) + (ref_range(dim+d)-ref_range(d))*control_pos_1D(i);
}
}
}
void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim) const
@@ -5197,6 +5305,467 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
return plb;
}
struct IntervalNode
{
real_t val_min;
real_t val_max;
Array<IntervalNode *> child;
IntervalNode(real_t vmin, real_t vmax)
: val_min(vmin), val_max(vmax)
{
child.SetSize(0);
}
void AddChild(IntervalNode *ch) { child.Append(ch); }
real_t GetChildMinLower()
{
if (child.Size() == 0)
{
return val_min;
}
real_t valmin = numeric_limits<real_t>::max();
for (int i = 0; i < child.Size(); i++)
{
real_t candidate = child[i]->GetChildMinLower();
valmin = std::min(valmin, candidate);
}
return valmin;
}
real_t GetChildMinUpper()
{
if (child.Size() == 0)
{
return val_max;
}
real_t valmax = numeric_limits<real_t>::max();
for (int i = 0; i < child.Size(); i++)
{
real_t candidate = child[i]->GetChildMinUpper();
valmax = std::min(valmax, candidate);
}
return valmax;
}
real_t GetChildMaxLower()
{
if (child.Size() == 0)
{
return val_min;
}
real_t valmin = numeric_limits<real_t>::lowest();
for (int i = 0; i < child.Size(); i++)
{
real_t candidate = child[i]->GetChildMaxLower();
valmin = std::max(valmin, candidate);
}
return valmin;
}
real_t GetChildMaxUpper()
{
if (child.Size() == 0)
{
return val_max;
}
real_t valmax = numeric_limits<real_t>::lowest();
for (int i = 0; i < child.Size(); i++)
{
real_t candidate = child[i]->GetChildMaxUpper();
valmax = std::max(valmax, candidate);
}
return valmax;
}
void DeleteChildren()
{
for (int i = 0; i < child.Size(); i++)
{
child[i]->DeleteChildren();
delete child[i];
}
child.SetSize(0);
}
};
struct SearchInterval
{
Vector ref_range;
int depth;
IntervalNode *node;
SearchInterval(const Vector &ref_range_in, int d, IntervalNode *n)
: ref_range(ref_range_in), depth(d), node(n)
{ }
};
struct IntervalCompareMin
{
bool operator()(const SearchInterval *a, const SearchInterval *b) const
{
return a->node->val_min > b->node->val_min;
}
};
struct IntervalCompareMax
{
bool operator()(const SearchInterval *a, const SearchInterval *b) const
{
return a->node->val_max < b->node->val_max;
}
};
std::pair<real_t, real_t> GridFunction::EstimateFunctionMinimum(
const int elem, const PLBound &plb, const int vdim,
const int max_depth, const real_t tol) const
{
real_t min_threshold = std::numeric_limits<real_t>::max();
return EstimateFunctionMinimum(elem, plb, vdim, max_depth, tol,
min_threshold);
}
std::pair<real_t, real_t> GridFunction::EstimateFunctionMinimum(
const int elem, const PLBound &plb, const int vdim,
const int max_depth, const real_t tol, real_t &min_threshold) const
{
const int dim = this->FESpace()->GetMesh()->Dimension();
const int ncp = plb.GetNControlPoints();
Vector pos_range(2*dim); pos_range = 0.0;
for (int d = 0; d < dim; d++) { pos_range(d+dim) = 1.0; }
Vector lower, upper, cp_ref_loc;
GetElementBoundsAtControlPoints(elem, plb, lower, upper, vdim);
real_t val_min = lower.Min();
real_t val_max = upper.Min();
min_threshold = std::min(min_threshold, val_max);
// Pruning: if the element's lower bound is greater than the current global
// upper bound, this element cannot contain the global minimum.
if (val_min >= min_threshold)
{
return std::make_pair(val_min, val_max);
}
if (val_min == val_max || max_depth == 0)
{
min_threshold = std::min(min_threshold, val_min);
return std::make_pair(val_min, val_max);
}
real_t abs_tol = tol*(val_max-val_min);
IntervalNode *initial_node = new IntervalNode(val_min, val_max);
SearchInterval *initial_interval = new SearchInterval(pos_range, 0,
initial_node);
std::priority_queue<SearchInterval*,
std::vector<SearchInterval*>, IntervalCompareMin> pq;
pq.push(initial_interval);
real_t min_upper_bound = upper.Min();
real_t min_lower_bound = lower.Min();
while (!pq.empty())
{
SearchInterval *current = pq.top();
pq.pop();
int curr_depth = current->depth;
// Reached max depth or this interval cannot contain the global minimum
if (current->node->val_min >= min_threshold || curr_depth >= max_depth)
{
delete current;
continue;
}
min_lower_bound = initial_node->GetChildMinLower();
if (min_upper_bound - min_lower_bound < abs_tol)
{
delete current;
break;
}
// Subdivide the interval and get bounds on it
GetElementBoundsAtControlPoints(elem, plb, current->ref_range,
vdim, lower, upper, cp_ref_loc);
// process the bounds and create sub-intervals
for (int k = 0; k < (dim == 3 ? ncp-1 : 1); k++)
{
for (int j = 0; j < (dim >= 2 ? ncp-1 : 1); j++)
{
for (int i = 0; i < ncp-1; i++)
{
real_t lv = 0.0, uv = 0.0;
if (dim == 1)
{
lv = std::min(lower(i), lower(i+1));
uv = std::min(upper(i), upper(i+1));
}
else if (dim == 2)
{
lv = std::min({lower(i + j*ncp), lower((i+1) + j*ncp),
lower(i + (j+1)*ncp),
lower((i+1) + (j+1)*ncp)});
uv = std::min({upper(i + j*ncp), upper((i+1) + j*ncp),
upper(i + (j+1)*ncp),
upper((i+1) + (j+1)*ncp)});
}
else if (dim == 3)
{
lv = std::min({lower(i + j*ncp + k*ncp*ncp),
lower((i+1) + j*ncp + k*ncp*ncp),
lower(i + (j+1)*ncp + k*ncp*ncp),
lower((i+1) + (j+1)*ncp + k*ncp*ncp),
lower(i + j*ncp + (k+1)*ncp*ncp),
lower((i+1) + j*ncp + (k+1)*ncp*ncp),
lower(i + (j+1)*ncp + (k+1)*ncp*ncp),
lower((i+1) + (j+1)*ncp + (k+1)*ncp*ncp)});
uv = std::min({upper(i + j*ncp + k*ncp*ncp),
upper((i+1) + j*ncp + k*ncp*ncp),
upper(i + (j+1)*ncp + k*ncp*ncp),
upper((i+1) + (j+1)*ncp + k*ncp*ncp),
upper(i + j*ncp + (k+1)*ncp*ncp),
upper((i+1) + j*ncp + (k+1)*ncp*ncp),
upper(i + (j+1)*ncp + (k+1)*ncp*ncp),
upper((i+1) + (j+1)*ncp + (k+1)*ncp*ncp)});
}
IntervalNode *child_node = new IntervalNode(lv, uv);
current->node->AddChild(child_node);
if (lv < min_threshold)
{
min_upper_bound = std::min(min_upper_bound, uv);
min_threshold = std::min(min_threshold, uv);
if (curr_depth < max_depth)
{
pos_range(0) = cp_ref_loc(i);
pos_range(0+dim) = cp_ref_loc(i+1);
if (dim >= 2)
{
pos_range(1) = cp_ref_loc(ncp + j);
pos_range(1+dim) = cp_ref_loc(ncp + j+1);
}
if (dim == 3)
{
pos_range(2) = cp_ref_loc(2*ncp + k);
pos_range(2+dim) = cp_ref_loc(2*ncp + k+1);
}
SearchInterval *child_interval =
new SearchInterval(pos_range, curr_depth + 1,
child_node);
pq.push(child_interval);
}
}
}
}
}
delete current;
}
// clean up remaining intervals in queue
while (!pq.empty())
{
delete pq.top();
pq.pop();
}
min_lower_bound = initial_node->GetChildMinLower();
initial_node->DeleteChildren();
delete initial_node;
min_threshold = std::min(min_threshold, min_lower_bound);
return std::make_pair(min_lower_bound, min_upper_bound);
}
std::pair<real_t, real_t> GridFunction::EstimateFunctionMaximum(
const int elem, const PLBound &plb, const int vdim,
const int max_depth, const real_t tol) const
{
real_t max_threshold = std::numeric_limits<real_t>::lowest();
return EstimateFunctionMaximum(elem, plb, vdim, max_depth, tol,
max_threshold);
}
std::pair<real_t, real_t> GridFunction::EstimateFunctionMaximum(
const int elem, const PLBound &plb, const int vdim,
const int max_depth, const real_t tol, real_t &max_threshold) const
{
const int dim = this->FESpace()->GetMesh()->Dimension();
const int ncp = plb.GetNControlPoints();
Vector pos_range(2*dim); pos_range = 0.0;
for (int d = 0; d < dim; d++) { pos_range(d+dim) = 1.0; }
Vector lower, upper, cp_ref_loc;
GetElementBoundsAtControlPoints(elem, plb, lower, upper, vdim);
real_t val_min = lower.Max();
real_t val_max = upper.Max();
max_threshold = std::max(max_threshold, val_min);
// Pruning: if the element's upper bound is less than the current global
// lower bound, this element cannot contain the global maximum.
if (val_max <= max_threshold)
{
return std::make_pair(val_min, val_max);
}
if (val_min == val_max || max_depth == 0)
{
max_threshold = std::max(max_threshold, val_max);
return std::make_pair(val_min, val_max);
}
real_t abs_tol = tol*(val_max-val_min);
IntervalNode *initial_node = new IntervalNode(val_min, val_max);
SearchInterval *initial_interval = new SearchInterval(pos_range, 0,
initial_node);
std::priority_queue<SearchInterval*,
std::vector<SearchInterval*>, IntervalCompareMax> pq;
pq.push(initial_interval);
real_t max_lower_bound = val_min;
real_t max_upper_bound = val_max;
while (!pq.empty())
{
SearchInterval *current = pq.top();
pq.pop();
int curr_depth = current->depth;
// Reached max depth or this interval cannot contain the global maximum.
if (current->node->val_max <= max_threshold || curr_depth >= max_depth)
{
delete current;
continue;
}
max_upper_bound = initial_node->GetChildMaxUpper();
if (max_upper_bound - max_lower_bound < abs_tol)
{
delete current;
break;
}
// Subdivide the interval and get bounds on it
GetElementBoundsAtControlPoints(elem, plb, current->ref_range,
vdim, lower, upper, cp_ref_loc);
// process the bounds and create sub-intervals
for (int k = 0; k < (dim == 3 ? ncp-1 : 1); k++)
{
for (int j = 0; j < (dim >= 2 ? ncp-1 : 1); j++)
{
for (int i = 0; i < ncp-1; i++)
{
real_t lv = 0.0, uv = 0.0;
if (dim == 1)
{
lv = std::max(lower(i), lower(i+1));
uv = std::max(upper(i), upper(i+1));
}
else if (dim == 2)
{
lv = std::max({lower(i + j*ncp), lower((i+1) + j*ncp),
lower(i + (j+1)*ncp),
lower((i+1) + (j+1)*ncp)});
uv = std::max({upper(i + j*ncp), upper((i+1) + j*ncp),
upper(i + (j+1)*ncp),
upper((i+1) + (j+1)*ncp)});
}
else if (dim == 3)
{
lv = std::max({lower(i + j*ncp + k*ncp*ncp),
lower((i+1) + j*ncp + k*ncp*ncp),
lower(i + (j+1)*ncp + k*ncp*ncp),
lower((i+1) + (j+1)*ncp + k*ncp*ncp),
lower(i + j*ncp + (k+1)*ncp*ncp),
lower((i+1) + j*ncp + (k+1)*ncp*ncp),
lower(i + (j+1)*ncp + (k+1)*ncp*ncp),
lower((i+1) + (j+1)*ncp + (k+1)*ncp*ncp)});
uv = std::max({upper(i + j*ncp + k*ncp*ncp),
upper((i+1) + j*ncp + k*ncp*ncp),
upper(i + (j+1)*ncp + k*ncp*ncp),
upper((i+1) + (j+1)*ncp + k*ncp*ncp),
upper(i + j*ncp + (k+1)*ncp*ncp),
upper((i+1) + j*ncp + (k+1)*ncp*ncp),
upper(i + (j+1)*ncp + (k+1)*ncp*ncp),
upper((i+1) + (j+1)*ncp + (k+1)*ncp*ncp)});
}
IntervalNode *child_node = new IntervalNode(lv, uv);
current->node->AddChild(child_node);
if (uv > max_threshold)
{
max_lower_bound = std::max(max_lower_bound, lv);
max_threshold = std::max(max_threshold, lv);
if (curr_depth < max_depth)
{
pos_range(0) = cp_ref_loc(i);
pos_range(0+dim) = cp_ref_loc(i+1);
if (dim >= 2)
{
pos_range(1) = cp_ref_loc(ncp + j);
pos_range(1+dim) = cp_ref_loc(ncp + j+1);
}
if (dim == 3)
{
pos_range(2) = cp_ref_loc(2*ncp + k);
pos_range(2+dim) = cp_ref_loc(2*ncp + k+1);
}
SearchInterval *child_interval =
new SearchInterval(pos_range, curr_depth + 1,
child_node);
pq.push(child_interval);
}
}
}
}
}
delete current;
}
// clean up remaining intervals in queue
while (!pq.empty())
{
delete pq.top();
pq.pop();
}
max_upper_bound = initial_node->GetChildMaxUpper();
initial_node->DeleteChildren();
delete initial_node;
max_threshold = std::max(max_threshold, max_upper_bound);
return std::make_pair(max_lower_bound, max_upper_bound);
}
std::pair<real_t, real_t> GridFunction::EstimateFunctionMinimum(
const int vdim, const PLBound &plb, const int max_depth,
const real_t tol) const
{
real_t global_min_lower = std::numeric_limits<real_t>::max();
real_t global_min_upper = std::numeric_limits<real_t>::max();
for (int i = 0; i < fes->GetNE(); i++)
{
std::pair<real_t, real_t> min_pair =
EstimateFunctionMinimum(i, plb, vdim, max_depth, tol,
global_min_lower);
global_min_upper = std::min(global_min_upper, min_pair.second);
}
return std::make_pair(global_min_lower, global_min_upper);
}
std::pair<real_t, real_t> GridFunction::EstimateFunctionMaximum(
const int vdim, const PLBound &plb, const int max_depth,
const real_t tol) const
{
real_t global_max_lower = std::numeric_limits<real_t>::lowest();
real_t global_max_upper = std::numeric_limits<real_t>::lowest();
for (int i = 0; i < fes->GetNE(); i++)
{
std::pair<real_t, real_t> max_pair =
EstimateFunctionMaximum(i, plb, vdim, max_depth, tol,
global_max_upper);
global_max_lower = std::max(global_max_lower, max_pair.first);
}
return std::make_pair(global_max_lower, global_max_upper);
}
}
+206 -24
View File
@@ -23,6 +23,7 @@
#include <limits>
#include <ostream>
#include <string>
#include <variant>
namespace mfem
{
@@ -79,10 +80,18 @@ protected:
bool wcoef,
int subdomain);
/** Project a discontinuous vector coefficient in a continuous space and
return in dof_attr the maximal attribute of the elements containing each
degree of freedom. */
void ProjectDiscCoefficient(VectorCoefficient &coeff, Array<int> &dof_attr);
/** @brief Project a discontinuous (vector) coefficient as a grid function on
a continuous finite element space. Return in dof_attr the maximal
attribute of the elements containing each degree of freedom. */
virtual void ProjectDiscCoefficient(
std::variant<Coefficient*, VectorCoefficient*> coeff, Array<int> &dof_attr);
/** @brief Project a discontinuous (vector) coefficient as a grid function on
a continuous finite element space. The values in shared dofs are
determined from the element with maximal attribute. */
virtual void ProjectDiscCoefficient(
std::variant<Coefficient*, VectorCoefficient*> coeff)
{ Array<int> dof_attr; ProjectDiscCoefficient(coeff, dof_attr); };
/** Helper function for ProjectCoefficientElementL2 */
void ProjectCoefficientElementL2_(Coefficient &coeff, Vector &sol, Vector &Va);
@@ -150,11 +159,13 @@ public:
FiniteElementCollection *OwnFEC() { return fec_owned; }
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the underlying #fes
int VectorDim() const;
/** @brief Shortcut for calling FiniteElementSpace::GetVectorDim() on the
underlying #fes */
int VectorDim() const { return fes->GetVectorDim(); }
/// Shortcut for calling FiniteElementSpace::GetCurlDim() on the underlying #fes
int CurlDim() const;
/** @brief Shortcut for calling FiniteElementSpace::GetCurlDim() on the
underlying #fes */
int CurlDim() const { return fes->GetCurlDim(); }
/// Read only access to the (optional) internal true-dof Vector.
const Vector &GetTrueVector() const
@@ -513,10 +524,17 @@ public:
but using an array of scalar coefficients for each component. */
void ProjectCoefficient(Coefficient *coeff[]);
/** @brief Project a discontinuous coefficient as a grid function on
a continuous finite element space. The values in shared dofs are
determined from the element with maximal attribute. */
virtual void ProjectDiscCoefficient(Coefficient &coeff)
{ ProjectDiscCoefficient(&coeff); }
/** @brief Project a discontinuous vector coefficient as a grid function on
a continuous finite element space. The values in shared dofs are
determined from the element with maximal attribute. */
virtual void ProjectDiscCoefficient(VectorCoefficient &coeff);
virtual void ProjectDiscCoefficient(VectorCoefficient &coeff)
{ ProjectDiscCoefficient(&coeff); }
enum AvgType {ARITHMETIC, HARMONIC};
/** @brief Projects a discontinuous coefficient so that the values in shared
@@ -532,6 +550,9 @@ public:
std::unique_ptr<GridFunction> ProlongateToMaxOrder() const;
protected:
void ProjectBdrCoefficientNormal(Coefficient *coeff, VectorCoefficient *vcoeff,
const Array<int> &attr);
/** @brief Accumulates (depending on @a type) the values of @a coeff at all
shared vdofs and counts in how many zones each vdof appears. */
void AccumulateAndCountZones(Coefficient &coeff, AvgType type,
@@ -564,6 +585,70 @@ protected:
/// P-refinement version of Update().
void UpdatePRef();
/** @brief Estimate the minimum value of the GridFunction in element @a elem
* if it is below a certain @a min_threshold.
*
* @details For a given element \p elem and grid function component \p vdim
* an estimate of the function minimum is the minimum of the piecewise
* linear lower bound obtained using the given PLBound object. The actual
* minimum is between [minimum lower bound, minimum upper bound]. We
* improve the estimate of the function minimum by recursively
* subdividing the interval with the lowest lower bound, and computing
* bounds on the sub-intervals.
* This process continues until (i) the maximum recursion depth is reached
* or (ii) the difference between the minimum upper bound and minimum lower
* bound is less than a certain tolerance (\p tol * [initial maximum
* upper bound - initial minimum lower bound]).
* The function also terminates if the lowest minima estimate is found
* to be above the given threshold \p min_threshold. This is useful when
* we are interested in computing the global minimum of the function
* over all elements. In this case we can reject elements where the lowest
* bound is above the current global minimum. In case the function
* minimum on the element is below the global minimum, we update
* \p min_threshold.
*
* We return a pair of values that bracket the actual minimum, i.e.
* [min_lower_bound, min_upper_bound].
*/
std::pair<real_t,real_t> EstimateFunctionMinimum(const int elem,
const PLBound &plb,
const int vdim,
const int max_depth,
const real_t tol,
real_t &min_threshold)const;
/** @brief Estimate the maximum value of the GridFunction in element @a elem
* if it is below a certain @a max_threshold.
*
* @details For a given element \p elem and grid function component \p vdim
* an estimate of the function maximum is the maximum of the piecewise
* linear upper bound obtained using the given PLBound object. The actual
* maximum is between [maximum lower bound, maximum upper bound]. We
* improve the estimate of the function maximum by recursively
* subdividing the interval with the highest upper bound, and computing
* bounds on the sub-intervals.
* This process continues until (i) the maximum recursion depth is reached
* or (ii) the difference between the maximum upper bound and maximum lower
* bound is less than a certain tolerance (\p tol * [initial maximum
* upper bound - initial maximum lower bound]).
* The function also terminates if the highest maxima estimate is found
* to be below the given threshold \p max_threshold. This is useful when
* we are interested in computing the global maximum of the function
* over all elements. In this case we can reject elements where the upper
* bound is below the current global maximum. In case the function
* maximum on the element is above the global maximum, we update
* \p max_threshold.
*
* We return a pair of values that bracket the actual maximum, i.e.
* [max_lower_bound, max_upper_bound].
*/
std::pair<real_t,real_t> EstimateFunctionMaximum(const int elem,
const PLBound &plb,
const int vdim,
const int max_depth,
const real_t tol,
real_t &max_threshold)const;
public:
/** @brief For each vdof, counts how many elements contain the vdof,
as containment is determined by FiniteElementSpace::GetElementVDofs(). */
@@ -592,15 +677,26 @@ public:
virtual void ProjectBdrCoefficient(Coefficient *coeff[],
const Array<int> &attr);
/** Project the normal component of the given VectorCoefficient on
the boundary. Only boundary attributes that are marked in
'bdr_attr' are projected. Assumes RT-type VectorFE GridFunction. */
/** @brief Project the normal component of the given VectorCoefficient on
the boundary. */
/** Only boundary attributes that are marked in @a bdr_attr are
projected. Assumes RT-type vector finite element GridFunction. */
void ProjectBdrCoefficientNormal(VectorCoefficient &vcoeff,
const Array<int> &bdr_attr);
const Array<int> &bdr_attr)
{ ProjectBdrCoefficientNormal(NULL, &vcoeff, bdr_attr); }
/** @brief Project the given Coefficient in the normal direction on the
boundary. */
/** Only boundary attributes that are marked in @a bdr_attr are projected.
Assumes RT-type vector finite element GridFunction. */
void ProjectBdrCoefficientNormal(Coefficient &coeff,
const Array<int> &bdr_attr)
{ ProjectBdrCoefficientNormal(&coeff, NULL, bdr_attr); }
/** @brief Project the tangential components of the given VectorCoefficient
on the boundary. Only boundary attributes that are marked in @a bdr_attr
are projected. Assumes ND-type VectorFE GridFunction. */
on the boundary. */
/** Only boundary attributes that are marked in @a bdr_attr
are projected. Assumes ND-type vector finite element GridFunction. */
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
const Array<int> &bdr_attr);
@@ -1662,21 +1758,21 @@ public:
*/
///@{
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the overall bounds for each
/// vdim (across all elements) in @b lower and @b upper. We also return the
/// points based on \p ref_factor, and returns the overall bounds for each
/// vdim (across all elements) in \p lower and \p upper. We also return the
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
/// We compute the bounds for each vdim if \p vdim < 1.
/// Note: For most cases, this method/interface will be sufficient.
virtual PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) const;
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each element
/// ordered byVDim:
/// ordered byNODES:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}. We also return the
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
/// We compute the bounds for each vdim if \p vdim < 1.
PLBound GetElementBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) const;
@@ -1687,6 +1783,18 @@ public:
Vector &lower, Vector &upper,
const int vdim = -1) const;
/** @brief Gets the bounds on given reference range inside an element.
*
* @details @a ref_range is a vector of size 2*dim that specifies the
* lower and upper limits in each dimension of the reference element.
* For example, in 2D, ref_range = [rmin, smin, rmax, smax].
*/
void GetElementBoundsAtControlPoints(const int elem, const PLBound &plb,
const Vector &ref_range,
const int vdim,
Vector &lower, Vector &upper,
Vector &control_pos) const;
/// Compute bounds on the grid function for the given element.
/// The bounds are stored in @b lower and @b upper.
void GetElementBounds(const int elem, const PLBound &plb,
@@ -1694,11 +1802,45 @@ public:
const int vdim = -1) const;
/// Compute bounds on the grid function for all the elements. The bounds
/// are returned in @b lower and @b upper, ordered byVDim:
/// are returned in @b lower and @b upper, ordered byNODES:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
const int vdim=-1) const;
/** @brief Estimate the minimum value of the GridFunction in element @a elem.
*
* @details See the protected version of EstimateFunctionMinimum for
* details.
*/
std::pair<real_t, real_t> EstimateFunctionMinimum(const int elem,
const PLBound &plb,
const int vdim,
const int max_depth,
const real_t tol) const;
/** @brief Estimate the minimum value of the GridFunction in element @a elem.
*
* @details See the protected version of EstimateFunctionMaximum for
* details.
*/
std::pair<real_t, real_t> EstimateFunctionMaximum(const int elem,
const PLBound &plb,
const int vdim,
const int max_depth,
const real_t tol) const;
/** @brief Estimate the GridFunction minimum across all elements. */
virtual std::pair<real_t,real_t> EstimateFunctionMinimum(const int vdim,
const PLBound &plb,
const int max_depth,
const real_t tol) const;
/** @brief Estimate the GridFunction maximum across all elements. */
virtual std::pair<real_t,real_t> EstimateFunctionMaximum(const int vdim,
const PLBound &plb,
const int max_depth,
const real_t tol) const;
///@}
/// Destroys grid function.
@@ -1804,7 +1946,7 @@ real_t ComputeElementLpDistance(real_t p, int i,
GridFunction& gf1, GridFunction& gf2);
/// Class used for extruding scalar GridFunctions
/// Class used for extruding a scalar coefficient
class ExtrudeCoefficient : public Coefficient
{
private:
@@ -1812,13 +1954,53 @@ private:
Mesh *mesh_in;
Coefficient &sol_in;
public:
/// Constructs an instance of VectorExtrudeCoefficient
/**
* @param m 1D mesh
* @param s 1D vector coefficient
* @param n_ number of transverse elements of the extruded mesh
*/
ExtrudeCoefficient(Mesh *m, Coefficient &s, int n_)
: n(n_), mesh_in(m), sol_in(s) { }
: n(n_), mesh_in(m), sol_in(s)
{ MFEM_VERIFY(n > 0, "Number of transverse elements must be positive!"); }
real_t Eval(ElementTransformation &T, const IntegrationPoint &ip) override;
virtual ~ExtrudeCoefficient() { }
};
/// Extrude a scalar 1D GridFunction, after extruding the mesh with Extrude1D.
/// Class used for extruding a vector coefficient
class VectorExtrudeCoefficient : public VectorCoefficient
{
private:
int n;
Mesh *mesh_in;
VectorCoefficient &sol_in;
public:
/// Constructs an instance of VectorExtrudeCoefficient
/**
* @param m 1D mesh
* @param s 1D vector coefficient
* @param n_ number of transverse elements of the extruded mesh
*/
VectorExtrudeCoefficient(Mesh *m, VectorCoefficient &s, int n_)
: VectorCoefficient(s.GetVDim()), n(n_), mesh_in(m), sol_in(s)
{ MFEM_VERIFY(n > 0, "Number of transverse elements must be positive!"); }
void Eval(Vector &v, ElementTransformation &T,
const IntegrationPoint &ip) override;
using VectorCoefficient::Eval;
virtual ~VectorExtrudeCoefficient() { }
};
/// Extrude a 1D GridFunction, after extruding the mesh with Extrude1D()
/**
* @param mesh 1D mesh
* @param mesh2d extruded mesh
* @param sol grid function
* @param ny number of transverse elements of the extruded mesh
*/
GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
GridFunction *sol, const int ny);
+2394 -161
View File
File diff suppressed because it is too large Load Diff
+351 -74
View File
@@ -21,6 +21,45 @@
#ifdef MFEM_USE_GSLIB
/* gslib license and copyright statement for code adapted from gslib:
Copyright (c) 2008-2024, UCHICAGO ARGONNE, LLC.
The UChicago Argonne, LLC as Operator of Argonne National
Laboratory holds copyright in the Software. The copyright holder
reserves all rights except those expressly granted to licensees,
and U.S. Government license rights.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions
are met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the disclaimer below.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the disclaimer (as noted below)
in the documentation and/or other materials provided with the
distribution.
3. Neither the name of ANL nor the names of its contributors
may be used to endorse or promote products derived from this software
without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL
UCHICAGO ARGONNE, LLC, THE U.S. DEPARTMENT OF
ENERGY OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED
TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*/
namespace gslib
{
struct comm;
@@ -86,7 +125,7 @@ protected:
void *fdataD;
struct gslib::crystal *cr; // gslib's internal data
struct gslib::comm *gsl_comm; // gslib's internal data
int dim, points_cnt; // mesh dimension and number of points
int dim, spacedim, points_cnt; // mesh dimension and number of points
Array<unsigned int> gsl_code, gsl_proc, gsl_elem, gsl_mfem_elem;
Vector gsl_mesh, gsl_ref, gsl_dist, gsl_mfem_ref;
Array<unsigned int> recv_proc, recv_index; // data for custom interpolation
@@ -104,18 +143,23 @@ protected:
bool gpu_to_cpu_fallback = false;
// Device specific data used for FindPoints
struct
struct DEV_STRUCT
{
bool setup_device = false;
bool find_device = false;
int local_hash_size, dof1d, dof1d_sol, h_o_size, h_nx;
int local_hash_size, dof1d, dof1d_sol, lh_nx, gh_nx;
double newt_tol; // Tolerance specified during setup for Newton solve
struct gslib::crystal *cr;
struct gslib::hash_data_3 *hash3;
struct gslib::hash_data_2 *hash2;
mutable Vector bb, wtend, gll1d, lagcoeff, gll1d_sol, lagcoeff_sol;
mutable Array<unsigned int> loc_hash_offset;
mutable Vector loc_hash_min, loc_hash_fac;
mutable Array<unsigned int> lh_offset, gh_offset;
mutable Vector lh_min, lh_fac, gh_min, gh_fac;
// Tolerance to mark points found on the surface as CODE_INTERNAL
// or CODE_BORDER. This is needed because we cannot only use reference
// space coordinates to determine if a point is located inside the
// element or not.
mutable double surf_dist_tol;
} DEV;
/// Use GSLIB for communication and interpolation
@@ -127,88 +171,157 @@ protected:
Vector &field_out,
const int field_out_ordering);
/// Since GSLIB is designed to work with quads/hexes, we split every
/// triangle/tet/prism/pyramid element into quads/hexes.
/** @brief Since GSLIB is designed to work with quads/hexes, we split every
* triangle/tet/prism/pyramid element into quads/hexes. */
virtual void SetupSplitMeshes();
/// Setup integration points that will be used to interpolate the nodal
/// location at points expected by GSLIB.
/** @brief Setup integration points that will be used to interpolate the
* nodal location at points expected by GSLIB. */
virtual void SetupIntegrationRuleForSplitMesh(Mesh *mesh,
IntegrationRule *irule,
int order);
/// Helper function that calls \ref SetupSplitMeshes and
/// \ref SetupIntegrationRuleForSplitMesh.
/** @brief Helper function that calls \ref SetupSplitMeshes and
* \ref SetupIntegrationRuleForSplitMesh. */
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
/// Get GridFunction value at the points expected by GSLIB.
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
/// find the original element number (that was split into micro quads/hexes)
/// during the setup phase.
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
* simplices, find the original element number (that was split into
* micro quads/hexes) during the setup phase. */
virtual void MapRefPosAndElemIndices();
// Device functions
// FindPoints locally on device for 3D.
/// FindPoints locally on device for 3D.
void FindPointsLocal3(const Vector &point_pos, int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
Vector &gsl_dist_l, int npt);
// FindPoints locally on device for 2D.
/// FindPoints locally on device for 2D.
void FindPointsLocal2(const Vector &point_pos, int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
Vector &gsl_dist_l, int npt);
// Interpolate on device for 3D.
/// FindPoints locally on device for 3D surface elements.
void FindPointsSurfLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// FindPoints locally on device for 3D edge elements.
void FindPointsEdgeLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// FindPoints locally on device for 2D edge elements.
void FindPointsEdgeLocal2(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// Interpolate on device for 3D.
void InterpolateLocal3(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1dsol);
// Interpolate on device for 2D.
int dof1dsol);
/// Interpolate on device for 2D.
void InterpolateLocal2(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1dsol);
int dof1dsol);
// Prepare data for device functions.
/// Interpolate on device for 1D.
void InterpolateLocal1(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp, int dof1dsol);
/// Prepare data for device execution for volume meshes.
void SetupDevice();
/** Searches positions given in physical space by @a point_pos.
/** @brief Searches positions given in physical space by @a point_pos.
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
void FindPointsOnDevice(const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES);
/** Interpolation of field values at prescribed reference space positions.
@param[in] field_in_evec E-vector of grid function to be interpolated.
Assumed ordering is NDOFSxVDIMxNEL
@param[in] nel Number of elements in the mesh.
@param[in] ncomp Number of components in the field.
@param[in] dof1dsol Number of degrees of freedom in each reference
space direction.
@param[in] ordering Ordering of the out field values: byNodes/byVDIM
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value. */
/** @brief Interpolation of field values at prescribed reference space
* positions.
* @param[in] field_in_evec E-vector of grid function to be interpolated.
* Assumed ordering is NDOFSxVDIMxNEL
* @param[in] nel Number of elements in the mesh.
* @param[in] ncomp Number of components in the field.
* @param[in] dof1dsol Number of degrees of freedom in each reference
* space direction.
* @param[in] ordering Ordering of the out field values: byNodes/byVDIM
*
* @param[out] field_out Interpolated values. For points that are not
* found the value is set to
* #default_interp_value. */
void InterpolateOnDevice(const Vector &field_in_evec, Vector &field_out,
const int nel, const int ncomp,
const int dof1dsol, const int ordering);
/** @brief Interpolation of field values at prescribed reference space
* positions for surface meshes. */
void InterpolateSurfBase(const Vector &field_in, Vector &field_out,
const int nel, const int ncomp,
const int dof1dsol, const int field_out_ordering);
/// Preprocess 2D surface mesh needed for FindPoints.
void findptsedge_setup_2(DEV_STRUCT &devs,
const double *const elx[2],
const unsigned n,
const uint nel,
const unsigned m,
const double bbox_tol,
const uint local_hash_size,
const uint global_hash_size);
/// Preprocess 3D surface mesh needed for FindPoints.
void findptssurf_setup_3(DEV_STRUCT &devs,
const double *const elx[3],
const unsigned n,
const uint nel,
const unsigned m,
const double bbox_tol,
const uint local_hash_size,
const uint global_hash_size,
const int rD);
public:
/// Serial constructor
FindPointsGSLIB();
/// Serial constructor + setup with given Mesh (see \ref Setup)
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
#ifdef MFEM_USE_MPI
/// Constructor for ParMesh
FindPointsGSLIB(MPI_Comm comm_);
/// Constructor + setup with given ParMesh (see \ref Setup)
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
@@ -218,8 +331,10 @@ public:
FindPointsGSLIB(const FindPointsGSLIB&) = delete;
FindPointsGSLIB& operator=(const FindPointsGSLIB&) = delete;
/** Initializes the internal mesh in gslib, by sending the positions of the
Gauss-Lobatto nodes of the input Mesh object \p m.
/** @brief Preprocess the internal mesh in gslib.
@details Initializes the internal mesh in gslib, by sending the
positions of the Gauss-Lobatto nodes of the input Mesh object \p m.
Note: not tested with periodic (L2).
Note: the input mesh \p m must have Nodes set.
@@ -230,13 +345,22 @@ public:
search methods.
@param[in] npt_max (Optional) Number of points for simultaneous
iteration. This alters performance and
memory footprint.*/
memory footprint.
*/
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** Searches positions given in physical space by \p point_pos.
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
/// Preprocess the surface mesh to compute data for FindPoints.
void SetupSurf(Mesh &m,
const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** @brief Searches positions given in physical space by \p point_pos.
@details These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) specified by \p point_pos_ordering.
This function populates the following member variables:
#gsl_code Return codes for each point: inside element (0),
element boundary (1), not found (2).
@@ -255,45 +379,72 @@ public:
#gsl_dist Distance between the sought and the found point
in physical space. */
void FindPoints(const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES);
int point_pos_ordering = Ordering::byNODES);
/// Convenience function when point positions are in a ParticleVector
void FindPoints(const ParticleVector &point_pos)
{
FindPoints(point_pos, point_pos.GetOrdering());
}
/** @brief Searches positions given in physical space by \p point_pos on
* surface mesh. */
void FindPointsSurf(const Vector &point_pos,
int point_pos_ordering = Ordering::byNODES);
/// Convenience function when point positions are in a ParticleVector
void FindPointsSurf(const ParticleVector &point_pos)
{
FindPointsSurf(point_pos, point_pos.GetOrdering());
}
/// Setup FindPoints and search positions
void FindPoints(Mesh &m, const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** Interpolation of field values at prescribed reference space positions.
/** @brief Interpolation of field values at prescribed reference space
* positions.
@param[in] field_in Function values that will be interpolated on the
reference positions. Note: it is assumed that
\p field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value. */
the value is set to #default_interp_value.
The output ordering is determined from field_in.*/
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
/// Interpolation of field values, with output ordering specification.
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
/// Interpolation of field values, with output ordering from ParticleVector
virtual void Interpolate(const GridFunction &field_in,
ParticleVector &field_out)
{
Interpolate(field_in, field_out, field_out.GetOrdering());
}
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
the output values in \p field_out corresponds to the ordering used
in the input GridFunction \p field_in. */
/** @brief Same as Interpolate but for surface meshes */
virtual void InterpolateSurf(const GridFunction &field_in,
Vector &field_out);
/** @brief Same as Interpolate but for surface meshes with specified output
ordering */
virtual void InterpolateSurf(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering);
/** @brief Search positions and interpolate.
*
* @details The ordering (byNODES or byVDIM) of the output values in
* \p field_out corresponds to the ordering used in the input
* GridFunction \p field_in.
*/
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
int point_pos_ordering = Ordering::byNODES);
/// Search positions and interpolate with given point and output ordering.
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out, const int point_pos_ordering,
const int field_out_ordering);
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
or byVDIM) of the output values in \p field_out corresponds to the
ordering used in the input GridFunction \p field_in. */
@@ -301,32 +452,36 @@ public:
const GridFunction &field_in, Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
/// Average type to be used for L2 functions in-case a point is located at
/// an element boundary where the function might be multi-valued.
/** @brief Average type to be used for L2 functions in-case a point is
* located at an element boundary where the function might be multi-valued.
*/
virtual void SetL2AvgType(AvgType avgtype_) { avgtype = avgtype_; }
/// Set the default interpolation value for points that are not found in the
/// mesh.
/** @brief Set the default interpolation value for points that are not found in the mesh. */
virtual void SetDefaultInterpolationValue(double interp_value_)
{
default_interp_value = interp_value_;
}
/// Set the tolerance for detecting points outside the 'curvilinear' boundary
/// that gslib may return as found on the boundary. Points found on boundary
/// with distance greater than @ bdr_tol are marked as not found.
/** @brief Tolerance for detecting points outside the 'curvilinear' boundary.
*
* @details When using FindPoints, gslib may return points as found on the
* boundary even when they are slightly outside the domain. This tolerance
* is used to filter such points based on the distance^2 value and mark them
* as not found.*/
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
{
bdr_tol = bdr_tol_;
}
/// Enable/Disable use of CPU functions for GPU data if the gslib version
/// is older.
/** @brief Enable/Disable use of CPU functions for GPU data if the gslib
* version is older. */
virtual void SetGPUtoCPUFallback(bool mode) { gpu_to_cpu_fallback = mode; }
/** Cleans up memory allocated internally by gslib.
Note that in parallel, this must be called before MPI_Finalize(), as it
calls MPI_Comm_free() for internal gslib communicators. FreeData is
/** @brief Cleans up memory allocated internally by gslib.
@details Note that in parallel, this must be called before MPI_Finalize,
as it calls MPI_Comm_free() for internal gslib communicators. FreeData is
also called by the class destructor and there are no memory leaks if the
destructor is called before MPI_Finalize(). If the destructor is called
after MPI_Finalize(), there will be an error because gslib will try to
@@ -334,8 +489,8 @@ public:
*/
virtual void FreeData();
/// Return code for each point searched by FindPoints: inside element (0), on
/// element boundary (1), or not found (2).
/** @brief Return code for each point searched by FindPoints:
* inside element (0), element boundary (1), or not found (2). */
virtual const Array<unsigned int> &GetCode() const { return gsl_code; }
/// Return element number for each point found by FindPoints.
virtual const Array<unsigned int> &GetElem() const { return gsl_mfem_elem; }
@@ -343,15 +498,15 @@ public:
virtual const Array<unsigned int> &GetProc() const { return gsl_proc; }
/// Return reference coordinates for each point found by FindPoints.
virtual const Vector &GetReferencePosition() const { return gsl_mfem_ref; }
/// Return distance between the sought and the found point in physical space,
/// for each point found by FindPoints.
/// Return distance between the sought and the found point in physical space.
virtual const Vector &GetDist() const { return gsl_dist; }
/// Return element number for each point found by FindPoints corresponding to
/// GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with simplices.
/** @brief Return element number for each point found by FindPoints
* corresponding to GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with
* simplices. */
virtual const Array<unsigned int> &GetGSLIBElem() const { return gsl_elem; }
/// Return reference coordinates in [-1,1] (internal range in GSLIB) for each
/// point found by FindPoints.
/** @brief Return reference coordinates in [-1,1] (internal range in GSLIB)
* for each point found by FindPoints. */
virtual const Vector &GetGSLIBReferencePosition() const { return gsl_ref; }
/// Get array of indices of not-found points.
@@ -394,7 +549,7 @@ public:
/// Return the axis-aligned bounding boxes (AABB) computed during \ref Setup.
/// The size of the returned vector is (nel x nverts x dim), where nel is the
/// number of elements (after splitting for simplcies), nverts is number of
/// number of elements (after splitting for simplicies), nverts is number of
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
@@ -408,6 +563,18 @@ public:
/// \p obbV, a vector of size (nel x nverts x dim) .
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
Vector &obbV) const;
/** @brief Return the bounding boxes as a mesh on rank 0.
*
* @param[in] type Bounding-box type: 0 - AABB, 1 - OBB.
*
* @return On rank 0, returns a newly allocated mesh containing the
* bounding boxes. The caller owns the returned pointer and is responsible
* for deleting it. On other ranks, returns nullptr.
*/
Mesh *GetBoundingBoxMesh(int type);
virtual const Vector &GetGLLMesh() const { return gsl_mesh; }
};
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
@@ -535,6 +702,116 @@ public:
void GS(Vector &senddata, GSOp op);
};
#if defined(MFEM_USE_MPI)
/** \brief Class to map a point in physical space to candidate ranks.
*
* This class builds a Cartesian-aligned tensor grid that covers the entire
* domain and precomputes which ranks have elements intersecting each
* grid cell. Given a point in physical space, the grid cell containing
* the point is determined, and the list of candidate ranks whose
* elements intersect that cell is returned. This yields a fast, conservative
* point-to-rank candidate query. This is used internally by FindPointsGSLIB
* to speed up point searches in parallel.
*
* See Mittal et al., "General Field Evaluation in High-Order Meshes on GPUs".
* (2025). Computers & Fluids. for technical details.
*
*/
class GlobalBBoxTensorGridMap
{
private:
struct gslib::crystal *cr = nullptr; // gslib's internal data
struct gslib::comm *gsl_comm = nullptr; // gslib's internal data
int sdim, n_local_cells, num_procs;
Array<int> gmap_n;
Vector gmap_bnd_min, gmap_bnd_max;
Vector gmap_fac;
Array<int> ggrid_map;
void SetupCrystal(const MPI_Comm &comm);
public:
/// Constructor for a given mesh and number of tensor grid divisions
GlobalBBoxTensorGridMap(ParMesh &pmesh, int nx);
/** @brief Constructor for given element bounds and spatial dimension.
*
* @details This constructor must be called collectively on \a comm.
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
*
* Assumes elmin, elmax Ordering::byNodes:
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
*
* When by_max_size=false, n gives the number of tensor-grid divisions in
* each direction. When by_max_size=true, n is a per-rank size hint used to
* derive a uniform global resolution. The communicator-wide sum of n is
* converted to nx = ceil(pow(sum(n), 1./sdim)) in each direction, so n is
* not a hard cap on ggrid_map.Size().
*/
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
Vector &elmax, int nel, int sdim, int n,
bool by_max_size);
/** @brief Constructor for given element bounds, spatial dimension, and
* tensor-grid divisions in each direction.
*
* @details This constructor must be called collectively on \a comm.
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
* Requires nx.Size() == sdim and positive entries in nx.
*
* Assumes elmin, elmax Ordering::byNodes:
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
*/
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
Vector &elmax, int nel, int sdim, Array<int> &nx);
~GlobalBBoxTensorGridMap();
/** @brief Get list of procs corresponding to the list of points.
*
* @details This method must be called collectively on the communicator
* used to construct the map. The input points can be ordered byNodes:
* (XXX...,YYY...,ZZZ) or byVDIM: (XYZ,XYZ,...), as specified by
* \a ordering.
*
* The output map contains one entry for each input point, keyed by the
* point's local index in \a xyz. Points with no candidate ranks, including
* points outside the global bounding box, have an empty list of candidate
* ranks.
*/
void MapPointsToProcs(Vector &xyz, int ordering,
std::map<int, std::vector<int>> &pt_to_procs) const;
// Some getters
const Array<int> &GetGridMap() const { return ggrid_map; }
const Vector &GetGridFac() const { return gmap_fac; }
const Vector &GetGridMin() const { return gmap_bnd_min; }
const Vector &GetGridMax() const { return gmap_bnd_max; }
const Array<int> &GetGridN() const { return gmap_n; }
private:
/// Setup the map given element bounds and number of tensor grid divisions.
void Setup(const MPI_Comm &comm, Vector &elmin, Vector &elmax,
int nel, Array<int> &nx);
/// Get global hash cell index for a given point.
int GetGlobalGridCellFromPoint(Vector &xyz) const;
/** @brief Get owning proc and local index on that proc for given global
* grid cell index. */
void GlobalGridCellToProcAndLocalIndex(int i, int &proc, int &idx) const;
/// Map a point to proc and local index of the corresponding grid cell
void GetProcAndLocalIndexFromPoint(Vector &xyz, int &proc, int &idx) const;
/// Given local cell index, return list of procs saved in the map
Array<int> MapCellToProcs(int l_idx) const;
};
#endif // MFEM_USE_MPI
} // namespace mfem
#endif // MFEM_USE_GSLIB
+15 -13
View File
@@ -254,7 +254,7 @@ get_edge(const double *elx[2], const double *wtend, int ei,
edge.dxdn[d] = workspace + (2 + d) * pN; //dxdn and dydn at DOFs along edge
}
if (side_init != (1u << ei))
if (static_cast<unsigned>(side_init) != (1u << ei))
{
#define ELX(d, j, k) elx[d][j + k * pN] // assumes lexicographic ordering
for (int d = 0; d < 2; ++d)
@@ -562,7 +562,7 @@ newton_area_fin:
int f = flags >> (2 * dd) & 3u;
res->r[dd] = f == 0 ? r0[dd] + dr[dd] : (f == 1 ? -1 : 1);
}
res->flags = flags | (p->flags << 5);
res->flags = flags | ((p->flags & FLAG_MASK) << 5);
}
// Full Newton solve on the face. One of r/s/t is constrained.
@@ -635,7 +635,8 @@ newton_edge_fin:
res->r[de] = nr;
res->r[dn]=p->r[dn];
res->dist2p = -v;
res->flags = flags | new_flags | (p->flags << 5);
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 5);
#undef EVAL
}
// Find closest mesh node to the sought point.
@@ -714,7 +715,6 @@ static void FindPointsLocal2D_Kernel(const int npt,
const double *lagcoeff,
const int pN = 0)
{
#define MAX_CONST(a, b) (((a) > (b)) ? (a) : (b))
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NE = D1D*D1D;
@@ -729,7 +729,7 @@ static void FindPointsLocal2D_Kernel(const int npt,
// 3D1D for seed, 10D1D+6 for area, 3D1D+9 for edge
constexpr int size1 = 10*MD1 + 6;
constexpr int size2 = MD1*4; // edge constraints
constexpr int size3 = MD1*MD1*MD1*DIM; // local element coordinates
constexpr int size3 = MD1*MD1*DIM; // local element coordinates
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED findptsElementPoint_t el_pts[2];
@@ -1162,9 +1162,9 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
auto pgslm = gsl_mesh.Read();
auto pwt = DEV.wtend.Read();
auto pbb = DEV.bb.Read();
auto plhm = DEV.loc_hash_min.Read();
auto plhf = DEV.loc_hash_fac.Read();
auto plho = DEV.loc_hash_offset.ReadWrite();
auto plhm = DEV.lh_min.Read();
auto plhf = DEV.lh_fac.Read();
auto plho = DEV.lh_offset.ReadWrite();
auto pcode = code.Write();
auto pelem = elem.Write();
auto pref = ref.Write();
@@ -1177,30 +1177,32 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
case 2:
return FindPointsLocal2D_Kernel<2>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 3:
return FindPointsLocal2D_Kernel<3>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 4:
return FindPointsLocal2D_Kernel<4>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 5:
return FindPointsLocal2D_Kernel<5>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
default:
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx,
plhm, plhf, plho, pcode, pelem,
pref, pdist, pgll1d, plc, DEV.dof1d);
}
}
#undef DIM2
#undef DIM
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
+16 -16
View File
@@ -294,7 +294,7 @@ get_face(const double *elx[3], const double *wtend, int fi, double *workspace,
face.dxdn[d] = workspace+(3+d)*p_Nfr;
}
if (side_init != (1u << fi))
if (static_cast<unsigned>(side_init) != (1u << fi))
{
const int e_stride[3] = {1, pN, pN*pN};
#define ELX(d, j, k, l) elx[d][j*e_stride[d1]+k*e_stride[d2]+l*e_stride[dn]]
@@ -342,7 +342,7 @@ get_edge(const double *elx[3], const double *wtend, int ei, double *workspace,
if (jidx >= 3*pN) { return edge; }
if (side_init != (64u << ei))
if (static_cast<unsigned>(side_init) != (64u << ei))
{
const int e_stride[3] = {1, pN, pN*pN};
#define ELX(d, j, k, l) elx[d][j*e_stride[de]+k*e_stride[dn1]+l*e_stride[dn2]]
@@ -706,7 +706,7 @@ newton_vol_fin:
int f = flags >> (2*dd) & 3u;
res->r[dd] = f == 0 ? r0[dd]+dr[dd] : (f == 1 ? -1 : 1);
}
res->flags = flags | (p->flags << 7);
res->flags = flags | ((p->flags & FLAG_MASK) << 7);
}
// Full Newton solve on the face. One of r/s/t is constrained.
@@ -889,7 +889,7 @@ newton_face_fin:
res->r[dn] = p->r[dn];
res->r[d1] = r[0];
res->r[d2] = r[1];
res->flags = new_flags | (p->flags << 7);
res->flags = new_flags | ((p->flags & FLAG_MASK) << 7);
}
// Full Newton solve on the edge. Two of r/s/t are constrained.
@@ -973,7 +973,8 @@ newton_edge_fin:
res->r[dn1] = p->r[dn1];
res->r[dn2] = p->r[dn2];
res->dist2p = -v;
res->flags = flags | new_flags | (p->flags << 7);
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 7);
#undef EVAL
}
// Find closest mesh node to the sought point.
@@ -1252,7 +1253,6 @@ static void FindPointsLocal3DKernel(const int npt,
case 0: // findpt_vol
{
double *wtr = r_workspace_ptr;
double *resid = wtr+6*D1D;
double *jac = resid+3;
double *resid_temp = jac+9;
@@ -1503,7 +1503,7 @@ static void FindPointsLocal3DKernel(const int npt,
// Hes_T is transposed version (i.e. in col major)
// n1*[2, 1, 1, 0, 0]
// j==1 => wt_j = wt+n1
double *wt_j = wt+D1D*(2-(row+1) / 2);
double *wt_j = wt+D1D*(2 - (row+1)/2);
const double *x = e_x[row+1][d];
hes_T[j] = 0.0;
for (int k = 0; k < D1D; ++k)
@@ -1522,7 +1522,6 @@ static void FindPointsLocal3DKernel(const int npt,
hes[j] += resid[d]*hes_T[j*3+d];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(l,x,1)
@@ -1780,6 +1779,7 @@ static void FindPointsLocal3DKernel(const int npt,
} //findpts_local
} //elp
});
#undef MAXC
}
void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
@@ -1796,9 +1796,9 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
auto pgslm = gsl_mesh.Read();
auto pwt = DEV.wtend.Read();
auto pbb = DEV.bb.Read();
auto plhm = DEV.loc_hash_min.Read();
auto plhf = DEV.loc_hash_fac.Read();
auto plho = DEV.loc_hash_offset.ReadWrite();
auto plhm = DEV.lh_min.Read();
auto plhf = DEV.lh_fac.Read();
auto plho = DEV.lh_offset.ReadWrite();
auto pcode = code.Write();
auto pelem = elem.Write();
auto pref = ref.Write();
@@ -1809,31 +1809,31 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
{
case 2:
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 3:
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 4:
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 5:
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
default:
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.h_nx, plhm, plhf,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc,
DEV.dof1d);
}
+725
View File
@@ -0,0 +1,725 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
#define sDIM 2
#define sDIM2 4
#define rDIM 1
struct findptsElementPoint_t
{
double x[sDIM], r, oldr, dist2, dist2p, tr;
int flags;
};
struct findptsElementGEdge_t
{
double *x[sDIM];
};
struct findptsElementGPT_t
{
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[sDIM], A[sDIM*sDIM];
dbl_range_t x[sDIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[sDIM];
double fac[sDIM];
unsigned int *offset;
};
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x-z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
p0[i] = lCoeff[i] * u0;
p1[i] = 2.0 * lCoeff[i] * u1;
p2[i] = 8.0 * lCoeff[i] * u2;
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
const double x[sDIM])
{
double b_d;
for (int d=0; d<sDIM; ++d)
{
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
if (b_d < 0) // if outside in any dimension
{
return b_d;
}
}
return b_d; // only positive if inside
}
/* positive when given point is possibly inside given obbox b */
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
const double x[sDIM])
{
const double bxyz = obbox_axis_test(b,x);
if (bxyz<0) // test if point is in AABB
{
return bxyz;
}
else // test OBB only if inside AABB
{
double dxyz[sDIM];
for (int d=0; d<sDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1;
for (int d=0; d<sDIM; ++d)
{
double rst = 0;
for (int e=0; e<sDIM; ++e)
{
rst += b->A[d*2 + e] * dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test<0 ? test : brst;
}
return test;
}
}
/* Hash index in the hash table to the elements that possibly contain the point x */
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[2])
{
const int n = p->hash_n;
int sum = 0;
for (int d=sDIM-1; d>=0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
{
return x[0] * x[0] + x[1] * x[1];
}
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
1 = 01b if r is constrained at -1, i.e., rmin
2 = 10b if r is constrained at +1, i.e., rmax
*/
#define CONVERGED_FLAG (1u<<2)
#define FLAG_MASK 0x07u // = 111b
/* returns 1 if r direction (the only free direction in 2D) is constrained.
returns 1 if either 1st or 2nd bit of flags is set.
*/
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
{
return ((flags | flags>>1) & 1u);
}
/* pi=0, r=-1; pi=1, r=+1 */
static MFEM_HOST_DEVICE inline int point_index(const int x)
{
return ((x>>1) & 1u);
}
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out->dist2, out->index, out->x, out->oldr in any event,
leaving out->r, out->dr, out->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
const double resid[2],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
out->x[0] = p->x[0];
out->x[1] = p->x[1];
out->oldr = p->r;
out->dist2 = dist2;
if (decr >= 0.01*pred)
{
if (decr >= 0.9*pred) // very good iteration
{
out->tr = p->tr*2;
}
else // somewhat good iteration
{
out->tr = p->tr;
}
return false;
}
else
{
/* reject step; note: the point will pass through this routine
again, and we set things up here so it gets classed as a
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out->tr = v0/4.0;
out->dist2 = p->dist2;
out->r = p->oldr;
out->flags = p->flags>>3;
out->dist2p = -HUGE_VAL;
if (pred < dist2*tol)
{
out->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
out,
const double jac[2],
const double rhess,
const double resid[2],
int flags,
const findptsElementPoint_t *const p,
const double tol )
{
const double tr = p->tr;
const double A = jac[0] * jac[0] + jac[1] * jac[1] -
rhess; // A = J^T J - resid_d H_d
const double y = jac[0]*resid[0] + jac[1]*resid[1]; // y = J^T resid
const double oldr = p->r;
double dr, newr, tdr, tnewr, v, tv;
int new_flags=0, tnew_flags=0;
#define EVAL(dr) ( (dr*A - 2*y) * dr )
if (A>0)
{
dr = y/A;
if (fabs(dr)<tol)
{
dr=0.0;
newr = oldr;
}
else
{
newr = oldr+dr;
}
if (fabs(dr)<tr && fabs(newr)<1)
{
v = EVAL(dr);
goto newton_edge_fin;
}
}
if ((newr=oldr-tr) > -1)
{
dr = -tr;
}
else
{
newr = -1, dr = -1-oldr, new_flags = flags|1u;
}
v = EVAL(dr);
if ((tnewr=oldr+tr) < 1)
{
tdr = tr;
}
else
{
tnewr = 1, tdr = 1-oldr, tnew_flags = flags|2u;
}
tv = EVAL(tdr);
if (tv<v)
{
newr = tnewr, dr = tdr, v = tv, new_flags = tnew_flags;
}
#undef EVAL
newton_edge_fin:
// check convergence by testing if change in r is less than tol
if (fabs(dr)<tol)
{
new_flags |= CONVERGED_FLAG;
}
out->r = newr;
out->dist2p = -v;
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
}
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
const double x[sDIM],
const double *z,
double *dist2,
double *r,
const int ir,
const int pN )
{
double dx[sDIM];
for (int d=0; d<sDIM; ++d)
{
dx[d] = x[d] - elx[d][ir];
}
dist2[ir] = HUGE_VAL;
const double dist2_rs = l2norm2(dx);
if (dist2[ir]>dist2_rs)
{
dist2[ir] = dist2_rs;
r[ir] = z[ir];
}
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal2D_Kernel( const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0 )
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NEL = nel*D1D;
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
const int nThreads = D1D*sDIM;
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
{
// 2D1D for seed, 3D1D + 7 for edge
constexpr int size1 = 3*MD1 + 7;
// edge coordinates = D1D*2
constexpr int size2 = 2*MD1;
// local element coordinates in shared memory
constexpr int size3 = MD1*sDIM;
MFEM_SHARED findptsElementPoint_t el_pts[2];
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED double constraint_workspace[size2];
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
double *r_workspace_ptr = r_workspace;
findptsElementPoint_t *fpt, *tmp;
fpt = el_pts + 0;
tmp = el_pts + 1;
// x and y coord index within point_pos for point i
int id_x = point_pos_ordering == 0 ? i : i*sDIM;
int id_y = point_pos_ordering == 0 ? i+npt : i*sDIM+1;
double x_i[2] = {x[id_x], x[id_y]};
unsigned int *code_i = code_base + i;
double *dist2_i = dist2_base + i;
//---------------- map_points_to_els --------------------
findptsLocalHashData_t hash;
for (int d=0; d<sDIM; ++d)
{
hash.bnd[d].min = hashMin[d];
hash.fac[d] = hashFac[d];
}
hash.hash_n = hash_n;
hash.offset = hashOffset;
const int hi = hash_index(&hash, x_i);
const unsigned int *elp = hash.offset + hash.offset[hi];
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
*code_i = CODE_NOT_FOUND;
*dist2_i = HUGE_VAL;
for (; elp!=ele; ++elp)
{
const unsigned int el = *elp;
obbox_t box;
int n_box_ents = 3*sDIM + sDIM2;
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
if (obbox_test(&box,x_i)>=0)
{
//------------ findpts_local ------------------
{
if (MD1 <= 6)
{
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
{
const int qp = j % D1D;
const int d = j / D1D;
elem_coords[qp + d*D1D] =
xElemCoord[qp + el*D1D + d*p_NEL];
}
MFEM_SYNC_THREAD;
}
const double *elx[sDIM];
for (int d=0; d<sDIM; d++)
{
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
xElemCoord + d*p_NEL + el*D1D;
}
MFEM_SYNC_THREAD;
//// findpts_el ////
{
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
fpt->dist2p = 0;
fpt->tr = 1;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
fpt->x[j] = x_i[j];
}
MFEM_SYNC_THREAD;
{
double *dist2_temp = r_workspace_ptr;
double *r_temp = dist2_temp + D1D;
MFEM_FOREACH_THREAD(j,x,D1D)
{
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
for (int ir=0; ir<D1D; ++ir)
{
if (dist2_temp[ir]<fpt->dist2)
{
fpt->dist2 = dist2_temp[ir];
fpt->r = r_temp[ir];
}
}
}
MFEM_SYNC_THREAD;
} //seed done
// Initialize tmp struct with fpt values before starting Newton iterations
MFEM_FOREACH_THREAD(j,x,1)
{
tmp->dist2 = HUGE_VAL;
tmp->dist2p = 0;
tmp->tr = 1;
tmp->flags = 0;
tmp->r = fpt->r;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
tmp->x[j] = fpt->x[j];
}
MFEM_SYNC_THREAD;
for (int step=0; step<50; step++)
{
int nc = num_constrained(tmp->flags & FLAG_MASK);
switch (nc)
{
case 0:
{
double *wt = r_workspace_ptr;
double *resid = wt + 3*D1D;
double *jac = resid + sDIM;
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
edge.x[d][j] = elx[d][j];
}
}
MFEM_SYNC_THREAD;
// compute basis function info upto 2nd derivative
MFEM_FOREACH_THREAD(j,x,D1D)
{
lag_eval_second_der(wt, tmp->r, j, gll1D,
lagcoeff, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,sDIM)
{
resid[j] = tmp->x[j];
jac[j] = 0.0;
hess[j] = 0.0;
for (int k=0; k<D1D; ++k)
{
resid[j] -= wt[ k]*edge.x[j][k];
jac[j] += wt[D1D+k]*edge.x[j][k];
hess[j] += wt[2*D1D+k]*edge.x[j][k];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
hess[2] = resid[0]*hess[0] + resid[1]*hess[1];
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
if (!reject_prior_step_q(fpt, resid, tmp, tol))
{
newton_edge(fpt, jac, hess[2], resid,
tmp->flags & FLAG_MASK, tmp, tol);
}
}
MFEM_SYNC_THREAD;
break;
}
case 1: // r is constrained to either -1 or 1
{
MFEM_FOREACH_THREAD(j,x,1)
{
const int pi = point_index(tmp->flags &
FLAG_MASK);
const double *wt = wtend + pi*3*D1D;
findptsElementGPT_t gpt;
for (int d=0; d<sDIM; ++d)
{
gpt.x[d] = elx[d][pi*(D1D-1)];
gpt.jac[d] = 0.0;
gpt.hes[d] = 0.0;
for (int k=0; k<D1D; ++k)
{
gpt.jac[d] += wt[D1D +k]*elx[d][k];
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
}
}
const double *const pt_x = gpt.x;
const double *const jac = gpt.jac;
const double *const hes = gpt.hes;
double resid[sDIM], steep, sr;
resid[0] = fpt->x[0] - pt_x[0];
resid[1] = fpt->x[1] - pt_x[1];
steep = jac[0]*resid[0] + jac[1]*resid[1];
sr = steep*tmp->r;
if ( !reject_prior_step_q(fpt, resid, tmp, tol) )
{
if (sr<0)
{
const double rhess = resid[0]*hes[0] +
resid[1]*hes[1];
newton_edge(fpt, jac, rhess,
resid, 0, tmp, tol);
}
else // sr==0
{
fpt->r = tmp->r;
fpt->dist2p = 0;
fpt->flags = tmp->flags | CONVERGED_FLAG;
}
}
}
MFEM_SYNC_THREAD;
break;
} // case 1
} //switch
if (fpt->flags & CONVERGED_FLAG)
{
break;
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
*tmp = *fpt;
}
MFEM_SYNC_THREAD;
} //for int step<50
} //findpts_el
bool converged_internal =
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
(fpt->dist2<dist2tol);
if (*code_i == CODE_NOT_FOUND || converged_internal ||
fpt->dist2 < *dist2_i)
{
MFEM_FOREACH_THREAD(j,x,1)
{
*(el_base+i) = el;
*code_i = converged_internal ? CODE_INTERNAL : CODE_BORDER;
*dist2_i = fpt->dist2;
*(r_base+i) = fpt->r;
}
MFEM_SYNC_THREAD;
if (converged_internal)
{
break;
}
}
} //findpts_local
} //obbox_test
} //elp
});
}
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt )
{
if (npt==0)
{
return;
}
MFEM_VERIFY(dim==1 && spacedim==2,"Function for 2D edges only");
bool use_dev = point_pos.UseDevice();
auto pp = point_pos.Read(use_dev);
auto pgslm = gsl_mesh.Read(use_dev);
auto pwt = DEV.wtend.Read(use_dev);
auto pbb = DEV.bb.Read(use_dev);
auto plhm = DEV.lh_min.Read(use_dev);
auto plhf = DEV.lh_fac.Read(use_dev);
auto plho = DEV.lh_offset.ReadWrite(use_dev);
auto pcode = code.Write(use_dev);
auto pelem = elem.Write(use_dev);
auto pref = ref.Write(use_dev);
auto pdist = dist.Write(use_dev);
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
switch (DEV.dof1d)
{
case 2:
return FindPointsEdgeLocal2D_Kernel<2>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
case 3:
return FindPointsEdgeLocal2D_Kernel<3>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
case 4:
return FindPointsEdgeLocal2D_Kernel<4>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
default:
return FindPointsEdgeLocal2D_Kernel(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
}
}
#undef sDIM
#undef rDIM
#undef sDIM2
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt ) {} ;
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
+733
View File
@@ -0,0 +1,733 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
#define sDIM 3
#define rDIM 1
#define sDIM2 (sDIM*sDIM)
#define rDIM2 (rDIM*rDIM)
struct findptsElementPoint_t
{
double x[sDIM], r, oldr, dist2, dist2p, tr;
int flags;
};
struct findptsElementGEdge_t
{
double *x[sDIM], *dxdn[sDIM], *d2xdn[sDIM];
};
struct findptsElementGPT_t
{
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
};
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[sDIM], A[sDIM*sDIM];
dbl_range_t x[sDIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[sDIM];
double fac[sDIM];
unsigned int *offset;
};
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j=0; j<pN; ++j)
{
if (i!=j)
{
double d_j = 2 * (x-z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
p0[i] = lCoeff[i] * u0;
p1[i] = 2.0 * lCoeff[i] * u1;
p2[i] = 8.0 * lCoeff[i] * u2;
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
const double x[sDIM])
{
double b_d;
for (int d=0; d<sDIM; ++d)
{
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
if (b_d < 0) // if outside in any dimension
{
return b_d;
}
}
return b_d; // only positive if inside in all dimensions
}
/* positive when possibly inside */
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
const double x[sDIM])
{
const double bxyz = obbox_axis_test(b, x);
if (bxyz<0)
{
return bxyz;
}
else
{
double dxyz[3];
// dxyz: distance of the point from the center of the OBB
for (int d=0; d<sDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
// transform dxyz to the local coordinate system of the OBB,
// and check if the point is inside the OBB [-1,1]^sDIM
double test = 1;
for (int d=0; d<sDIM; ++d)
{
double rst = 0;
for (int e=0; e<sDIM; ++e)
{
rst += b->A[d*sDIM + e] * dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test<0 ? test : brst;
}
return test;
}
}
/* Hash index in the hash table to the elements that possibly contain the point x */
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[sDIM])
{
const int n = p->hash_n;
int sum = 0;
for (int d=sDIM-1; d>=0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
{
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2] );
}
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
1 = 01b if r is constrained at -1, i.e., rmin
2 = 10b if r is constrained at +1, i.e., rmax
*/
#define CONVERGED_FLAG (1u<<2)
#define FLAG_MASK 0x07u
/* returns the number of constrained reference coordinates, max 2
*/
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
{
const int y = (flags | flags>>1);
return (y & 1u) + (y>>2 & 1u);
}
static MFEM_HOST_DEVICE inline int point_index(const int x)
{
return ((x>>1)&1u) | ((x>>2)&2u);
}
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out->dist2, out->index, out->x, out->oldr in any event,
leaving out->r, out->dr, out->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
const double resid[3],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = norm2(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d=0; d<sDIM; ++d)
{
out->x[d] = p->x[d];
}
out->oldr = p->r;
out->dist2 = dist2;
if (decr>=0.01*pred)
{
if (decr>=0.9*pred) // very good iteration
{
out->tr = 2*p->tr;
}
else // good iteration
{
out->tr = p->tr;
}
return false;
}
else // if the iteration in not good
{
/* reject step; note: the point will pass through this routine
again, and we set things up here so it gets classed as a
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out->tr = v0/4.0;
out->dist2 = p->dist2;
out->r = p->oldr;
out->flags = p->flags>>3;
out->dist2p = -HUGE_VAL;
if (pred<dist2*tol)
{
out->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
out,
const double jac[sDIM*rDIM],
const double rhes,
const double resid[sDIM],
int flags,
const findptsElementPoint_t *const p,
const double tol)
{
const double tr = p->tr;
/* A = J^T J - resid_d H_d */
const double A = jac[0]*jac[0]+ jac[1] * jac[1] + jac[2] * jac[2]
- rhes;
/* y = J^T r */
const double y = jac[0]*resid[0] + jac[1]*resid[1] + jac[0+2]*resid[2];
const double oldr = p->r;
double dr, nr, tdr, tnr;
double v, tv;
int new_flags = 0, tnew_flags = 0;
#define EVAL(dr) (dr*A - 2*y)*dr
/* if A is not SPD, quadratic model has no minimum */
if (A>0)
{
dr = y/A;
if (fabs(dr)<tol)
{
dr=0.0;
nr = oldr;
}
else
{
nr = oldr+dr;
}
if ( fabs(dr)<tr && fabs(nr)<1 )
{
v = EVAL(dr);
goto newton_edge_fin;
}
}
if ( (nr=oldr-tr)>-1 )
{
dr = -tr;
}
else
{
nr = -1, dr = -1-oldr, new_flags = flags | 1u;
}
v = EVAL(dr);
if ( (tnr = oldr+tr)<1 )
{
tdr = tr;
}
else
{
tnr = 1, tdr = 1-oldr, tnew_flags = flags | 2u;
}
tv = EVAL(tdr);
if (tv<v)
{
nr = tnr, dr = tdr, v = tv, new_flags = tnew_flags;
}
newton_edge_fin:
/* check convergence */
if ( fabs(dr)<tol )
{
new_flags |= CONVERGED_FLAG;
}
out->r = nr;
out->dist2p = -v;
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
#undef EVAL
}
static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
const double x[sDIM],
const double *z,
double *dist2,
double *r,
const int ir,
const int pN)
{
if (ir>=pN)
{
return;
}
double dx[sDIM];
for (int d=0; d<sDIM; ++d)
{
dx[d] = x[d] - elx[d][ir];
}
dist2[ir] = norm2(dx);;
r[ir] = z[ir];
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal3D_Kernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NEL = nel*D1D;
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
const int nThreads = D1D*sDIM;
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
{
constexpr int size1 = 3*MD1 + 13;
constexpr int size2 = 3*MD1;
constexpr int size3 = MD1*sDIM;
MFEM_SHARED findptsElementPoint_t el_pts[2];
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED double constraint_workspace[size2];
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
double *r_workspace_ptr = r_workspace;
findptsElementPoint_t *fpt, *tmp;
fpt = el_pts + 0;
tmp = el_pts + 1;
int id_x = point_pos_ordering==0 ? i : i*sDIM;
int id_y = point_pos_ordering==0 ? npt+i : 1+i*sDIM;
int id_z = point_pos_ordering==0 ? 2*npt+i : 2+i*sDIM;
double x_i[3] = {x[id_x], x[id_y], x[id_z]};
unsigned int *code_i = code_base + i;
double *dist2_i = dist2_base + i;
//// map_points_to_els ////
findptsLocalHashData_t hash;
for (int d=0; d<sDIM; ++d)
{
hash.bnd[d].min = hashMin[d];
hash.fac[d] = hashFac[d];
}
hash.hash_n = hash_n;
hash.offset = hashOffset;
const unsigned int hi = hash_index(&hash, x_i);
const unsigned int *elp = hash.offset + hash.offset[hi];
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
*code_i = CODE_NOT_FOUND;
*dist2_i = HUGE_VAL;
for (; elp!=ele; ++elp)
{
const unsigned int el = *elp;
obbox_t box;
int n_box_ents = 3*sDIM + sDIM2;
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
if (obbox_test(&box, x_i)>=0)
{
//// findpts_local ////
{
if (MD1 <= 6)
{
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
{
const int qp = j % D1D;
const int d = j / D1D;
elem_coords[qp + d*D1D] =
xElemCoord[qp + el*D1D + d*p_NEL];
}
MFEM_SYNC_THREAD;
}
const double *elx[sDIM];
for (int d=0; d<sDIM; d++)
{
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
xElemCoord + d*p_NEL + el*D1D;
}
MFEM_SYNC_THREAD;
//// findpts_el ////
{
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
fpt->dist2p = 0;
fpt->tr = 1.0;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
fpt->x[j] = x_i[j];
}
MFEM_SYNC_THREAD;
//// seed ////
{
double *dist2_temp = r_workspace_ptr;
double *r_temp = dist2_temp + D1D;
MFEM_FOREACH_THREAD(j,x,nThreads)
{
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
for (int ir=0; ir<D1D; ++ir)
{
if (dist2_temp[ir] < fpt->dist2)
{
fpt->dist2 = dist2_temp[ir];
fpt->r = r_temp[ir];
}
}
}
MFEM_SYNC_THREAD;
} //seed done
MFEM_FOREACH_THREAD(j,x,1)
{
tmp->dist2 = HUGE_VAL;
tmp->dist2p = 0;
tmp->tr = 1;
tmp->flags = 0;
tmp->r = fpt->r;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
tmp->x[j] = fpt->x[j];
}
MFEM_SYNC_THREAD;
for (int step=0; step<50; step++)
{
switch (num_constrained(tmp->flags & FLAG_MASK))
{
case 0:
{
double *wt = r_workspace_ptr;
double *resid = wt + 3*D1D;
double *jac = resid + sDIM;
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
edge.x[d][j] = elx[d][j];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,D1D)
{
lag_eval_second_der(wt, tmp->r, j, gll1D,
lagcoeff, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,sDIM)
{
resid[j] = tmp->x[j];
jac[j] = 0.0;
hess[j] = 0.0;
for (int k=0; k<D1D; ++k)
{
resid[j] -= wt[ k]*edge.x[j][k];
jac[j] += wt[D1D+k]*edge.x[j][k];
hess[j] += wt[2*D1D+k]*edge.x[j][k];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
hess[3] = resid[0]*hess[0] + resid[1]*hess[1] +
resid[2]*hess[2];
}
MFEM_FOREACH_THREAD(l,x,1)
{
if (!reject_prior_step_q(fpt,resid,tmp,tol))
{
newton_edge(fpt,jac,hess[3],resid,
tmp->flags&FLAG_MASK,tmp,tol);
}
}
MFEM_SYNC_THREAD;
break;
}
case 1:
{
MFEM_FOREACH_THREAD(j,x,1)
{
const int pi = point_index(tmp->flags &
FLAG_MASK);
const double *wt = wtend + pi*3*D1D;
findptsElementGPT_t gpt;
for (int d=0; d<sDIM; ++d)
{
gpt.x[d] = elx[d][pi*(D1D-1)];
gpt.jac[d] = 0.0;
gpt.hes[d] = 0.0;
for (int k=0; k<D1D; ++k)
{
gpt.jac[d] += wt[D1D +k]*elx[d][k];
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
}
}
const double *const pt_x = gpt.x;
const double *const jac = gpt.jac;
const double *const hes = gpt.hes;
double resid[sDIM], steep, sr;
resid[0] = fpt->x[0] - pt_x[0];
resid[1] = fpt->x[1] - pt_x[1];
resid[2] = fpt->x[2] - pt_x[2];
steep = jac[0]*resid[0] + jac[1]*resid[1] +
jac[2]*resid[2];
sr = steep*tmp->r;
if (!reject_prior_step_q(fpt, resid, tmp, tol))
{
if (sr<0)
{
const double rhess = resid[0]*hes[0] +
resid[1]*hes[1] +
resid[2]*hes[2];
newton_edge(fpt, jac, rhess,
resid, 0, tmp, tol);
}
else // sr==0
{
fpt->r = tmp->r;
fpt->dist2p = 0;
fpt->flags = tmp->flags | CONVERGED_FLAG;
}
}
}
MFEM_SYNC_THREAD;
break;
} // case 1
} //switch
if (fpt->flags & CONVERGED_FLAG)
{
break;
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
*tmp = *fpt;
}
MFEM_SYNC_THREAD;
} // for step<50
} // findpts_el
bool converged_internal =
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
(fpt->dist2<dist2tol);
if (*code_i==CODE_NOT_FOUND || converged_internal ||
fpt->dist2<*dist2_i)
{
MFEM_FOREACH_THREAD(j,x,1)
{
*(el_base+i) = el;
*code_i = converged_internal?CODE_INTERNAL:CODE_BORDER;
*dist2_i = fpt->dist2;
*(r_base+i) = fpt->r;
}
MFEM_SYNC_THREAD;
if (converged_internal)
{
break;
}
}
} // findpts_local
} // obbox_test
} // elp
});
}
void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt)
{
if (npt == 0)
{
return;
}
MFEM_VERIFY(spacedim==3 && dim == 1,"Function for 3D edges only");
bool use_dev = point_pos.UseDevice();
auto pp = point_pos.Read(use_dev);
auto pgslm = gsl_mesh.Read(use_dev);
auto pwt = DEV.wtend.Read(use_dev);
auto pbb = DEV.bb.Read(use_dev);
auto plhm = DEV.lh_min.Read(use_dev);
auto plhf = DEV.lh_fac.Read(use_dev);
auto plho = DEV.lh_offset.ReadWrite(use_dev);
auto pcode = code.Write(use_dev);
auto pelem = elem.Write(use_dev);
auto pref = ref.Write(use_dev);
auto pdist = dist.Write(use_dev);
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
switch (DEV.dof1d)
{
case 2:
return FindPointsEdgeLocal3D_Kernel<2>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
case 3:
return FindPointsEdgeLocal3D_Kernel<3>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
case 4:
return FindPointsEdgeLocal3D_Kernel<4>(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc);
default:
return FindPointsEdgeLocal3D_Kernel(
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
}
}
#undef rDIM2
#undef sDIM2
#undef rDIM
#undef sDIM
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::FindPointsEdgeLocal3( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt ) {} ;
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
File diff suppressed because it is too large Load Diff
+157
View File
@@ -0,0 +1,157 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/kernels.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j=0; j<p_Nq; ++j)
{
p_i *= j==i ? 1 : x-z[j];
}
p0[i] = lagrangeCoeff[i] * p_i;
}
template<int T_D1D = 0>
static void InterpolateLocal1DKernel(const double *const gf_in,
int *const el,
double *const r,
double *const int_out,
const int npt,
const int nfields,
double *gll1D,
double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_Nq = D1D;
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
// for each point of the npt points, create a thread block of size dof1Dsol
mfem::forall_2D(npt, D1D, 1, [=] MFEM_HOST_DEVICE (int i)
{
MFEM_SHARED double wtr[MD1];
MFEM_SHARED double sums[MD1];
// Evaluate basis functions at the reference space coordinates
MFEM_FOREACH_THREAD(j,x,D1D)
{
lagrange_eval(wtr, r[i], j, p_Nq, gll1D, lagcoeff);
}
MFEM_SYNC_THREAD;
for (int fld=0; fld<nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Nq + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
const int elemOffset = el[i]*nfields*p_Nq + fld*p_Nq;
MFEM_FOREACH_THREAD(j,x,D1D)
{
sums[j] = wtr[j] * gf_in[elemOffset + j];
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
double sumv = 0.0;
// sum the contributions of each lagrange polynomial
for (int jj=0; jj<D1D; ++jj)
{
sumv += sums[jj];
}
int_out[fld*npt + i] = sumv;
}
MFEM_SYNC_THREAD;
}
});
}
void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt,
int ncomp,
int dof1Dsol )
{
MFEM_VERIFY(dim == 1, "Kernel for edges only.");
if (npt == 0) { return; }
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp,
pgll, plcf, dof1Dsol);
}
}
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::InterpolateLocal1(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol) {};
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
+19 -19
View File
@@ -52,8 +52,6 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
double *const int_out,
const int npt,
const int ncomp,
const int nel,
const int gf_offset,
double *gll1D,
double *lagcoeff,
const int pN = 0)
@@ -64,6 +62,8 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
const int p_Np = D1D*D1D;
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
mfem::forall_2D(npt, D1D, D1D, [=] MFEM_HOST_DEVICE (int i)
{
@@ -82,9 +82,9 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
for (int fld = 0; fld < Nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
//if using R->Mult for L -> E-Vec use below: NDOFSxVDIMxNEL
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Np + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
MFEM_FOREACH_THREAD(j,x,D1D)
{
@@ -120,32 +120,32 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1Dsol)
int dof1Dsol)
{
if (npt == 0) { return; }
const int gf_offset = field_in.Size()/ncomp;
auto pfin = field_in.Read();
auto pgsl = gsl_elem_dev_l.ReadWrite();
auto pgslr = gsl_ref_l.ReadWrite();
auto pfout = field_out.Write();
auto pgll = DEV.gll1d_sol.ReadWrite();
auto plcf = DEV.lagcoeff_sol.ReadWrite();
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf, dof1Dsol);
}
}
@@ -160,7 +160,7 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1Dsol) {};
int dof1Dsol) {};
#endif
} // namespace mfem
+18 -19
View File
@@ -52,8 +52,6 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
double *const int_out,
const int npt,
const int ncomp,
const int nel,
const int gf_offset,
double *gll1D,
double *lagcoeff,
const int pN = 0)
@@ -84,9 +82,9 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
for (int fld = 0; fld < Nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
//if using R->Mult for L -> E-Vec use below.
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Np + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
MFEM_FOREACH_THREAD(j,x,D1D)
{
@@ -125,37 +123,38 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1Dsol)
int dof1Dsol)
{
if (npt == 0) { return; }
const int gf_offset = field_in.Size()/ncomp;
auto pfin = field_in.Read();
auto pgsle = gsl_elem_dev_l.ReadWrite();
auto pgslr = gsl_ref_l.ReadWrite();
auto pfout = field_out.Write();
auto pgll = DEV.gll1d_sol.ReadWrite();
auto plcf = DEV.lagcoeff_sol.ReadWrite();
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsle = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf);
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
npt, ncomp,
pgll, plcf, dof1Dsol);
}
}
#undef MAXC
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
@@ -165,7 +164,7 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int nel, int dof1Dsol) {};
int dof1Dsol) {};
#endif
} // namespace mfem
+8 -5
View File
@@ -1004,13 +1004,16 @@ inline void SmemPADiffusionApply3D(const int NE,
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
MFEM_VERIFY(D1D <= max_d1d, "");
MFEM_VERIFY(Q1D <= max_q1d, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto g = Reshape(g_.Read(), Q1D, D1D);
const auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
const auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_VERIFY(D1D <= Q1D, "THREAD_DIRECT requires D1D <= Q1D");
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_3D<T_Q1D*T_Q1D*T_Q1D>(NE,
Q1D, Q1D, Q1D,
[=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
+18 -8
View File
@@ -197,15 +197,21 @@ static void EAHdivAssemble3D(const int NE,
// Assemble (one row per thread)
MFEM_FOREACH_THREAD(idx_i, x, NDOF)
{
// NOTE: due to an llvm backend bug, usage of the modulus operator
// has been removed from this foreach section.
const int ic = idx_i / NDOF_C;
const int idx_ii = idx_i % NDOF_C;
const int idx_ii = idx_i - ic * NDOF_C; // idx_i % NDOF_C
const int nx_i = (ic == 0) ? D1D : D1D-1;
const int ny_i = (ic == 1) ? D1D : D1D-1;
const int ix = idx_ii % nx_i;
const int iy = (idx_ii / nx_i) % ny_i;
const int iz = (idx_ii / nx_i) / ny_i;
const int qx_i = idx_ii / nx_i;
const int ix = idx_ii - qx_i * nx_i; // idx_ii % nx_i
const int qy_i = qx_i / ny_i;
const int iy = qx_i - qy_i * ny_i; // (idx_ii / nx_i) % ny_i
const int iz = qy_i; // (idx_ii / nx_i) / ny_i
const real_t (&Bi1)[MQ1][MD1] = (ic == 0) ? r_Bc : r_Bo;
const real_t (&Bi2)[MQ1][MD1] = (ic == 1) ? r_Bc : r_Bo;
@@ -214,14 +220,18 @@ static void EAHdivAssemble3D(const int NE,
for (int idx_j = 0; idx_j < NDOF; ++idx_j)
{
const int jc = idx_j / NDOF_C;
const int idx_jj = idx_j % NDOF_C;
const int idx_jj = idx_j - jc * NDOF_C; // idx_j % NDOF_C
const int nx_j = (jc == 0) ? D1D : D1D-1;
const int ny_j = (jc == 1) ? D1D : D1D-1;
const int jx = idx_jj % nx_j;
const int jy = (idx_jj / nx_j) % ny_j;
const int jz = (idx_jj / nx_j) / ny_j;
const int qx_j = idx_jj / nx_j;
const int jx = idx_jj - qx_j * nx_j; // idx_jj % nx_j
const int qy_j = qx_j / ny_j;
const int jy = qx_j - qy_j * ny_j; // (idx_jj / nx_j) % ny_j
const int jz = qy_j; // (idx_jj / nx_j) / ny_j
const real_t (&Bj1)[MQ1][MD1] = (jc == 0) ? r_Bc : r_Bo;
const real_t (&Bj2)[MQ1][MD1] = (jc == 1) ? r_Bc : r_Bo;
+81 -53
View File
@@ -181,6 +181,12 @@ constexpr int NBZ(int D1D)
{
return ipow(2, D(D1D) >= 0 ? D(D1D) : 0);
}
constexpr int NBZ3D(int MDQ)
{
return MDQ > 0 ? std::min<int>(
(128 + MDQ * MDQ * MDQ - 1) / (MDQ * MDQ * MDQ), 64)
: 1;
}
}
// Shared memory PA Mass Diagonal 2D kernel
@@ -804,19 +810,23 @@ void PAMassApply3D_Element(const int e,
}
}
template<int T_D1D, int T_Q1D, bool ACCUMULATE = true>
MFEM_HOST_DEVICE inline
void SmemPAMassApply3D_Element(const int e,
const int NE,
const real_t *b_,
const real_t *d_,
const real_t *x_,
real_t *y_,
const int d1d = 0,
const int q1d = 0)
template <int T_D1D, int T_Q1D, int TBATCH, bool ACCUMULATE = true>
MFEM_HOST_DEVICE inline void
SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
const real_t *d_, const real_t *x_, real_t *y_,
int d1d = 0, int q1d = 0)
{
constexpr int D1D = T_D1D ? T_D1D : d1d;
constexpr int Q1D = T_Q1D ? T_Q1D : q1d;
static_assert(TBATCH > 0, "TBATCH must be positive");
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
constexpr int tbatch = TBATCH;
const int tidz = MFEM_THREAD_ID(z);
#else
// host always batch size 1
constexpr int tbatch = 1;
constexpr int tidz = 0;
#endif
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
@@ -829,33 +839,37 @@ void SmemPAMassApply3D_Element(const int e,
MFEM_SHARED real_t sDQ[MQ1*MD1];
real_t (*B)[MD1] = (real_t (*)[MD1]) sDQ;
real_t (*Bt)[MQ1] = (real_t (*)[MQ1]) sDQ;
MFEM_SHARED real_t sm0[MDQ*MDQ*MDQ];
MFEM_SHARED real_t sm1[MDQ*MDQ*MDQ];
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm0;
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) sm1;
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm0;
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm1;
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) sm0;
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm1;
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_SHARED real_t sm0[tbatch][MDQ*MDQ*MDQ];
MFEM_SHARED real_t sm1[tbatch][MDQ*MDQ*MDQ];
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+tidz);
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) (sm1+tidz);
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm0+tidz);
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm1+tidz);
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) (sm0+tidz);
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm1+tidz);
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(dx, x, D1D)
{
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
X[dz][dy][dx] = x(dx, dy, dz, e);
}
}
MFEM_FOREACH_THREAD(dx,x,Q1D)
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
B[dx][dy] = b(dx,dy);
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -880,9 +894,9 @@ void SmemPAMassApply3D_Element(const int e,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -907,9 +921,9 @@ void SmemPAMassApply3D_Element(const int e,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -929,22 +943,22 @@ void SmemPAMassApply3D_Element(const int e,
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
{
QQQ[qz][qy][qx] = u[qz] * d(qx,qy,qz,e);
QQQ[qz][qy][qx] = u[qz] * d(qx, qy, qz, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(di,y,D1D)
if (tidz == 0)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
MFEM_FOREACH_THREAD(di, y, D1D)
{
Bt[di][q] = b(q,di);
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[di][q] = b(q, di); }
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -969,9 +983,9 @@ void SmemPAMassApply3D_Element(const int e,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -996,9 +1010,9 @@ void SmemPAMassApply3D_Element(const int e,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(dx, x, D1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -1020,11 +1034,11 @@ void SmemPAMassApply3D_Element(const int e,
{
if (ACCUMULATE)
{
y(dx,dy,dz,e) += u[dz];
y(dx, dy, dz, e) += u[dz];
}
else
{
y(dx,dy,dz,e) = u[dz];
y(dx, dy, dz, e) = u[dz];
}
}
}
@@ -1115,8 +1129,8 @@ inline void PAMassApply3D(const int NE,
});
}
// Shared memory PA Mass Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
// Shared memory PA Mass Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0, int TBATCH=1>
inline void SmemPAMassApply3D(const int NE,
const Array<real_t> &b_,
const Array<real_t> &bt_,
@@ -1126,6 +1140,9 @@ inline void SmemPAMassApply3D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
static_assert(T_D1D > 0, "T_D1D must be positive");
static_assert(T_Q1D > 0, "T_Q1D must be positive");
static_assert(TBATCH > 0, "TBATCH must be positive");
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -1133,13 +1150,15 @@ inline void SmemPAMassApply3D(const int NE,
const int max_d1d = T_D1D ? T_D1D : DeviceDofQuadLimits::Get().MAX_D1D;
MFEM_VERIFY(D1D <= max_d1d, "");
MFEM_VERIFY(Q1D <= max_q1d, "");
auto b = b_.Read();
auto d = d_.Read();
auto x = x_.Read();
const auto b = b_.Read();
const auto d = d_.Read();
const auto x = x_.Read();
auto y = y_.ReadWrite();
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_2D_batch<T_Q1D * T_Q1D * TBATCH>(NE, Q1D, Q1D, TBATCH,
[=] MFEM_HOST_DEVICE(int e)
{
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
internal::SmemPAMassApply3D_Element<T_D1D, T_Q1D, TBATCH>(e, NE, b, d, x,
y, d1d, q1d);
});
}
@@ -1156,8 +1175,8 @@ inline void EAMassAssemble1D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
const auto B = Reshape(basis.Read(), Q1D, D1D);
const auto D = Reshape(padata.Read(), Q1D, NE);
auto M = Reshape(add ? eadata.ReadWrite() : eadata.Write(), D1D, D1D, NE);
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
{
@@ -1394,7 +1413,16 @@ ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassApply3D<T_D1D, T_Q1D>; }
else if constexpr (DIM == 3)
{
constexpr int MDQ = T_D1D >= T_Q1D ? T_D1D : T_Q1D;
// max 64 threads in z limit in cuda and hip
if constexpr (MDQ > 0)
{
return internal::SmemPAMassApply3D<T_D1D, T_Q1D,
internal::mass::NBZ3D(MDQ)>;
}
}
MFEM_ABORT("");
}
+5 -5
View File
@@ -54,7 +54,7 @@ void SmemPAVectorDiffusionApply2D(const int NE,
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -120,7 +120,7 @@ void SmemPAVectorDiffusionApply3D(const int NE,
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -171,15 +171,15 @@ template<int DIM, int T_SDIM, int T_D1D, int T_Q1D>
VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
{
if (DIM == 2)
if constexpr (DIM == 2)
{
return internal::SmemPAVectorDiffusionApply2D<T_SDIM, T_D1D, T_Q1D>;
}
else if (DIM == 3)
else if constexpr (DIM == 3)
{
return internal::SmemPAVectorDiffusionApply3D<T_SDIM, T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
MFEM_ABORT("Unsupported kernel");
}
inline VectorDiffusionIntegrator::ApplyKernelType
+5 -5
View File
@@ -51,7 +51,7 @@ void SmemPAVectorMassApply2D(const int NE,
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -119,7 +119,7 @@ void SmemPAVectorMassApply3D(const int NE,
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
@@ -182,15 +182,15 @@ template<int DIM, int T_D1D, int T_Q1D>
VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Kernel()
{
if (DIM == 2)
if constexpr (DIM == 2)
{
return internal::SmemPAVectorMassApply2D<T_D1D,T_Q1D>;
}
else if (DIM == 3)
else if constexpr (DIM == 3)
{
return internal::SmemPAVectorMassApply3D<T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
MFEM_ABORT("Unsupported kernel");
}
inline VectorMassIntegrator::VectorMassAddMultPAType
+6 -10
View File
@@ -301,18 +301,14 @@ template <int DIM, int T_D1D, int T_Q1D>
DomainLFIntegrator::AssembleKernelType
DomainLFIntegrator::AssembleKernels::Kernel()
{
switch (DIM)
{
case 1:
return DLFEvalAssemble1D<T_D1D, T_Q1D>;
case 2:
return DLFEvalAssemble2D<T_D1D, T_Q1D>;
case 3:
return DLFEvalAssemble3D<T_D1D, T_Q1D>;
}
if constexpr (DIM == 1) { return DLFEvalAssemble1D<T_D1D, T_Q1D>; }
if constexpr (DIM == 2) { return DLFEvalAssemble2D<T_D1D, T_Q1D>; }
if constexpr (DIM == 3) { return DLFEvalAssemble3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
#endif // MFEM_LININTEG_DOMAIN_KERNELS_HPP
+811 -327
View File
File diff suppressed because it is too large Load Diff
+63 -64
View File
@@ -43,56 +43,52 @@ public:
index = i;
}
void Set3w(const real_t x1, const real_t x2, const real_t x3, const real_t w)
{ x = x1; y = x2; z = x3; weight = w; }
void Set2w(const real_t x1, const real_t x2, const real_t w)
{ x = x1; y = x2; weight = w; }
void Set1w(const real_t x1, const real_t w)
{ x = x1; weight = w; }
void Set3w(const real_t *p) { Set3w(p[0], p[1], p[2], p[3]); }
void Set2w(const real_t *p) { Set2w(p[0], p[1], p[2]); }
void Set1w(const real_t *p) { Set1w(p[0], p[1]); }
void Set3(const real_t x1, const real_t x2, const real_t x3)
{ x = x1; y = x2; z = x3; }
void Set2(const real_t x1, const real_t x2)
{ x = x1; y = x2; }
void Set1(const real_t x1)
{ x = x1; }
void Set3(const real_t *p) { Set3(p[0], p[1], p[2]); }
void Set2(const real_t *p) { Set2(p[0], p[1]); }
void Set1(const real_t *p) { Set1(p[0]); }
void Set(const real_t x1, const real_t x2, const real_t x3, const real_t w)
{ Set3w(x1, x2, x3, w); }
void Set(const real_t *p, const int dim)
{
MFEM_ASSERT(1 <= dim && dim <= 3, "invalid dim: " << dim);
x = p[0];
if (dim > 1)
switch (dim)
{
y = p[1];
if (dim > 2)
{
z = p[2];
}
case 3: Set3(p); break;
case 2: Set2(p); break;
case 1: Set1(p); break;
}
}
void Get(real_t *p, const int dim) const
{
MFEM_ASSERT(1 <= dim && dim <= 3, "invalid dim: " << dim);
p[0] = x;
if (dim > 1)
switch (dim)
{
p[1] = y;
if (dim > 2)
{
p[2] = z;
}
case 3: p[2] = z;
case 2: p[1] = y;
case 1: p[0] = x;
}
}
void Set(const real_t x1, const real_t x2, const real_t x3, const real_t w)
{ x = x1; y = x2; z = x3; weight = w; }
void Set3w(const real_t *p) { x = p[0]; y = p[1]; z = p[2]; weight = p[3]; }
void Set3(const real_t x1, const real_t x2, const real_t x3)
{ x = x1; y = x2; z = x3; }
void Set3(const real_t *p) { x = p[0]; y = p[1]; z = p[2]; }
void Set2w(const real_t x1, const real_t x2, const real_t w)
{ x = x1; y = x2; weight = w; }
void Set2w(const real_t *p) { x = p[0]; y = p[1]; weight = p[2]; }
void Set2(const real_t x1, const real_t x2) { x = x1; y = x2; }
void Set2(const real_t *p) { x = p[0]; y = p[1]; }
void Set1w(const real_t x1, const real_t w) { x = x1; weight = w; }
void Set1w(const real_t *p) { x = p[0]; weight = p[1]; }
};
/// Class for an integration rule - an Array of IntegrationPoint.
@@ -125,18 +121,6 @@ private:
void AddTriPoints3b(const int off, const real_t b, const real_t weight)
{ AddTriPoints3(off, (1. - b)/2., b, weight); }
void AddTriPoints3R(const int off, const real_t a, const real_t b,
const real_t c, const real_t weight)
{
IntPoint(off + 0).Set2w(a, b, weight);
IntPoint(off + 1).Set2w(c, a, weight);
IntPoint(off + 2).Set2w(b, c, weight);
}
void AddTriPoints3R(const int off, const real_t a, const real_t b,
const real_t weight)
{ AddTriPoints3R(off, a, b, 1. - a - b, weight); }
void AddTriPoints6(const int off, const real_t a, const real_t b,
const real_t c, const real_t weight)
{
@@ -183,14 +167,6 @@ private:
AddTetPoints3(off + 1, a, 1. - 3.*a, weight);
}
// given b, add the permutations of (a,a,a,b), where 3*a + b = 1
void AddTetPoints4b(const int off, const real_t b, const real_t weight)
{
const real_t a = (1. - b)/3.;
IntPoint(off).Set(a, a, a, weight);
AddTetPoints3(off + 1, a, b, weight);
}
// add the permutations of (a,a,b,b), 2*(a + b) = 1
void AddTetPoints6(const int off, const real_t a, const real_t weight)
{
@@ -209,14 +185,37 @@ private:
AddTetPoints6(off + 6, a, bc, cb, weight);
}
// given (b,c), add the permutations of (a,a,b,c), 2*a + b + c = 1
void AddTetPoints12bc(const int off, const real_t b, const real_t c,
const real_t weight)
// add all 24 permutations of (a,b,c,d) where a+b+c+d = 1, all distinct
void AddTetPoints24(const int off, const real_t a, const real_t b,
const real_t c, const real_t weight)
{
const real_t a = (1. - b - c)/2.;
AddTetPoints3(off, a, b, weight);
AddTetPoints3(off + 3, a, c, weight);
AddTetPoints6(off + 6, a, b, c, weight);
const real_t d = 1. - a - b - c;
// all 24 permutations of 4 distinct barycentric coordinates
// permuting which coordinate goes to x, y, z (4th is 1-x-y-z)
IntPoint(off + 0).Set(a, b, c, weight);
IntPoint(off + 1).Set(a, b, d, weight);
IntPoint(off + 2).Set(a, c, b, weight);
IntPoint(off + 3).Set(a, c, d, weight);
IntPoint(off + 4).Set(a, d, b, weight);
IntPoint(off + 5).Set(a, d, c, weight);
IntPoint(off + 6).Set(b, a, c, weight);
IntPoint(off + 7).Set(b, a, d, weight);
IntPoint(off + 8).Set(b, c, a, weight);
IntPoint(off + 9).Set(b, c, d, weight);
IntPoint(off + 10).Set(b, d, a, weight);
IntPoint(off + 11).Set(b, d, c, weight);
IntPoint(off + 12).Set(c, a, b, weight);
IntPoint(off + 13).Set(c, a, d, weight);
IntPoint(off + 14).Set(c, b, a, weight);
IntPoint(off + 15).Set(c, b, d, weight);
IntPoint(off + 16).Set(c, d, a, weight);
IntPoint(off + 17).Set(c, d, b, weight);
IntPoint(off + 18).Set(d, a, b, weight);
IntPoint(off + 19).Set(d, a, c, weight);
IntPoint(off + 20).Set(d, b, a, weight);
IntPoint(off + 21).Set(d, b, c, weight);
IntPoint(off + 22).Set(d, c, a, weight);
IntPoint(off + 23).Set(d, c, b, weight);
}
public:
+3 -1
View File
@@ -297,7 +297,8 @@ void LinearForm::Assemble()
tr = mesh->GetBdrFaceTransformations(i);
if (tr != NULL)
{
fes -> GetElementVDofs (tr -> Elem1No, vdofs);
mfem::DofTransformation doftrans;
fes -> GetElementVDofs (tr -> Elem1No, vdofs, doftrans);
for (int k = 0; k < boundary_face_integs.Size(); k++)
{
if (boundary_face_integs_marker[k] &&
@@ -307,6 +308,7 @@ void LinearForm::Assemble()
boundary_face_integs[k]->
AssembleRHSElementVect(*fes->GetFE(tr->Elem1No),
*tr, elemvect);
doftrans.TransformDual(elemvect);
AddElementVector (vdofs, elemvect);
}
}
+2 -2
View File
@@ -164,8 +164,8 @@ private:
public:
/// Constructs the domain integrator $ (Q, \nabla v) $
DomainLFGradIntegrator(VectorCoefficient &QF)
: DeltaLFIntegrator(QF), Q(QF) { }
DomainLFGradIntegrator(VectorCoefficient &QF, const IntegrationRule *ir = NULL)
: DeltaLFIntegrator(QF, ir), Q(QF) { }
bool SupportsDevice() const override { return true; }
+6 -5
View File
@@ -158,15 +158,16 @@ void LORBase::ConstructLocalDofPermutation(Array<int> &perm_) const
int i;
i = dofmap_lor[off_lor + i1 + i2*2];
int s1 = i < 0 ? -1 : 1;
int idof_lor = vdof_lor[absdof(i)];
int idof_lor = vdof_lor[UnsignIndex(i)];
i = dofmap_ho[off_ho + i1*n1 + i2*n2];
int s2 = i < 0 ? -1 : 1;
int idof_ho = vdof_ho[absdof(i)];
int idof_ho = vdof_ho[UnsignIndex(i)];
int s3 = idof_lor < 0 ? -1 : 1;
int s4 = idof_ho < 0 ? -1 : 1;
int s = s1*s2*s3*s4;
i = absdof(idof_ho);
perm_[absdof(idof_lor)] = s < 0 ? -1-absdof(i) : absdof(i);
i = UnsignIndex(idof_ho);
perm_[UnsignIndex(idof_lor)] = s < 0 ? -1-UnsignIndex(i) :
UnsignIndex(i);
}
}
};
@@ -232,7 +233,7 @@ void LORBase::ConstructDofPermutation() const
int j = l_perm[i];
int s = j < 0 ? -1 : 1;
int t_i = pfes_lor->GetLocalTDofNumber(i);
int t_j = pfes_ho->GetLocalTDofNumber(absdof(j));
int t_j = pfes_ho->GetLocalTDofNumber(UnsignIndex(j));
// Either t_i and t_j both -1, or both non-negative
if ((t_i < 0 && t_j >=0) || (t_j < 0 && t_i >= 0))
{
-2
View File
@@ -57,8 +57,6 @@ private:
/// values (after temporarily changing them for LOR assembly).
void ResetIntegrationRules(GetIntegratorsFn get_integrators);
static inline int absdof(int i) { return i < 0 ? -1-i : i; }
protected:
enum FESpaceType { H1, ND, RT, L2, INVALID };
+4 -4
View File
@@ -284,12 +284,12 @@ GeometricMultigrid::GeometricMultigrid(
ownedProlongations.SetSize(nlevels - 1);
ownedProlongations = have_ess_bdr;
if (have_ess_bdr)
essentialTrueDofs.SetSize(nlevels);
for (int level = 0; level < nlevels; ++level)
{
essentialTrueDofs.SetSize(nlevels);
for (int level = 0; level < nlevels; ++level)
essentialTrueDofs[level] = new Array<int>;
if (have_ess_bdr)
{
essentialTrueDofs[level] = new Array<int>;
fespaces.GetFESpaceAtLevel(level).GetEssentialTrueDofs(
ess_bdr, *essentialTrueDofs[level]);
}
+1 -2
View File
@@ -187,8 +187,7 @@ public:
/// mesh boundary element attributes that define the essential DOFs.
///
/// If @a ess_bdr is empty, or all its entries are 0, then no essential
/// boundary conditions are imposed and the protected array essentialTrueDofs
/// remains empty.
/// boundary conditions are imposed.
GeometricMultigrid(const FiniteElementSpaceHierarchy& fespaces_,
const Array<int> &ess_bdr);
+72 -50
View File
@@ -349,6 +349,7 @@ void ParFiniteElementSpace::GetGroupComm(
}
}
bool have_sign_flips = false;
if (g_ldof_sign)
{
g_ldof_sign->SetSize(GetNDofs());
@@ -424,10 +425,11 @@ void ParFiniteElementSpace::GetGroupComm(
{
if (ind[l] < 0)
{
dofs[l] = m + (-1-ind[l]);
dofs[l] = m + FlipIndexSign(ind[l]);
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -462,10 +464,11 @@ void ParFiniteElementSpace::GetGroupComm(
{
if (ind[l] < 0)
{
dofs[l] = m + (-1-ind[l]);
dofs[l] = m + FlipIndexSign(ind[l]);
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -500,10 +503,11 @@ void ParFiniteElementSpace::GetGroupComm(
{
if (ind[l] < 0)
{
dofs[l] = m + (-1-ind[l]);
dofs[l] = m + FlipIndexSign(ind[l]);
if (g_ldof_sign)
{
(*g_ldof_sign)[dofs[l]] = -1;
have_sign_flips = true;
}
}
else
@@ -527,27 +531,33 @@ void ParFiniteElementSpace::GetGroupComm(
group_ldof.GetI()[gr+1] = group_ldof_counter;
}
if (g_ldof_sign && have_sign_flips == false)
{
g_ldof_sign->DeleteAll();
}
gc.Finalize();
}
void ParFiniteElementSpace::ApplyLDofSigns(Array<int> &dofs) const
{
MFEM_ASSERT(Conforming(), "wrong code path");
if (!HaveDofSigns()) { return; }
for (int i = 0; i < dofs.Size(); i++)
{
if (dofs[i] < 0)
{
if (ldof_sign[-1-dofs[i]] < 0)
if (ldof_sign[FlipIndexSign(dofs[i])] < 0)
{
dofs[i] = -1-dofs[i];
dofs[i] = FlipIndexSign(dofs[i]);
}
}
else
{
if (ldof_sign[dofs[i]] < 0)
{
dofs[i] = -1-dofs[i];
dofs[i] = FlipIndexSign(dofs[i]);
}
}
}
@@ -559,6 +569,24 @@ void ParFiniteElementSpace::ApplyLDofSigns(Table &el_dof) const
ApplyLDofSigns(all_dofs);
}
void ParFiniteElementSpace::ApplyDofSigns(real_t *h_data) const
{
if (!HaveDofSigns()) { return; }
const bool byvdim = (ordering == Ordering::byVDIM);
for (int i = 0; i < ndofs; i++)
{
if (ldof_sign[i] < 0)
{
for (int d = 0; d < vdim; d++)
{
const int idx = byvdim ? d+vdim*i : i+ndofs*d;
h_data[idx] = -h_data[idx];
}
}
}
}
void ParFiniteElementSpace::GetElementDofs(int i, Array<int> &dofs,
DofTransformation &doftrans) const
{
@@ -699,7 +727,8 @@ void ParFiniteElementSpace::GetSharedEdgeDofs(
for (int i = 0; i < dofs.Size(); i++)
{
const int di = dofs[i];
dofs[i] = (di >= 0) ? rdofs[di] : -1-rdofs[-1-di];
dofs[i] = di >= 0 ? rdofs[di] :
FlipIndexSign(rdofs[FlipIndexSign(di)]);
}
}
}
@@ -723,7 +752,8 @@ void ParFiniteElementSpace::GetSharedTriangleDofs(
for (int i = 0; i < dofs.Size(); i++)
{
const int di = dofs[i];
dofs[i] = (di >= 0) ? rdofs[di] : -1-rdofs[-1-di];
dofs[i] = di >= 0 ? rdofs[di] :
FlipIndexSign(rdofs[FlipIndexSign(di)]);
}
}
}
@@ -747,7 +777,8 @@ void ParFiniteElementSpace::GetSharedQuadrilateralDofs(
for (int i = 0; i < dofs.Size(); i++)
{
const int di = dofs[i];
dofs[i] = (di >= 0) ? rdofs[di] : -1-rdofs[-1-di];
dofs[i] = (di >= 0) ? rdofs[di] :
FlipIndexSign(rdofs[FlipIndexSign(di)]);
}
}
}
@@ -1190,15 +1221,15 @@ void ParFiniteElementSpace::GetEssentialTrueDofsVar(const Array<int>
MFEM_VERIFY(IsVariableOrder() && R,
"GetEssentialTrueDofsVar is only for variable-order spaces");
true_ess_dofs.SetSize(R->Height(), Device::GetDeviceMemoryType());
true_ess_dofs.SetSize(R->Height());
true_ess_dofs.HostWrite();
true_ess_dofs = 0;
const int ntdofs = tdof2ldof.Size();
MFEM_VERIFY(vdim * ntdofs == R->NumRows() &&
vdim * ntdofs == true_ess_dofs.Size(), "");
MFEM_VERIFY(ldof_ltdof.Size() == ndofs && ess_dofs.Size() == vdim * ndofs, "");
true_ess_dofs = 0;
const bool bynodes = (ordering == Ordering::byNODES);
const int vdim_factor = bynodes ? 1 : vdim;
const int num_true_dofs = R->NumRows() / vdim;
@@ -1487,7 +1518,7 @@ void ParFiniteElementSpace::ExchangeFaceNbrData()
GetElementVDofs(my_elems[i], ldofs);
for (int j = 0; j < ldofs.Size(); j++)
{
int ldof = (ldofs[j] >= 0 ? ldofs[j] : -1-ldofs[j]);
int ldof = UnsignIndex(ldofs[j]);
if (ldof_marker[ldof] != fn)
{
@@ -1548,7 +1579,7 @@ void ParFiniteElementSpace::ExchangeFaceNbrData()
GetElementVDofs(my_elems[i], ldofs);
for (int j = 0; j < ldofs.Size(); j++)
{
int ldof = (ldofs[j] >= 0 ? ldofs[j] : -1-ldofs[j]);
int ldof = UnsignIndex(ldofs[j]);
if (ldof_marker[ldof] != fn)
{
@@ -1573,14 +1604,15 @@ void ParFiniteElementSpace::ExchangeFaceNbrData()
for (int i = 0; i < num_ldofs; i++)
{
int ldof = (ldofs_fn[i] >= 0 ? ldofs_fn[i] : -1-ldofs_fn[i]);
int ldof = UnsignIndex(ldofs_fn[i]);
ldof_marker[ldof] = i;
}
for ( ; j < j_end; j++)
{
int ldof = (send_J[j] >= 0 ? send_J[j] : -1-send_J[j]);
send_J[j] = (send_J[j] >= 0 ? ldof_marker[ldof] : -1-ldof_marker[ldof]);
const int ldof = UnsignIndex(send_J[j]);
send_J[j] = (send_J[j] >= 0 ? ldof_marker[ldof] :
FlipIndexSign(ldof_marker[ldof]));
}
}
@@ -1672,12 +1704,7 @@ void ParFiniteElementSpace::ExchangeFaceNbrData()
{
for (int j_end = face_nbr_ldof.GetI()[fn+1]; j < j_end; j++)
{
int ldof = face_nbr_ldof.GetJ()[j];
if (ldof < 0)
{
ldof = -1-ldof;
}
const int ldof = UnsignIndex(face_nbr_ldof.GetJ()[j]);
face_nbr_glob_dof_map[j] = dof_face_nbr_offsets[fn] + ldof;
}
}
@@ -1721,7 +1748,7 @@ void ParFiniteElementSpace::GetFaceNbrFaceVDofs(int i, Array<int> &vdofs) const
MFEM_ASSERT(Nonconforming() && i >= pmesh->GetNumFaces(), "");
int el1, el2, inf1, inf2;
pmesh->GetFaceElements(i, &el1, &el2);
el2 = -1 - el2;
el2 = FlipIndexSign(el2);
pmesh->GetFaceInfos(i, &inf1, &inf2);
MFEM_ASSERT(0 <= el2 && el2 < face_nbr_element_dof.Size(), "");
const int nd = face_nbr_element_dof.RowSize(el2);
@@ -1737,7 +1764,8 @@ void ParFiniteElementSpace::GetFaceNbrFaceVDofs(int i, Array<int> &vdofs) const
for (int j = 0; j < vdofs.Size(); j++)
{
const int ldof = vdofs[j];
vdofs[j] = (ldof >= 0) ? vol_vdofs[ldof] : -1-vol_vdofs[-1-ldof];
vdofs[j] = (ldof >= 0) ? vol_vdofs[ldof] :
FlipIndexSign(vol_vdofs[FlipIndexSign(ldof)]);
}
}
@@ -2061,8 +2089,8 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
for (int j = 0; j < ne; j++)
{
dofs[offset++] = (ind[j] >= 0) ? (first + ind[j])
/* */ : (-1 - (first + (-1 - ind[j])));
dofs[offset++] = (ind[j] >= 0) ? (first + ind[j]) :
FlipIndexSign(first + FlipIndexSign(ind[j]));
}
}
else
@@ -2072,8 +2100,8 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
const int *ind = fec->DofOrderForOrientation(Geometry::SEGMENT, Eo[i]);
for (int j = 0; j < ne; j++)
{
dofs[offset++] = (ind[j] >= 0) ? (first + ind[j])
/* */ : (-1 - (first + (-1 - ind[j])));
dofs[offset++] = (ind[j] >= 0) ? (first + ind[j]) :
FlipIndexSign(first + FlipIndexSign(ind[j]));
}
}
}
@@ -2866,7 +2894,7 @@ void NeighborRowMessage::Encode(int rank)
if (ind && (edof = ind[edof]) < 0)
{
edof = -1 - edof;
edof = FlipIndexSign(edof);
s = -1;
}
@@ -3067,10 +3095,10 @@ void NeighborRowMessage::Decode(int rank)
// If edof arrived with a negative index, flip it, and the scaling.
real_t s = (edof < 0) ? -1.0 : 1.0;
edof = (edof < 0) ? -1 - edof : edof;
edof = UnsignIndex(edof);
if (ind && (edof = ind[edof]) < 0)
{
edof = -1 - edof;
edof = FlipIndexSign(edof);
s *= -1.0;
}
@@ -3121,10 +3149,10 @@ void NeighborRowMessage::Decode(int rank)
// If edof arrived with a negative index, flip it, and the scaling.
s = (edof < 0) ? -1.0 : 1.0;
edof = (edof < 0) ? -1 - edof : edof;
edof = UnsignIndex(edof);
if (ind && (edof = ind[edof]) < 0)
{
edof = -1 - edof;
edof = FlipIndexSign(edof);
s *= -1.0;
}
@@ -4405,12 +4433,9 @@ ParFiniteElementSpace::RebalanceMatrix(int old_ndofs,
{
for (int j = 0; j < dofs.Size(); j++)
{
int row = DofToVDof(dofs[j], vd);
if (row < 0) { row = -1 - row; }
int col = DofToVDof(old_dofs[j], vd, old_ndofs);
if (col < 0) { col = -1 - col; }
const int row = UnsignIndex(DofToVDof(dofs[j], vd));
const int col = UnsignIndex(DofToVDof(old_dofs[j], vd,
old_ndofs));
i_diag[row] = col;
}
}
@@ -4435,9 +4460,7 @@ ParFiniteElementSpace::RebalanceMatrix(int old_ndofs,
{
for (int j = 0; j < dofs.Size(); j++)
{
int row = DofToVDof(dofs[j], vd);
if (row < 0) { row = -1 - row; }
const int row = UnsignIndex(DofToVDof(dofs[j], vd));
if (i_diag[row] == i_diag[row+1]) // diag row empty?
{
i_offd[row] = old_dofs[j + vd * dofs.Size()];
@@ -4546,9 +4569,9 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
{
const Embedding &emb = dtrans.embeddings[k];
int fine_rank = old_ranks[k];
int coarse_rank = (emb.parent < 0) ? (-1 - emb.parent)
: old_pncmesh->ElementRank(emb.parent);
const int fine_rank = old_ranks[k];
const int coarse_rank = (emb.parent < 0) ? FlipIndexSign(emb.parent)
: old_pncmesh->ElementRank(emb.parent);
if (coarse_rank != MyRank && fine_rank == MyRank)
{
@@ -4636,8 +4659,8 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
{
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
const int r = DofToVDof(dofs[i], vd);
const int m = UnsignIndex(r);
if (is_dg || !mark[m])
{
@@ -4686,8 +4709,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
{
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
const int m = UnsignIndex(DofToVDof(dofs[i], vd));
if (is_dg || !mark[m])
{
+14 -2
View File
@@ -340,8 +340,20 @@ public:
inline ParMesh *GetParMesh() const { return pmesh; }
int GetDofSign(int i)
{ return NURBSext || Nonconforming() ? 1 : ldof_sign[VDofToDof(i)]; }
/** @brief Return true if the parallel FE space has DOFs with signs opposite
of the DOFs in the respective serial FE space. */
bool HaveDofSigns() const { return ldof_sign.Size() != 0; }
/** @brief Apply the DOF signs to the given host data @a h_data which must be
of size GetVSize() if HaveDofSigns() is true. If HaveDofSigns() is false,
this method is no-op and returns immediately. */
void ApplyDofSigns(real_t *h_data) const;
/** @brief Return -1 if the given (vector) DOF @a i has a sign opposite of
the DOF in the respecive serial FE space. Otherwise, return 1. */
int GetDofSign(int i) const
{ return !HaveDofSigns() ? 1 : ldof_sign[VDofToDof(i)]; }
HYPRE_BigInt *GetDofOffsets() const { return dof_offsets; }
HYPRE_BigInt *GetTrueDofOffsets() const { return tdof_offsets; }
HYPRE_BigInt GlobalVSize() const
+66 -10
View File
@@ -80,6 +80,8 @@ ParGridFunction::ParGridFunction(ParMesh *pmesh, std::istream &input)
fes->GetOrdering());
delete fes;
fes = pfes;
pfes->ApplyDofSigns(HostReadWrite());
}
void ParGridFunction::Update()
@@ -545,6 +547,8 @@ void ParGridFunction::GetElementDofValues(int el, Vector &dof_vals) const
void ParGridFunction::ProjectCoefficient(Coefficient &coeff, ProjectType type)
{
MFEM_VERIFY(VectorDim() == 1,
"Cannot project scalar coefficient onto vector ParGridFunction");
DeltaCoefficient *delta_c = dynamic_cast<DeltaCoefficient *>(&coeff);
if (delta_c == NULL)
@@ -715,7 +719,8 @@ void ParGridFunction::ProjectCoefficientElementL2(VectorCoefficient &vcoeff)
}
void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
void ParGridFunction::ProjectDiscCoefficient(
std::variant<Coefficient*, VectorCoefficient*> coeff)
{
// local maximal element attribute for each dof
Array<int> ldof_attr;
@@ -761,6 +766,9 @@ void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
void ParGridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
{
MFEM_VERIFY(
VectorDim() == 1,
"Cannot project scalar coefficient onto a vector ParGridFunction");
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
@@ -786,6 +794,8 @@ void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &vcoeff,
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
// Number of zones that contain a given dof.
Array<int> zones_per_vdof;
AccumulateAndCountZones(vcoeff, type, zones_per_vdof);
@@ -858,6 +868,12 @@ void ParGridFunction::ProjectBdrCoefficient(
#endif
}
void ParGridFunction::ProjectBdrCoefficient(VectorCoefficient &vcoeff,
const Array<int> &attr)
{
ProjectBdrCoefficient(NULL, &vcoeff, attr);
}
void ParGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
const Array<int> &bdr_attr)
{
@@ -1068,18 +1084,17 @@ real_t ParGridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
void ParGridFunction::Save(std::ostream &os) const
{
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < size; i++)
{
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
}
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t *h_data = const_cast<real_t*>(HostRead());
pfes->ApplyDofSigns(h_data);
GridFunction::Save(os);
for (int i = 0; i < size; i++)
{
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
}
pfes->ApplyDofSigns(h_data);
}
void ParGridFunction::Save(const char *fname, int precision) const
@@ -1250,7 +1265,13 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
int *nfdofs = new int[NRanks];
int *nrdofs = new int[NRanks];
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t * h_data = const_cast<real_t *>(this->HostRead());
pfes->ApplyDofSigns(h_data); // temporarily flip the dof signs
values[0] = h_data;
nv[0] = pfes -> GetVSize();
@@ -1357,6 +1378,8 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
MPI_Send(h_data, nv[0], MPITypeMap<real_t>::mpi_type, 0, 460, MyComm);
}
pfes->ApplyDofSigns(h_data); // restore the original h_data
delete [] values;
delete [] nv;
delete [] nvdofs;
@@ -1568,6 +1591,39 @@ PLBound ParGridFunction::GetBounds(Vector &lower, Vector &upper,
return plb;
}
std::pair<real_t, real_t> ParGridFunction::EstimateFunctionMinimum(
const int vdim, const PLBound &plb, const int max_depth,
const real_t tol) const
{
std::pair<real_t, real_t> minmax =
GridFunction::EstimateFunctionMinimum(vdim, plb, max_depth, tol);
real_t glob_min_lower = minmax.first;
real_t glob_min_upper = minmax.second;
MPI_Allreduce(MPI_IN_PLACE, &glob_min_lower, 1,
MFEM_MPI_REAL_T, MPI_MIN, pfes->GetComm());
MPI_Allreduce(MPI_IN_PLACE, &glob_min_upper, 1,
MFEM_MPI_REAL_T, MPI_MIN, pfes->GetComm());
return std::make_pair(glob_min_lower, glob_min_upper);
}
std::pair<real_t, real_t> ParGridFunction::EstimateFunctionMaximum(
const int vdim, const PLBound &plb, const int max_depth,
const real_t tol) const
{
std::pair<real_t, real_t> minmax =
GridFunction::EstimateFunctionMaximum(vdim, plb, max_depth, tol);
real_t glob_max_lower = minmax.first;
real_t glob_max_upper = minmax.second;
MPI_Allreduce(MPI_IN_PLACE, &glob_max_lower, 1,
MFEM_MPI_REAL_T, MPI_MAX, pfes->GetComm());
MPI_Allreduce(MPI_IN_PLACE, &glob_max_upper, 1,
MFEM_MPI_REAL_T, MPI_MAX, pfes->GetComm());
return std::make_pair(glob_max_lower, glob_max_upper);
}
} // namespace mfem
#endif // MFEM_USE_MPI
+19 -7
View File
@@ -63,6 +63,12 @@ protected:
void ProjectBdrCoefficient(Coefficient *coeff[], VectorCoefficient *vcoeff,
const Array<int> &attr);
/** @brief Project a discontinuous (vector) coefficient as a grid function on
a continuous finite element space. The values in shared dofs are
determined from the element with maximal attribute. */
virtual void ProjectDiscCoefficient(
std::variant<Coefficient*, VectorCoefficient*> coeff) override;
public:
ParGridFunction() { pfes = NULL; }
@@ -268,11 +274,6 @@ public:
ProjectType type = ProjectType::DEFAULT) override;
using GridFunction::ProjectDiscCoefficient;
/** @brief Project a discontinuous vector coefficient as a grid function on
a continuous finite element space. The values in shared dofs are
determined from the element with maximal attribute. */
void ProjectDiscCoefficient(VectorCoefficient &coeff) override;
void ProjectDiscCoefficient(Coefficient &coeff, AvgType type) override;
void ProjectDiscCoefficient(VectorCoefficient &vcoeff, AvgType type) override;
@@ -280,8 +281,7 @@ public:
using GridFunction::ProjectBdrCoefficient;
void ProjectBdrCoefficient(VectorCoefficient &vcoeff,
const Array<int> &attr) override
{ ProjectBdrCoefficient(NULL, &vcoeff, attr); }
const Array<int> &attr) override;
void ProjectBdrCoefficient(Coefficient *coeff[],
const Array<int> &attr) override
@@ -609,6 +609,18 @@ public:
PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) const override;
/** @brief Estimate the GridFunction minimum across all elements. */
std::pair<real_t, real_t> EstimateFunctionMinimum(const int vdim,
const PLBound &plb,
const int max_depth,
const real_t tol) const override;
/** @brief Estimate the GridFunction maximum across all elements. */
std::pair<real_t, real_t> EstimateFunctionMaximum(const int vdim,
const PLBound &plb,
const int max_depth,
const real_t tol) const override;
/** Save the local portion of the ParGridFunction. This differs from the
serial GridFunction::Save in that it takes into account the signs of
the local dofs. */
+11 -5
View File
@@ -321,12 +321,17 @@ void ParL2FaceRestriction::DoubleValuedConformingMult(
const int vd = vdim;
const bool t = byvdim;
const int threshold = ndofs;
const int nsdofs = pfes.GetFaceNbrVSize();
const int nsdofs = pfes.GetFaceNbrVSize() / vd;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
auto d_x_shared = Reshape(face_nbr_data.Read(),
t?vd:nsdofs, t?nsdofs:vd);
const int ne_shared = nsdofs / elem_dofs;
const int nedof = elem_dofs;
// Note: the shape of face_nbr_data, as determined by
// ParFiniteElementSpace::ExchangeFaceNbrData, is (elem_dofs, vdim,
// ne_shared), independent of the ordering (byNODES or byVDIM) of the finite
// element space.
auto d_x_shared = Reshape(face_nbr_data.Read(), elem_dofs, vd, ne_shared);
auto d_y = Reshape(y.Write(), nface_dofs, vd, 2, nf);
mfem::forall(nfdofs, [=] MFEM_HOST_DEVICE (int i)
{
@@ -346,8 +351,9 @@ void ParL2FaceRestriction::DoubleValuedConformingMult(
}
else if (idx2>=threshold) // shared boundary
{
d_y(dof, c, 1, face) = d_x_shared(t?c:(idx2-threshold),
t?(idx2-threshold):c);
const int e_shared = (idx2 - threshold) / nedof;
const int i_shared = (idx2 - threshold) % nedof;
d_y(dof, c, 1, face) = d_x_shared(i_shared,c,e_shared);
}
else // true boundary
{
+1 -4
View File
@@ -271,10 +271,7 @@ inline void QuadratureFunction::GetValues(
const int s_offset = qspace->Offset(idx);
const int sl_size = qspace->Offset(idx + 1) - s_offset;
// Make the values matrix memory an alias of the quadrature function memory
Memory<real_t> &values_mem = values.GetMemory();
values_mem.Delete();
values_mem.MakeAlias(GetMemory(), vdim*s_offset, vdim*sl_size);
values.SetSize(vdim, sl_size);
values.MakeRef(GetMemory(), vdim*s_offset, vdim, sl_size);
}
inline void QuadratureFunction::GetValues(
+8 -9
View File
@@ -334,17 +334,16 @@ template<int DIM, int SDIM, int D1D, int Q1D>
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1)
if constexpr (DIM == 1)
{
if (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 2>; }
else if (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 3>; }
else { MFEM_ABORT(""); }
if constexpr (SDIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if constexpr (SDIM == 2) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 2>; }
else if constexpr (SDIM == 3) { return internal::quadrature_interpolator::Det1DSurface<D1D, Q1D, 3>; }
}
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
else if constexpr (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if constexpr (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
MFEM_ABORT("");
}
/// @endcond

Some files were not shown because too many files have changed in this diff Show More