Compare commits

...
Author SHA1 Message Date
Eric B. Chin 5a954e2b48 Merge branch 'master' into lor-consistent-mass-dev 2026-06-24 11:16:52 -07:00
Tzanio Kolev 867c0a0e7e Merge pull request #5174 from mfem/tmop-mesh-validity
Ensuring mesh validity during mesh optimization using bounds on Jacobian determinant
2026-06-24 11:02:33 -07:00
Tzanio Kolev 270c4d5175 Merge pull request #5374 from mfem/fix-macos-runner
Fix MacOS runner issue
2026-06-24 10:59:51 -07:00
Veselin Dobrev 0779fbbc72 Merge pull request #5302 from lindsayad/bel-sorting
ReorderElements: sort boundary elements by face index after reordering
2026-06-24 10:31:39 -07:00
Eric B. Chin 7d80dfd93d Merge branch 'master' into lor-consistent-mass-dev 2026-06-23 15:16:33 -07:00
Eric B. Chin a4638d6d61 fix msan issue 2026-06-23 14:54:54 -07:00
Tara Drwenski 5cd7f88da0 Remove XCode version setup which breaks on runner update 2026-06-23 11:46:48 -07:00
Tzanio Kolev cc6a34b575 Merge pull request #5358 from mfem/ci-num-tasks
In CI, auto-detect the number of build tasks
2026-06-23 11:45:03 -07:00
Eric B. Chin 40ae2ffb7b clean up formatting changes 2026-06-22 22:36:44 -07:00
Eric B. Chin b500757b68 formatting 2026-06-22 22:19:30 -07:00
Eric B. Chin 445a8ba801 Merge branch 'master' into lor-consistent-mass-dev 2026-06-22 22:18:25 -07:00
Eric B. Chin 519e9185df make sure destruction happens in the right order; avoid unneded copy 2026-06-22 17:15:01 -07:00
Eric B. Chin 2d98f8c8ae make the first entry a nullptr for empty meshes 2026-06-22 10:24:38 -07:00
Veselin Dobrev fa9fdc6576 Address reviewer comments 2026-06-22 10:24:28 -07:00
Tzanio Kolev e25f778f7b Reorganized CHANGELOG 2026-06-22 08:56:58 -07:00
Tzanio Kolev b5baebd0b3 Merge branch 'master' into tmop-mesh-validity 2026-06-22 08:40:47 -07:00
Tzanio Kolev c9ecc5b466 Merge pull request #5205 from mfem/trace-p-ref
PRefinementTransferOperator for trace spaces
2026-06-22 08:40:35 -07:00
Tzanio Kolev f38de5059e Merge branch 'master' into tmop-mesh-validity 2026-06-22 08:39:13 -07:00
Tzanio Kolev 0ff0e1c76b Merge pull request #5223 from mfem/jdongg/pa-simplices
Partial assembly algorithms for Bernstein basis on simplicial meshes [pa-simplices-dev]
2026-06-22 08:37:01 -07:00
Eric B. Chin 983096e0a4 move operator into L2ProjectionH1Space 2026-06-19 23:03:35 -07:00
Eric B. Chin d4bd939578 add a preconditioner for parallel 2026-06-19 16:51:54 -07:00
Eric B. Chin 91aa10b7eb add comments, clean up variables 2026-06-19 15:24:55 -07:00
Eric B. Chin b527d78a2c remove unneded variables 2026-06-19 10:39:48 -07:00
Veselin Dobrev fcca0eaa71 Replace the 'mfem/github-actions' branch name 'update-num-tasks' with the
tag 'v2.7'.
2026-06-18 14:58:00 -07:00
Ketan Mittal 0bfa06f87d merge with master and resolve conflicts 2026-06-18 12:40:44 -07:00
Ketan Mittal 3aed584bf7 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-18 08:59:48 -07:00
Ketan Mittal c4d944cfad restore tolerance for barrier 2026-06-18 08:59:33 -07:00
Socratis Petrides c05114555f Merge branch 'trace-p-ref' of github.com:mfem/mfem into trace-p-ref 2026-06-17 18:41:27 -07:00
Socratis Petrides dfc15d77e5 Tzanio/codex review 2026-06-17 18:40:16 -07:00
Tzanio Kolev 25e5ac7db0 Merge branch 'master' into trace-p-ref 2026-06-17 09:14:57 -07:00
John Camier d0533903f3 Merge branch 'master' into bel-sorting 2026-06-17 07:25:22 -07:00
Socratis Petrides 48e258a60c add ownership documentation 2026-06-16 17:37:39 -07:00
Socratis Petrides e1b9c355e6 fix local prolonation issue in case of vdim > 1 2026-06-16 17:36:49 -07:00
Socratis Petrides c3db6ed873 fix headers 2026-06-16 17:35:50 -07:00
Socratis PetridesandTzanio Kolev 56b3e1ecc6 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:58:22 -07:00
Socratis PetridesandTzanio Kolev 5385e090f7 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:54 -07:00
Socratis PetridesandTzanio Kolev ecb7767b70 Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:21 -07:00
Socratis PetridesandTzanio Kolev 42bf39cd8c Update miniapps/dpg/makefile
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2026-06-16 11:57:01 -07:00
Tzanio Kolev 716f92d636 Merge pull request #5309 from mfem/fix-rocm7-hipsparse
Fix hipsparse in ROCm 7
2026-06-16 11:28:45 -07:00
Tzanio Kolev efdc17ce07 Merge pull request #5339 from mfem/tmop-AL-multiZ
TMOP - allow multiple fields in adaptive limiting.
2026-06-16 11:24:22 -07:00
John Camier ffab9d82fa Merge branch 'master' into bel-sorting 2026-06-16 07:23:55 -07:00
camierjs 891eaf2829 Fix missing B/Bt for OCCA backend 2026-06-16 07:07:34 -07:00
Ketan Mittal 053804f759 Merge branch 'master' into tmop-mesh-validity 2026-06-15 13:03:44 -07:00
Andrew Ho a8af93c4ec Merge branch 'master' into trace-p-ref 2026-06-15 12:19:38 -07:00
jdongg 00a1f5ed64 Add explicit return outside of if statement for mass and diffusion kernels 2026-06-12 08:05:30 -07:00
Veselin Dobrev a76b0dbc0e In GitHub CI, in the sanitizer builds of hypre and metis, do not set the
environment variables CXXFLAGS and LDFLAGS -- there are not needed and were
causing the hypre build to fail.
2026-06-12 02:43:18 -07:00
Veselin Dobrev 59ce107b33 In GitHub CI, fix caching key 2026-06-12 02:21:59 -07:00
Veselin Dobrev ce1036ccc8 Another try to define ACTIONS_VERSION in
.github/actions/sanitize/config/action.yml
in a way that it will be picked up by other sanitizer actions.
2026-06-12 01:06:06 -07:00
Veselin Dobrev b6861ed26f In GitHub CI, define ACTIONS_VERSION in
.github/actions/sanitize/config/action.yml
instead of
   .github/workflows/sanitizers.yml
where it was not picked up by some called actions.
2026-06-12 00:53:39 -07:00
Veselin Dobrev 8522cceb89 In CI, variables like ${{env.VAR}} cannot be used in "uses:" fields. 2026-06-11 23:46:43 -07:00
Veselin Dobrev dec9509a5d In GitHub CI, use and environment variable to set the version (branch or tag)
of the 'mfem/github-actions' to use.
2026-06-11 18:56:46 -07:00
Vladimir Z Tomov 5f02c6f64b fixed nvcc error for protected functions 2026-06-11 14:50:52 -07:00
Eric B. Chin 8271da1517 formatting 2026-06-10 23:15:29 -07:00
Eric B. Chin 05c0941cbb add a test 2026-06-10 22:54:19 -07:00
Eric B. Chin 8e565163ec add consistent LO mass matrix Mult() and MultTranspose() 2026-06-10 22:54:02 -07:00
jdongg 8a69b52f48 Add explicit return in diffusion and mass kernels in else branch. 2026-06-10 20:52:49 -07:00
Tzanio Kolev 51a0f8accb Merge branch 'master' into jdongg/pa-simplices 2026-06-10 15:54:26 -07:00
Veselin Dobrev 8777bc0810 In CI, set the number of build tasks for the 'build-mfem' action automatically 2026-06-09 18:20:58 -07:00
Ketan Mittal de97775bd0 Merge branch 'master' into tmop-mesh-validity 2026-06-09 16:11:59 -07:00
John Camier ad453255a5 Merge branch 'master' into bel-sorting 2026-06-09 06:46:23 -07:00
John Camier 2143ed5ca8 Merge branch 'master' into fix-rocm7-hipsparse 2026-06-09 06:46:06 -07:00
John Camier 2aa8372cdc Merge branch 'master' into jdongg/pa-simplices 2026-06-09 06:43:22 -07:00
Ketan Mittal 0176c7664c merge and resolve conflicts 2026-06-08 19:27:48 -07:00
Ketan Mittal e1c0705785 reviewer comments 2026-06-08 19:24:21 -07:00
Vladimir Z Tomov a4d114ad6e renamed a function, improved some comments 2026-06-08 10:04:43 -07:00
Ketan Mittal 1e5609f6c9 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-08 09:16:47 -07:00
Ketan Mittal c21d90589b Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-08 09:16:28 -07:00
Ketan Mittal bef7183ce7 documentation and CHANGELOG 2026-06-08 09:16:23 -07:00
Tzanio Kolev f6eda1fce6 Merge branch 'master' into tmop-mesh-validity 2026-06-05 16:40:10 -07:00
Tzanio Kolev a1bcd7443d Merge branch 'master' into jdongg/pa-simplices 2026-06-05 16:39:29 -07:00
Ketan Mittal 9c4d742db8 update gridfunction after h-adaptivity 2026-06-05 09:41:42 -07:00
Ketan Mittal 16ac6c40a3 refactor to have TMOPNewtonSolver manage bounding detJ 2026-06-05 09:33:21 -07:00
jdongg ee2484383c Remove own_rules from intrules.cpp 2026-06-04 15:43:36 -07:00
jdongg 1863dafeca Remove own rules flag and function. 2026-06-04 15:18:35 -07:00
Vladimir Z Tomov a4c7828ac2 pr review comments 2026-06-04 15:07:12 -07:00
Ketan Mittal 0a02f5737a address reviewer comments 2026-06-04 14:00:41 -07:00
Ketan Mittal 154a1840d0 Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-06-04 12:41:56 -07:00
Ketan Mittal 13e83cbd14 minor 2026-06-04 12:41:53 -07:00
Vladimir Z Tomov f64405ebb6 minor 2026-06-03 17:56:37 -07:00
Ketan Mittal 92f2057650 Merge branch 'master' into tmop-mesh-validity 2026-06-03 17:48:17 -07:00
Ketan Mittal f3558cb78c Merge branch 'master' into tmop-AL-multiZ 2026-06-03 10:15:35 -07:00
John Camier d5ddb83c9d Merge branch 'master' into jdongg/pa-simplices 2026-06-02 06:08:39 -07:00
camierjs 869b0e48a6 Add simplices tests with D1D > Q1D 2026-06-01 09:31:24 -07:00
John Camier 75180606f8 Merge branch 'master' into bel-sorting 2026-06-01 08:07:45 -07:00
John Camier 3cac4326e4 Merge branch 'master' into fix-rocm7-hipsparse 2026-06-01 08:07:06 -07:00
Ketan Mittal da5e8b5844 Merge branch 'master' into tmop-mesh-validity 2026-05-31 22:52:14 -07:00
Vladimir Z Tomov 081a90d1fa style 2026-05-31 14:45:56 -07:00
Vladimir Z Tomov d06a3eec7f use Vectors instead of allocating hypre vectors in AdvectorCG 2026-05-29 11:14:50 -07:00
jdongg 2766e56f2d Remove GaussJacobi out-of-range error check on alpha and beta before hard-coded cases. Add error check for unsupported geometries in StroudIntegrationRules class. 2026-05-28 12:58:02 -07:00
jdongg 0c781120a2 Remove references to InverseDuffyTrans in comments. Add description of on-the-fly inverse Duffy transform in GetRaggedTensorDofToQuad. Fix typos in GaussJacobi MFEM_ABORT. Remove unnecessary if statement in GaussJacobi routine when floating point type is undefined. 2026-05-28 11:51:50 -07:00
camierjs 449c9f5903 Avoid duplicate meshes in test_pa_simplices 2026-05-28 06:27:50 -07:00
Tzanio Kolev 6f793edfb7 Merge branch 'master' into tmop-AL-multiZ 2026-05-27 09:34:31 -07:00
John Camier 6455a0c1fc Merge branch 'master' into jdongg/pa-simplices 2026-05-27 06:31:47 -07:00
Vladimir Z Tomov b19c3c7e01 avoid Makeref 2026-05-26 22:23:20 -07:00
Vladimir Z Tomov fb7c8af59e more Newton iterations in the tmop unit test 2026-05-26 20:32:55 -07:00
Mittal, Ketan f6ea1ea0da minor 2026-05-26 10:01:57 -07:00
Vladimir TomovandCopilot Autofix powered by AI ef6510b42e typo
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-25 19:09:36 -07:00
Vladimir TomovandCopilot Autofix powered by AI 27ce32de07 typo
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-05-25 19:08:24 -07:00
Vladimir Z Tomov 341f6ec683 minor 2026-05-25 19:06:57 -07:00
Vladimir Z Tomov 8c88e1e26a style 2026-05-25 19:01:17 -07:00
Vladimir Z Tomov cabdb42bb4 Remap the adaptive limiting fields as a multi-component vector. 2026-05-25 18:52:04 -07:00
Vladimir Z Tomov e4a69c6245 vis in the mesh-optimizer, shorter 3d runs 2026-05-25 17:38:27 -07:00
John Camier e447f090cc Merge branch 'master' into jdongg/pa-simplices 2026-05-24 20:14:10 -07:00
Vladimir Z Tomov c3096d08bf Better example with 2 fields in the miniapps. 2026-05-24 18:07:53 -07:00
Vladimir Z Tomov 67b906ea3b Restored the optimization when the coefficients are ConstantCoefficients. 2026-05-24 17:46:37 -07:00
Vladimir Z Tomov 9167fc0c58 style 2026-05-22 18:03:28 -07:00
Vladimir Z Tomov e48ffe3579 cleanup 2026-05-22 18:00:32 -07:00
Vladimir Z Tomov a7697c1db1 cleanup 2026-05-22 17:34:05 -07:00
Vladimir Z Tomov 36c1cce132 delta_max per specified function. 2026-05-22 17:09:04 -07:00
Vladimir Z Tomov 38f257cf87 Multiple fields for adaptive limiting - initial pass. 2026-05-22 16:32:34 -07:00
camierjs a0022e0330 Fix number of threads in simplices kernels to allow D1D > Q1D, revert Stroud int rules in ex1[p], fix thread race in SmemPADiffusionApplyTetrahedron 2026-05-22 06:45:02 -07:00
camierjs 3a2f286e2d Select Stroud integration rule ine ex1[p] when needed, add verification in simplicial kernels that D1D <= Q1D 2026-05-21 08:09:33 -07:00
jdongg 7031f7cb80 Remove pullback flag from simplex unit tests 2026-05-19 22:24:45 -07:00
jdongg afaced7cd6 Rebase InverseDuffyTrans removal onto master. 2026-05-19 21:58:22 -07:00
jdongg d671ca4712 Remove InverseDuffyTrans and perform inverse mapping on the fly in GetRaggedTensorDofToQuad. Remove Pullback option from StroudIntRules object. Remove unused T array from RaggedDofToQuad object and mass+diffusion integrators. 2026-05-19 21:57:06 -07:00
John Camier 1095aee346 Merge branch 'master' into jdongg/pa-simplices 2026-05-19 20:28:53 -07:00
camierjs ba8764300c Revert the nvcc warning fix that will be addressed in another PR 2026-05-19 11:43:11 -07:00
camierjs 510477a204 Fix nvcc with gcc use of literal operator warnings 2026-05-19 09:23:17 -07:00
camierjs 0b9ae2202c Fix nvcc identifier warnings: preceded by whitespace in a literal operator declaration 2026-05-19 08:38:55 -07:00
camierjs 60641259b2 Revert ex1[p] refinements 2026-05-19 08:27:20 -07:00
camierjs 99e2b5023d Simplify ex1[p] with IsSimplexMesh 2026-05-19 08:26:01 -07:00
John Camier 4c59f6cfe6 Merge branch 'master' into jdongg/pa-simplices 2026-05-18 06:04:46 -07:00
Ketan Mittal cee913cc04 Merge branch 'master' into tmop-mesh-validity 2026-05-15 10:34:26 -07:00
Alex Lindsay bd5d928084 Pare down doxygen for GetFace group 2026-05-14 14:48:09 -07:00
John Camier 5e62d62c3d Merge branch 'master' into jdongg/pa-simplices 2026-05-14 13:21:04 -07:00
Ketan Mittal 5faba43544 Merge branch 'master' into bel-sorting 2026-05-14 10:34:41 -07:00
Alex Lindsay 20805d87b4 Prevent comment lines from going past column 80 2026-05-13 11:33:48 -07:00
Alex Lindsay 0119d25dfc Factor out common testing code 2026-05-13 11:32:12 -07:00
John Camier 22c15267f9 Merge branch 'master' into bel-sorting 2026-05-12 06:58:24 -07:00
John Camier 6c83fec2da Merge branch 'master' into fix-rocm7-hipsparse 2026-05-12 06:58:03 -07:00
John Camier bc845844ea Merge branch 'master' into jdongg/pa-simplices 2026-05-09 11:31:25 -07:00
camierjs 35ceed5193 Cleanup 2026-05-07 11:10:58 -07:00
camierjs ce52e1f51f Use MFEM_FOREACH_THREAD_DIRECT for simplices kernels 2026-05-07 11:01:25 -07:00
camierjs 16a21b366e Fix missing sync thread in smem mass simplices 2026-05-07 10:41:00 -07:00
John Camier d540fa5a12 Merge branch 'master' into bel-sorting 2026-05-07 08:06:44 -07:00
Tzanio Kolev 1221dea58e Merge branch 'master' into fix-rocm7-hipsparse 2026-05-07 07:29:53 -07:00
camierjs a5b9a7948f Add ex1p device simplices sample runs 2026-05-06 17:22:44 -07:00
camierjs 386e7e8c6a Add ex1 device simplices sample runs 2026-05-06 17:15:27 -07:00
camierjs 26077ab9aa Remove the ex_pa_simplices miniapps 2026-05-06 14:22:44 -07:00
camierjs 87bdb90bcc Merge branch 'master' into jdongg/pa-simplices 2026-05-05 11:20:27 -07:00
camierjs 848b54d13a Add pa-simplices to the selected files for formatting 2026-05-05 11:20:15 -07:00
camierjs 62214f61ae Merge branch 'master' into jdongg/pa-simplices 2026-05-05 06:29:17 -07:00
John Camier 9b93f1c1e1 Merge branch 'master' into bel-sorting 2026-05-05 06:19:44 -07:00
John Camier be03c9703f Merge branch 'master' into fix-rocm7-hipsparse 2026-05-05 06:14:52 -07:00
Tom Stitt b885fdc50f fix 2026-05-04 17:18:39 -07:00
Tom Stitt e4c0069ad9 better documentation 2026-05-04 14:41:49 -07:00
Tom StittandAndrew Ho 9fe53c2403 Apply suggestion from @helloworld922
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2026-05-04 14:14:01 -07:00
Ketan Mittal 00b9678b1d Merge branch 'master' into tmop-mesh-validity 2026-05-04 10:31:41 -07:00
camierjs 5e15a2e29b Cleanup PA simplices tests 2026-04-30 20:56:34 -07:00
camierjs d168f5b489 Revert intrules in PA simplices unit tests 2026-04-30 20:06:24 -07:00
camierjs 7a5d13e86a Fix & simplify ceed benchmark tests 2026-04-30 18:58:43 -07:00
camierjs 2e0d1697b9 Use specific simplices diffusion specializations when Q=P
Use integrations rules from integrators for unit testing
2026-04-30 18:25:13 -07:00
camierjs 87168f51ab Meld back toward master, removing un-warnings 2026-04-30 14:43:49 -07:00
jdongg 2558e32ee0 Switch std::beta to std::tgamma for compliances with C++11 standard 2026-04-30 13:28:32 -07:00
jdongg 7521efeff9 Add unit tests for Stroud quadrature rules as well as general Gauss-Jacobi rules 2026-04-30 13:11:57 -07:00
camierjs 3b23fd4941 Add extra meshs for PA simplices tests 2026-04-30 10:54:44 -07:00
camierjs d6a084322b Use new StroudIntRules for PA simplices tests 2026-04-30 09:34:01 -07:00
camierjs d2a38dd3d4 Merge branch 'jdongg/pa-simplices' of github.com:mfem/mfem into jdongg/pa-simplices 2026-04-30 09:27:34 -07:00
camierjs 17aab43082 Fix PA simplices tests with ir, fix PADiffusionApplyTriangle accumulate 2026-04-30 09:27:11 -07:00
jdongg e5c3cf4ee3 Move Duffy transform from IntegrationRule class to free-standing functions. Create StroudIntegrationRules class for caching Stroud rules. 2026-04-30 07:10:38 -07:00
camierjs b09cc88bab Just avoid the funct_coeff 2026-04-29 19:06:33 -07:00
camierjs 620904124e Init also y in test_pa_simplices 2026-04-29 18:50:03 -07:00
camierjs 6e55cfdc39 Shuffle back bilininteg mass pa simplices 2026-04-29 18:46:16 -07:00
camierjs bef23c5770 Split bilininteg mass pa simplices 2026-04-29 18:38:24 -07:00
camierjs a6fcf162a6 Split bilininteg diffusion pa simplices
and add initial unit tests
2026-04-29 18:20:12 -07:00
John Camier ba9af41877 Merge branch 'master' into jdongg/pa-simplices 2026-04-29 17:04:48 -07:00
John Camier 2762b9dbfc Merge branch 'master' into bel-sorting 2026-04-29 17:04:40 -07:00
John Camier d75a153df8 Merge branch 'master' into fix-rocm7-hipsparse 2026-04-29 17:04:16 -07:00
John Camier b170c6ae54 Merge branch 'master' into jdongg/pa-simplices 2026-04-27 08:12:29 -07:00
John Camier c4b4ad3224 Merge branch 'master' into bel-sorting 2026-04-27 08:11:59 -07:00
John Camier cecd75aff6 Merge branch 'master' into fix-rocm7-hipsparse 2026-04-27 08:11:53 -07:00
jdongg 50b197e754 Remove unused variables 2026-04-20 13:05:50 -07:00
jdongg 69dc2b5142 Make style 2026-04-20 12:56:55 -07:00
jdongg fc26f0a773 Switch temporary 1d GJ rules for Stroud construction from heap to stack memory. Clean up diffusion and mass kernels so they all use transposed basis arrays in quad-to-dof operation. Enforce uniform loop index notation across all kernels. 2026-04-20 12:53:00 -07:00
jdongg d160b88706 Guard against 0-size shared memory arrays in simplex diffusion kernels 2026-04-17 19:09:46 -07:00
jdongg 007d5e3e43 Re-add specializations for D1D=1 and D1D=9 since they no longer cause issues 2026-04-17 18:55:58 -07:00
jdongg fd78d48dfb Fix memory leaks (missing virtual destructor for DofToQuad after new derived class); fix race condition in diffusion integrator 2026-04-17 14:38:53 -07:00
camierjs c1d1df3d80 Merge branch 'master' into jdongg/pa-simplices 2026-04-17 07:04:10 -07:00
Tom Stitt c0d32918b0 add explicit call to hipsparseSpMV_preprocess to avoid runtime errors in rocsparse with ROCm 7
add checks to all cu/hipsparse calls, which i've confirmed would have cought this

add a static toggle for vendor libsparse usage
2026-04-16 12:36:23 -07:00
Socratis Petrides 03937d0dc2 Fix typo 2026-04-16 12:01:56 -07:00
jdongg 63e24c2a25 Fix documentation of RAGGED_TENSOR mode in DofToQuad. Move Bernstein-specific fields to derived class RaggedDofToQuad. Guard against overwriting Stroud rules in IntegrationRules::Set. 2026-04-15 20:05:06 -07:00
Socratis Petrides c202f9244a more fixes from review comments 2026-04-15 15:07:55 -07:00
Alex LindsayandClaude Sonnet 4.6 76cf9a2d00 Add Doxygen for Mesh::GetFaceElements()
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-14 15:50:11 -07:00
Alex LindsayandClaude Sonnet 4.6 4f796b3708 ReorderElements: sort boundary elements by face index after reordering
After calling ReorderElements the elements[] array is reordered but
boundary[] was left in its original arbitrary order.  be_to_face is
correctly rebuilt by vertex lookup, so individual face lookups remained
correct, but the boundary element ordering was inconsistent with the new
volume element ordering.

Sort boundary[] and be_to_face[] by face index (be_to_face[i]) after
rebuilding the face tables.  Face indices are assigned by
GetElementToFaceTable in the new element order, so sorting by face index
produces the same result as GenerateBoundaryElements would on a mesh
originally stored in the reordered element order.  Because each boundary
face has exactly one boundary element the sort is a strict total order
with no ties, making the output deterministic: two meshes with the same
geometry but different initial numberings produce identical mesh files
after Hilbert reordering.

For 3D meshes bel_to_edge (boundary-element-to-edge table) is also
permuted to stay consistent with the new boundary element ordering.

Add unit tests covering 3D hex, 2D quad, and 3D tet meshes that verify
the adjacent-element indices are non-decreasing across the sorted
boundary element list.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-14 15:32:32 -07:00
Socratis Petrides 2399e5e294 refactored trace tests to avoid code dublication 2026-04-06 18:53:36 -07:00
Socratis Petrides 205b694162 Merge branch 'master' into trace-p-ref 2026-04-06 18:30:43 -07:00
Socratis Petrides f2b661ca05 more fixes 2026-04-06 18:30:15 -07:00
Socratis Petrides c8ef971f03 review fixes 2026-04-06 18:11:06 -07:00
Socratis Petrides 10aa1133b2 review fixes 2026-04-06 13:22:59 -07:00
John Camier 05c3a2c83f Merge branch 'master' into jdongg/pa-simplices 2026-04-04 07:11:19 -07:00
jdongg d194a26542 Fix memory leak 2026-04-02 06:12:50 -07:00
Justin Dong a3a368dfd9 Make style 2026-04-02 05:26:08 -07:00
jdongg e8f39487b9 Fix doxygen formatting 2026-04-02 05:08:15 -07:00
jdongg ec6e2a7c74 Address PR reviews, pt. 1: removed unused lines, improve documentation, remove individual IntegrationRule objects for Stroud components, remove IsBernsteinSimplexSpace function from fespace and combine with UsesRaggedTensorBasis, ensuring checks for entire -mesh geoemtries are done, introduce MAX_Q1D_SIMPLEX and
MAX_D1D_SIMPLEX.
2026-04-02 04:51:31 -07:00
John Camier cb1f54ed2b Merge branch 'master' into jdongg/pa-simplices 2026-03-29 17:31:29 -07:00
John Camier 9a18da5aaa Merge branch 'master' into jdongg/pa-simplices 2026-03-24 07:58:41 -07:00
camierjs 4e31111827 Add BP'7' for simplices 2026-03-20 17:50:05 -07:00
camierjs 0cbf5b5562 CEED bench GLL simplices BP5 fix 2026-03-20 16:39:12 -07:00
camierjs d81bbc1214 BP5 hex/tet 2026-03-20 16:34:25 -07:00
camierjs 41255f308e Fix Benchmark internal namespace 2026-03-20 14:44:22 -07:00
camierjs 96be0cafdf Add benchmark tests with simplices 2026-03-20 08:30:23 -07:00
camierjs 55fc2f806c Merge branch 'master' into jdongg/pa-simplices 2026-03-20 05:59:14 -07:00
Ketan Mittal 9917fa8306 Merge branch 'master' into tmop-mesh-validity 2026-03-06 10:12:51 -08:00
camierjs eb95b46fad Missing Stroud flags 2026-03-05 09:42:14 -08:00
camierjs a0778c759e StroudFlag to bool 2026-03-05 09:04:05 -08:00
camierjs a922c1f2c3 Merge remote-tracking branch 'origin/catch-tests' into jdongg/pa-simplices 2026-03-05 08:31:35 -08:00
camierjs 17600f59f2 make style 2026-03-05 07:02:26 -08:00
camierjs 6cfec88f47 Merge branch 'master' into jdongg/pa-simplices 2026-03-05 07:01:30 -08:00
jdongg f11009d674 Switch 'or' to || for windows tests 2026-03-05 02:02:00 -08:00
jdongg 5df384f6ef Remove unused variables 2026-03-05 01:33:54 -08:00
jdongg daa555b68d Change pullback of Stroud rule to reference cube in AssemblePA routines to avoid assigning pointer to integration rule to a const integration rule 2026-03-05 00:59:36 -08:00
jdongg 76cfafe700 Switch 2D mass and diffusion kernels on triangles to stack memory 2026-03-04 18:51:54 -08:00
Socratis Petrides 58976c6f41 add relaxation in the smoother to fix non-PD issue of the preconditioner 2026-03-03 22:18:50 -08:00
camierjs 0114739e5e Fix more warnings 2026-02-21 18:14:23 -08:00
camierjs 708b642be2 Merge branch 'master' into jdongg/pa-simplices 2026-02-21 13:00:45 -08:00
camierjs d78357494d Fix some other warnings 2026-02-21 11:59:33 -08:00
camierjs 9398b1a6e0 Fix warnings 2026-02-21 11:52:23 -08:00
camierjs b23a087f42 Test for fec in IsBernsteinSimplexSpace 2026-02-21 11:27:58 -08:00
Ketan Mittal 72203ebb5c Merge branch 'master' into tmop-mesh-validity 2026-02-17 14:24:27 -08:00
Tzanio Kolev 86b2ad3a61 Merge branch 'master' into jdongg/pa-simplices 2026-02-17 08:29:10 -08:00
Socratis Petrides 35ba876ecf CI fix 2026-02-14 15:15:31 -08:00
Socratis Petrides 2886c4b211 pass ess_bdr_marker in pmg 2026-02-13 17:11:10 -08:00
Socratis Petrides 9455ed83b2 add ess_tdof treatement in pmg 2026-02-13 17:10:44 -08:00
Mittal, Ketan cc46bfdfcc Merge branch 'tmop-mesh-validity' of https://github.com/mfem/mfem into tmop-mesh-validity 2026-02-13 13:03:04 -08:00
Mittal, Ketan 010e96b6f0 remove shadow variable 2026-02-13 13:02:52 -08:00
Ketan Mittal edf551c40a Merge branch 'master' into tmop-mesh-validity 2026-02-13 09:47:32 -08:00
Mittal, Ketan 2bbb317831 minor 2026-02-10 16:35:22 -08:00
Mittal, Ketan 36835e62e0 remove unused variable 2026-02-10 16:25:01 -08:00
Mittal, Ketan 70867ab87b minor 2026-02-10 16:04:18 -08:00
Mittal, Ketan 4b59c90b08 merge and resolve conflicts 2026-02-10 15:56:43 -08:00
Ketan Mittal b538e9d34c Merge branch 'master' into tmop-mesh-validity 2026-02-10 15:48:42 -08:00
Socratis Petrides 505661eb5f minor 2026-02-09 19:25:36 -08:00
camierjs 5e9d32a0f7 Fix CMake PA simplices test, add visualization 2026-02-07 14:11:09 -08:00
camierjs 724866141a make style, remove unused-variable & shadows 2026-02-07 13:54:12 -08:00
John Camier f17c9eef12 Merge branch 'master' into jdongg/pa-simplices 2026-02-07 13:29:43 -08:00
Socratis Petrides 69c1d0d15b changelog 2026-02-07 12:52:47 -08:00
Socratis Petrides 5f7d138aef style 2026-02-06 18:18:19 -08:00
Tzanio Kolev 4659602e63 Merge branch 'master' into jdongg/pa-simplices 2026-02-06 07:11:45 -08:00
jdongg 1771a26426 Updated new function documentation and CHANGELOG 2026-02-05 15:04:39 -08:00
jdongg b42a383201 Modified Stroud rules to output nodes in the simplex, with InverseDuffyTrans routine added to pull back to reference cube when necessary 2026-02-05 13:23:18 -08:00
Socratis Petrides deae7e4997 minor 2026-02-05 10:33:10 -08:00
psocratis 6142015168 refactoring 2026-02-04 22:46:27 -08:00
psocratis 305a7f1e02 added pmg and cpmg options to dpg miniapps 2026-02-04 18:40:51 -08:00
psocratis 0bff07026c additional pmg and complex pmg options added including coarse solve option 2026-02-04 18:40:17 -08:00
psocratis 32c3e5dfd8 add complexoperator-->complexhypre 2026-02-04 18:39:05 -08:00
psocratis ee78679277 add blockoperator-->monolythic 2026-02-04 18:38:08 -08:00
jdongg 185f8b099d Small bug fixes to 2D mass and diffusion simplex kernels 2026-02-04 16:48:07 -08:00
psocratis 89a0f18b30 small leak 2026-02-04 11:14:18 -08:00
Socratis Petrides 4cbc345ba6 fix makefile and cmakelists.txt 2026-02-03 21:58:34 -08:00
Socratis Petrides abe3843712 typo 2026-02-03 21:47:04 -08:00
Socratis Petrides 3c2d1e4814 shadow var fix 2026-02-03 21:11:13 -08:00
Socratis Petrides a8c4ed3c79 expand p-MG to the complex case 2026-02-03 18:51:04 -08:00
Socratis Petrides 0aa73fa285 simplify precond construction in pmax 2026-02-03 15:12:01 -08:00
Socratis Petrides f1652b5ba0 add convenient functions for solver construction 2026-02-03 12:35:58 -08:00
Socratis Petrides fed0baf6b6 fix leak in complex case 2026-02-03 12:34:23 -08:00
Socratis Petrides 49a8ee5b54 fix possible leak in case of static_cond and add method to retrun trace_fes 2026-02-03 12:32:40 -08:00
Socratis Petrides c9569a0629 shadow var fix 2026-02-02 23:50:36 -08:00
Socratis Petrides a4e83b5836 p-mg cleanup 2026-02-02 22:30:02 -08:00
Socratis Petrides 8b285047bd some helper function for constructor order of fecols + Clone impl 2026-02-02 22:29:34 -08:00
psocratis f999be0372 valgrind fixes for p-mg 2026-02-01 18:12:32 -08:00
psocratis e639cfd75b style 2026-02-01 15:49:29 -08:00
psocratis 2753692303 Merge branch 'master' into trace-p-ref 2026-02-01 15:42:07 -08:00
psocratis 2722e979fc fix mg-transpose issue 2026-02-01 15:41:26 -08:00
psocratis 144c0ce106 CI 2026-02-01 14:49:58 -08:00
psocratis b64172e12e remove no longer needed code 2026-02-01 13:27:06 -08:00
psocratis c0072fcb84 fix windows CI issue 2026-02-01 13:02:18 -08:00
psocratis 37cd34a0e1 fix shadow var 2026-01-30 22:50:44 -08:00
psocratis 005be84af4 valgrind fixes 2026-01-30 22:44:48 -08:00
Socratis Petrides 0ef682c156 fixing dim for AMSvsADS in smoother 2026-01-30 17:03:26 -08:00
Socratis Petrides 3faa838774 expanding unit tests to serial/parallel true transfer 2026-01-30 17:02:53 -08:00
Socratis Petrides f10c2793c9 adding NC case in serial 2026-01-30 17:02:18 -08:00
Socratis Petrides 9f4b83362a p-ref MG for dpg diffusion 2026-01-28 00:02:13 -08:00
Socratis Petrides 8ef423ca65 traces p-ref example for p-diffusion 2026-01-28 00:01:00 -08:00
Socratis Petrides aff5173656 fix CI 2026-01-26 22:45:47 -08:00
Socratis Petrides 2164f07f01 starte p-mg for dpg-diffusion 2026-01-26 22:08:38 -08:00
Socratis Petrides fc0037da31 minor 2026-01-25 22:06:19 -08:00
Socratis Petrides 795a1a29a3 minor print fix 2026-01-25 00:11:44 -08:00
Socratis Petrides 476d9c5111 fix serial case 2026-01-25 00:06:18 -08:00
Socratis Petrides 64625f7333 tests for parallel trace/assemble prefoperator 2026-01-24 18:12:49 -08:00
Socratis Petrides 4ae8c1bc2b started on p-multigrid in serial 2026-01-24 18:12:11 -08:00
Socratis Petrides 4ae97f1834 hpp file changes for pref assembly 2026-01-24 18:11:37 -08:00
Socratis Petrides 816e8e4bea add option to get the preftransfer operator in parallel as a hypreparmatrix 2026-01-24 18:11:12 -08:00
Socratis Petrides c14d630f14 tet ND case fixed for assembled Transfer P 2026-01-24 00:03:09 -08:00
Socratis Petrides 899433f79f added PrerinementTransfer SparseMatrix. Still need to address the ND tet case wrt to doftrans 2026-01-23 18:54:42 -08:00
Ketan Mittal ff9294e5b0 Merge branch 'master' into tmop-mesh-validity 2026-01-23 15:41:31 -08:00
Socratis Petrides 16db883427 style 2026-01-22 22:11:46 -08:00
Socratis Petrides 1834268091 complete pref-tests 2026-01-22 22:11:25 -08:00
Socratis Petrides 947e25f769 clean up of ProjectTrace methods 2026-01-22 22:10:59 -08:00
Socratis Petrides 9fbe90527a fix trace pref prolongation 2026-01-22 22:09:35 -08:00
jdongg 076f1450e5 Cleanup junk files 2026-01-22 00:46:41 -08:00
Socratis Petrides bfa672c644 error computation 2026-01-22 00:24:33 -08:00
Justin Dong 30ac5d8d38 Fixed style issues 2026-01-21 20:35:36 -08:00
jdongg 5a2b05cc35 Remove unused variables, cleanup after rebasing. 2026-01-21 20:00:49 -08:00
Socratis Petrides afe388a573 style 2026-01-21 19:35:22 -08:00
Socratis Petrides 50a9cce9f4 first tests on project 2026-01-21 19:35:03 -08:00
Socratis Petrides 8ee2fdbfd9 Project coeff for trace/skeleton 2026-01-21 19:34:40 -08:00
Socratis Petrides d6b4594150 Pref mat-free for trace space 2026-01-21 19:34:06 -08:00
jdongg 771e263db8 Rebase onto main 2026-01-21 18:10:45 -08:00
jdongg ee13fec158 Added shared memory implementation of diffusion and mass integrators with partial assembly 2026-01-21 17:43:41 -08:00
Justin Dong 0162a6727d Optimized ragged tensor nested for loops by collapsing to 1d loops with appropriate forward and inverse maps to recover nested indices 2026-01-21 17:36:37 -08:00
Justin Dong 31fae9d195 Adding support for partial assembly of tetrahedrons for mass and diffusion integrators with Bernstein basis 2026-01-21 17:34:25 -08:00
Justin Dong 6ce5fe8fac Initial implementation of partial assembly on triangles for Bernstein basis
Rebase onto master
2026-01-21 17:33:07 -08:00
jdongg 1739cebfdb Added shared memory implementation of diffusion and mass integrators with partial assembly 2026-01-21 17:03:17 -08:00
Mittal, Ketan a7ca72c59f update get jacobian function 2025-12-22 12:24:15 -08:00
Mittal, Ketan 8267c7ea64 Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-22 12:18:20 -08:00
Mittal, Ketan d2cf8fef73 remove unneeded function 2025-12-22 12:18:17 -08:00
Mittal, Ketan 2df3fc8ddb remove unneeded function 2025-12-22 12:16:05 -08:00
Mittal, Ketan 59ee0d93c6 incorporate bounds from other branch 2025-12-14 16:57:34 -08:00
Mittal, Ketan d8164fd158 wip: check if function positive 2025-12-14 16:40:44 -08:00
Mittal, Ketan 23ce5f08e7 Merge branch 'plbound-extremum' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-13 17:37:42 -08:00
Mittal, Ketan e059253549 Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-13 17:35:13 -08:00
kmittal2 c89807da1e use in TMOP solver 2025-12-10 20:20:17 -08:00
kmittal2 53de3bde2c Merge branch 'master' of https://github.com/mfem/mfem into tmop-mesh-validity 2025-12-06 18:08:33 -08:00
kmittal2 56414350e2 initial commit - function to extract Jacobian determinant gridfunction in mesh/pmesh 2025-12-06 18:07:57 -08:00
Justin Dong 41d15a3a0a Optimized ragged tensor nested for loops by collapsing to 1d loops with appropriate forward and inverse maps to recover nested indices 2025-07-22 12:13:15 -07:00
Justin Dong daca70bf80 Adding support for partial assembly of tetrahedrons for mass and diffusion integrators with Bernstein basis 2025-07-02 11:28:39 -07:00
Justin Dong 0e02aa947a Initial implementation of partial assembly on triangles for Bernstein basis 2025-06-30 21:22:25 -07:00
92 changed files with 7516 additions and 979 deletions
@@ -94,6 +94,16 @@ inputs:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
# Unfortunately, "uses:" fields cannot have references to variables like
# ${{env.MFEM_ACTIONS_VERSION}}, so the branch/tag name has to be hard coded.
# Therefore, in the future, when updating the version of the
# mfem/github-actions to use, we'll have to replace:
# - all definitions of MFEM_ACTIONS_VERSION and
# - all "uses:" fields that refer to mfem/github-actions.
MFEM_ACTIONS_VERSION:
description: Version (branch or tag) of the mfem/github-actions to use.
default: v2.7
runs:
using: 'composite'
steps:
@@ -118,6 +128,7 @@ runs:
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
echo MFEM_ACTIONS_VERSION=${{inputs.MFEM_ACTIONS_VERSION}} >> $GITHUB_ENV
shell: bash
- name: Env (dir)
+1 -1
View File
@@ -53,7 +53,7 @@ runs:
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- uses: mfem/github-actions/build-mfem@v2.6
- uses: mfem/github-actions/build-mfem@v2.7
if: ${{steps.debug.outputs.cache-hit != 'true'}}
env:
CXXFLAGS: ${{env.CXXFLAGS}}
+6
View File
@@ -12,6 +12,11 @@
name: 'Install MPI'
description: 'Installs MPI and set up its environment variables'
inputs:
NO_FLAGS:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
runs:
using: 'composite'
steps:
@@ -27,6 +32,7 @@ runs:
shell: bash
- name: Env (bis)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
+2 -2
View File
@@ -37,14 +37,14 @@ runs:
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{env.MFEM_ACTIONS_VERSION}}
- uses: actions/cache/restore@v5 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
- name: Hypre/Metis links
if: ${{inputs.par == 'true'}}
+8 -21
View File
@@ -40,6 +40,7 @@ env:
METIS_ARCHIVE_MAC: metis-4.0.3-mac.tgz
METIS_TOP_DIR: metis-4.0.3
MFEM_TOP_DIR: mfem
MFEM_ACTIONS_VERSION: v2.7
# Note for future improvements:
#
@@ -170,20 +171,6 @@ jobs:
env
shell: bash
# For info on Xcode see:
# - https://github.com/actions/runner-images/issues/12541
# - https://github.com/actions/runner-images/blob/releases/macos-15-arm64/20250811/images/macos/macos-15-arm64-Readme.md#xcode
- name: Xcode version setup (MacOS)
if: matrix.os == 'macos-latest'
run: |
XCODE_PATH="/Applications/Xcode_16.4.app"
echo "> sudo xcode-select -s ${XCODE_PATH}"
sudo xcode-select -s ${XCODE_PATH}
echo "> g++ -v"
g++ -v
echo "> clang++ -v"
clang++ -v
# Only get MPI if defined for the job.
# TODO: It would be nice to have only one step, e.g. with a dedicated
# action, but I (@adrienbernede) don't see how at the moment.
@@ -228,11 +215,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-${{ env.MFEM_ACTIONS_VERSION }}
- name: get hypre
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.6
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -242,7 +229,7 @@ jobs:
- name: get hypre (Windows)
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.6
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -258,11 +245,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
- name: install metis
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.6
uses: mfem/github-actions/build-metis@v2.7
with:
archive: ${{ matrix.os != 'macos-latest' && env.METIS_ARCHIVE || env.METIS_ARCHIVE_MAC }}
dir: ${{ env.METIS_TOP_DIR }}
@@ -304,7 +291,7 @@ jobs:
# MFEM build and test
- name: build
uses: mfem/github-actions/build-mfem@v2.6
uses: mfem/github-actions/build-mfem@v2.7
env:
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
with:
@@ -375,7 +362,7 @@ jobs:
# Code coverage (process and upload reports)
- name: codecov
if: matrix.codecov == 'YES'
uses: mfem/github-actions/upload-coverage@v2.6
uses: mfem/github-actions/upload-coverage@v2.7
with:
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
project_dir: ${{ env.MFEM_TOP_DIR }}
+7 -5
View File
@@ -32,6 +32,7 @@ env:
METIS_ARCHIVE: metis-4.0.3.tar.gz
METIS_TOP_DIR: metis-4.0.3
COVERAGE_ENV: mfem-coverage
MFEM_ACTIONS_VERSION: v2.7
jobs:
gitignore:
@@ -53,33 +54,34 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
- name: Get Hypre
if: steps.hypre-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.6
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
target: int32
precision: fp64
- name: Cache Metis Install
id: metis-cache
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
- name: Install Metis
if: steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.6
uses: mfem/github-actions/build-metis@v2.7
with:
archive: ${{ env.METIS_ARCHIVE }}
dir: ${{ env.METIS_TOP_DIR }}
# MFEM build and test
- name: build-mfem
uses: mfem/github-actions/build-mfem@v2.6
uses: mfem/github-actions/build-mfem@v2.7
with:
os: ${{ runner.os }}
target: opt
+6 -2
View File
@@ -19,18 +19,22 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.6
uses: mfem/github-actions/build-hypre@v2.7
with:
archive: ${{env.HYPRE_TGZ}}
dir: ${{env.HYPRE_DIR}}
+6 -2
View File
@@ -19,18 +19,22 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.6
uses: mfem/github-actions/build-metis@v2.7
with:
archive: ${{env.METIS_TGZ}}
dir: ${{env.METIS_DIR}}
+51 -30
View File
@@ -11,35 +11,45 @@
Version 4.9.1 (development)
===========================
- Policy for AI-assisted contribution added to CONTRIBUTING.md
- Added policy for AI-assisted contribution to CONTRIBUTING.md.
Discretization improvements
---------------------------
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
integrators.
- Replaced legacy simplex quadrature rules with symmetric positive weight rules
for triangles (orders 0-25) and tetrahedra (orders 0-20). These rules
guarantee all-positive weights and interior quadrature points, improving
numerical stability. Higher orders fall back to Grundmann-Moller.
* Triangle rules: Witherden and Vincent, DOI: 10.1016/j.camwa.2015.03.017
* Tet rules (d=1-13): Witherden and Vincent (same as above)
* Tet rules (d=14-20): Chuluunbaatar et al., DOI: 10.1016/j.camwa.2022.08.016
- Added support for general 1D Gauss-Jacobi quadrature rules and Stroud conical
quadrature rules on triangles and tetrahedra.
- Improved the GridFunction projection routines. Projections work for Scalar,
Vector and VectorFE, also NURBS versions. Optionally different types of
projections can be selected, default behavior has not changed.
- Added GridFunction projection methods for trace spaces, i.e., project
coefficients on the mesh skeleton.
- Added methods to estimate function extremum using piecewise linear bounds plus
recursive subdivision.
- Extend FindPointsGSLIB to support surface meshes.
- Replaced legacy simplex quadrature rules with symmetric positive-weight
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
rules guarantee all-positive weights and interior quadrature points,
improving numerical stability. Higher orders fall back to Grundmann-Moller.
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
2015.
Tet rules (d=1-13): Witherden & Vincent (ibid).
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
2022.
- Improved the gridfunction projection routines. Projections work for Scalar,
Vector and VectorFE, also NURBS versions. Optionally different types of
projections can be selected, default behaviour has not changed.
- Added methods to estimate function extremum using piecewise linear bounds +
recursive subdivision.
Meshing improvements
--------------------
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
bounds on the determinant of the mesh transformation Jacobian.
- Added PA support for TMOP's adaptive limiting functionality. Multiple
GridFunctions and Coefficients can be combined to form a composite term.
- Improved support for 1D NURBS meshes with variable order, including using
the patches construct for 1D NURBS meshes.
@@ -48,22 +58,33 @@ Meshing improvements
parallel visualization, e.g. with GLVis. This is supported by both the Print
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
New and updated examples and miniapps
-------------------------------------
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
capability.
Linear and nonlinear solvers
----------------------------
- Added support for trace spaces in PRefinementTransferOperator. This is used in
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
DPG miniapps).
GPU computing
-------------
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
New and updated examples and miniapps
-------------------------------------
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
leverage the ParticleSet capability.
- Added (Complex)PRefinementMultigrid solver option in the DPG miniapps.
Miscellaneous
-------------
- Fixed signed DOF handling in parallel grid-function reading (read constructor)
and saving via ParGridFunction::SaveAsOne(). Simplified the process of
applying the DOF signs by using the new method ApplyDofSigns() in class
ParFiniteElementSpace -- the method will return immediately if no sign flips
are needed.
- Fixed signed DOF handling in ParGridFunction reading (read constructor) and
saving via SaveAsOne(). Simplified the process of applying the DOF signs by
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
method will return immediately if no sign flips are needed.
Version 4.9, released on Dec 11, 2025
+13 -12
View File
@@ -50,6 +50,10 @@
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cpu
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
//
// Device simplices sample runs:
// ex1 -pa -d gpu -m ../data/inline-tet.mesh
// ex1 -pa -d gpu -m ../data/inline-tri.mesh
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
@@ -138,25 +142,25 @@ int main(int argc, char *argv[])
}
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
// Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
FiniteElementCollection *fec;
bool delete_fec;
auto basis_type = (pa && mesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
fec = new H1_FECollection(order, dim, basis_type);
}
else if (mesh.GetNodes())
{
fec = mesh.GetNodes()->OwnFEC();
delete_fec = false;
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
fec = new H1_FECollection(order = 1, dim, basis_type);
}
FiniteElementSpace fespace(&mesh, fec);
cout << "Number of finite element unknowns: "
@@ -292,10 +296,7 @@ int main(int argc, char *argv[])
}
// 15. Free the used memory.
if (delete_fec)
{
delete fec;
}
if (order > 0) { delete fec; }
return 0;
}
+14 -13
View File
@@ -42,7 +42,11 @@
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/square-mixed.mesh
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/fichera-mixed.mesh
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
// mpirun -np 4 ex1p -pa -d ceed-cpu -m ../data/beam-tet.mesh
//
// Device simplices sample runs:
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tet.mesh
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tri.mesh
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
@@ -165,19 +169,20 @@ int main(int argc, char *argv[])
}
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
// use continuous Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
FiniteElementCollection *fec;
bool delete_fec;
auto basis_type = (pa && pmesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
fec = new H1_FECollection(order, dim, basis_type);
}
else if (pmesh.GetNodes())
{
fec = pmesh.GetNodes()->OwnFEC();
delete_fec = false;
if (myid == 0)
{
cout << "Using isoparametric FEs: " << fec->Name() << endl;
@@ -185,8 +190,7 @@ int main(int argc, char *argv[])
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
fec = new H1_FECollection(order = 1, dim, basis_type);
}
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_BigInt size = fespace.GlobalTrueVSize();
@@ -333,10 +337,7 @@ int main(int argc, char *argv[])
}
// 17. Free the used memory.
if (delete_fec)
{
delete fec;
}
if (order > 0) { delete fec; }
return 0;
}
+2
View File
@@ -195,12 +195,14 @@ set(HDRS
integ/bilininteg_dgtrace_kernels.hpp
integ/bilininteg_vecdiffusion_kernels.hpp
integ/bilininteg_convection_kernels.hpp
integ/bilininteg_diffusion_pa_simplices.hpp
integ/bilininteg_diffusion_kernels.hpp
integ/bilininteg_elasticity_kernels.hpp
integ/bilininteg_hcurl_kernels.hpp
integ/bilininteg_hdiv_kernels.hpp
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
integ/bilininteg_mass_pa_simplices.hpp
integ/bilininteg_vecdiffusion_pa.hpp
integ/bilininteg_vecmass_pa.hpp
coefficient.hpp
+22 -4
View File
@@ -1345,7 +1345,8 @@ real_t DiffusionIntegrator::ComputeFluxEnergy
}
const IntegrationRule &DiffusionIntegrator::GetRule(
const FiniteElement &trial_fe, const FiniteElement &test_fe)
const FiniteElement &trial_fe, const FiniteElement &test_fe,
const bool stroud)
{
int order;
if (trial_fe.Space() == FunctionSpace::Pk)
@@ -1362,7 +1363,15 @@ const IntegrationRule &DiffusionIntegrator::GetRule(
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
}
MassIntegrator::MassIntegrator(const IntegrationRule *ir)
@@ -1449,7 +1458,8 @@ void MassIntegrator::AssembleElementMatrix2(
const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans)
const ElementTransformation &Trans,
const bool stroud)
{
// int order = trial_fe.GetOrder() + test_fe.GetOrder();
const int order = trial_fe.GetOrder() + test_fe.GetOrder() + Trans.OrderW();
@@ -1458,7 +1468,15 @@ const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
}
+40 -2
View File
@@ -2184,11 +2184,22 @@ public:
const Vector&, const Vector&,
Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const bool, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&,
Vector&, const int, const int);
using DiagonalKernelType = void(*)(const int, const bool, const Array<real_t>&,
const Array<real_t>&, const Vector&, Vector&,
const int, const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
struct Kernels { Kernels(); };
@@ -2341,7 +2352,8 @@ public:
void AddMultPatchPA(const int patch, const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
const FiniteElement &test_fe,
const bool stroud = false);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2352,6 +2364,13 @@ public:
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
const IntegrationRule* GetDefaultIntegrationRule(
@@ -2388,11 +2407,22 @@ public:
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&, Vector&,
const int, const int);
using DiagonalKernelType = void(*)(const int, const Array<real_t>&,
const Vector&, Vector&, const int,
const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
struct Kernels { Kernels(); };
@@ -2441,7 +2471,8 @@ public:
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans);
const ElementTransformation &Trans,
const bool stroud = false);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2452,6 +2483,13 @@ public:
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
+42 -1
View File
@@ -167,7 +167,15 @@ public:
/** @brief Full multidimensional representation which does not use tensor
product structure. The ordering of the degrees of freedom is the
same as TENSOR, but the sizes of B and G are the same as FULL.*/
LEXICOGRAPHIC_FULL
LEXICOGRAPHIC_FULL,
/** @brief Ragged tensor product representation using 1D matrices/tensors
with dimensions using 1D number of quadrature points and ragged tensor degrees of
freedom. */
/** Used only for partial assembly of the H1 positive basis. The
size of B is d1d x qnpt x dim. Since different Gauss-Jacobi quadrature rules
are employed in each dimension, we need to store dim arrays. */
RAGGED_TENSOR
};
/// Describes the contents of the #B, #Bt, #G, and #Gt arrays, see #Mode.
@@ -228,6 +236,39 @@ public:
const Array<DofToQuad*> &dof2quad_array,
const IntegrationRule &ir,
DofToQuad::Mode mode);
virtual ~DofToQuad() = default;
};
/** @brief Structure representing the matrices/tensors needed to evaluate (in
reference space) the values, gradients, divergences, or curls of a positive
FiniteElement on simplices at the quadrature points of Stroud conical quadrature. */
class RaggedDofToQuad : public DofToQuad
{
public:
/** @brief Special basis function structures for positive (Bernstein) basis with
partial assembly. The storage layout of Ba1 is ndof x nqpt for scalar elements.
The storage layout of Ba2 is ndof x ndof x nqpt. In particular, we have
Ba2(iqpt, a1, a2) = B^{p-a1}_{a2}(x_{iqpt}). */
Array<real_t> Ba1, Ba2, Ba3;
Array<real_t> Ba1t, Ba2t, Ba3t;
/** @brief Special structures for gradients of positive basis with partial assembly.
The gradient arrays exploit properties of the Bernstein basis which allow grad(B^p_alpha)
to be expressed as the sum of products of B^{p-1}_alpha and the barycentric coordinates.
Thus, Ga1 and Ga2 simply contain the ragged tensor product components of B^{p-1}_alpha */
Array<real_t> Ga1, Ga2, Ga3;
Array<real_t> Ga1t, Ga2t, Ga3t;
/** @brief Mapping from the Bernstein multi-index (a_1, ..., a_d) to the lexicographic
dof index. */
Array<int> lex_map;
Array<int> forward_map2d_diff, forward_map3d_diff;
Array<int> inverse_map2d_diff, inverse_map3d_diff;
Array<int> forward_map2d_mass, forward_map3d_mass;
Array<int> inverse_map2d_mass, inverse_map3d_mass;
};
/// Describes the function space on each element
+302
View File
@@ -557,6 +557,101 @@ H1Pos_TriangleElement::H1Pos_TriangleElement(const int p)
}
}
const DofToQuad &H1Pos_TriangleElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt*ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize((int)nqpt*ndof*ndof);
rd2q->Ba1t.SetSize(nqpt*ndof);
rd2q->Ba2t.SetSize((int)nqpt*ndof*ndof);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt*(ndof -1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->Ga1t.SetSize(nqpt*(ndof -1));
rd2q->Ga2t.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->lex_map.SetSize(ndof * ndof);
Vector shape_a1(ndof), shape_a2(ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2((ndof-1) * (ndof-1));
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). Additionally, the Bernstein PA algorithms expect evaluation of the
// component 1D bases at the Stroud nodes pulled back to the unit square, so perform the pullback
// on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
rd2q->Ba2t[i + nqpt*(j + ndof*k)] = rd2q->Ba2[k + ndof*(j + ndof*i)] = shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
rd2q->Ga2t[i + nqpt*(j + (ndof-1)*k)] = rd2q->Ga2[k + (ndof-1)*(j +
(ndof-1)*i)] = shape_Ga2(k);
}
}
}
}
// stores the mapping from 2D Bernstein multi-index (i,j,p-i-j) to the
// lexicographic DOF ordering
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
int idx = ((2 * (ndof-1) + 3) - j) * j / 2 + i;
rd2q->lex_map[j + ndof*i] = idx;
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TriangleElement::CalcShape(
const int p, const real_t l1, const real_t l2, real_t *shape)
@@ -749,6 +844,213 @@ H1Pos_TetrahedronElement::H1Pos_TetrahedronElement(const int p)
}
}
const DofToQuad &H1Pos_TetrahedronElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
const int basis_dim2d = ndof*(ndof+1) / 2;
const int basis_dim3d = ndof*(ndof+1)*(ndof+2) / 6;
const int basis_dim2d_diff = (ndof-1)*(ndof) / 2;
const int basis_dim3d_diff = (ndof-1)*(ndof)*(ndof+1) / 6;
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt * ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize(nqpt * basis_dim2d);
// third component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba3.SetSize(nqpt * basis_dim3d);
rd2q->Ba1t.SetSize(nqpt * ndof);
rd2q->Ba2t.SetSize(nqpt * basis_dim2d);
rd2q->Ba3t.SetSize(nqpt * basis_dim3d);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt * (ndof-1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt * basis_dim2d_diff);
// stores third component of ragged tensor basis with order p-1
rd2q->Ga3.SetSize(nqpt * basis_dim3d_diff);
rd2q->Ga1t.SetSize(nqpt * (ndof-1));
rd2q->Ga2t.SetSize(nqpt * basis_dim2d_diff);
rd2q->Ga3t.SetSize(nqpt * basis_dim3d_diff);
rd2q->lex_map.SetSize(ndof * ndof * ndof);
rd2q->forward_map2d_diff.SetSize((ndof-1) * (ndof-1));
rd2q->forward_map3d_diff.SetSize((ndof-1) * (ndof-1) * (ndof-1));
rd2q->inverse_map2d_diff.SetSize(2 * basis_dim2d_diff);
rd2q->inverse_map3d_diff.SetSize(3 * basis_dim3d_diff);
rd2q->forward_map2d_mass.SetSize(ndof * ndof);
rd2q->forward_map3d_mass.SetSize(ndof * ndof * ndof);
rd2q->inverse_map2d_mass.SetSize(2 * basis_dim2d);
rd2q->inverse_map3d_mass.SetSize(2 * basis_dim3d);
// forward and inverse maps for multi-index to collpased 1d index for diffusion, can combine
// these four loops, but need four idx's and clause for shorter diff loops
int idx = 0;
for (int i = 0; i < ndof-1; i++)
{
for (int j = 0; j < ndof-i-1; j++)
{
rd2q->forward_map2d_diff[j + (ndof-1)*i] = idx;
rd2q->inverse_map2d_diff[2*idx] = i;
rd2q->inverse_map2d_diff[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof-1; k++)
{
for (int j = 0; j < ndof-k-1; j++)
{
for (int i = 0; i < ndof-k-j-1; i++)
{
rd2q->forward_map3d_diff[k + (ndof-1)*(j + (ndof-1)*i)] = idx;
rd2q->inverse_map3d_diff[3*idx] = i;
rd2q->inverse_map3d_diff[1 + 3*idx] = j;
rd2q->inverse_map3d_diff[2 + 3*idx] = k;
idx++;
}
}
}
// forward and inverse maps for multi-index to collpased 1d index for mass
idx = 0;
for (int j = 0; j < ndof; j++)
{
for (int i = 0; i < ndof-j; i++)
{
rd2q->forward_map2d_mass[j + ndof*i] = idx;
rd2q->inverse_map2d_mass[2*idx] = i;
rd2q->inverse_map2d_mass[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof; k++)
{
for (int j = 0; j < ndof-k; j++)
{
for (int i = 0; i < ndof-k-j; i++)
{
rd2q->forward_map3d_mass[k + ndof*(j + ndof*i)] = idx;
rd2q->inverse_map3d_mass[2*idx] = i;
rd2q->inverse_map3d_mass[1 + 2*idx] = j;
// d2q->inverse_map3d_mass[2 + 3*idx] = k;
idx++;
}
}
}
Vector shape_a1(ndof), shape_a2(ndof * ndof), shape_a3(ndof * ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2(ndof-1), shape_Ga3(ndof-1);
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). The first 'nqpt' points in the third dimension have the same z-coordinates
// as those of the 1D rule for the third dimension (i.e. Gauss-Legendre rule). Additionally,
// the Bernstein PA algorithms expect evaluation of the component 1D bases at the Stroud nodes
// pulled back to the unit cube, so perform the pullback on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
const real_t z = ir.IntPoint(nqpt*nqpt*i).z / (1.0 - ir.IntPoint(
nqpt*nqpt*i).x - ir.IntPoint(nqpt*nqpt*i).y);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
const int a_2d_mass = rd2q->forward_map2d_mass[k + ndof*j];
rd2q->Ba2t[i + nqpt*a_2d_mass] = rd2q->Ba2[a_2d_mass + basis_dim2d*i] =
shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
const int a_2d_diff = rd2q->forward_map2d_diff[k + (ndof-1)*j];
rd2q->Ga2t[i + nqpt*a_2d_diff] = rd2q->Ga2[a_2d_diff + basis_dim2d_diff*i] =
shape_Ga2(k);
Poly_1D::CalcBernstein(ndof-2-j-k, z, shape_Ga3);
}
Poly_1D::CalcBernstein(ndof-1-j-k, z, shape_a3);
for (int m = 0; m < ndof-j-k; m++)
{
const int a_3d_mass = rd2q->forward_map3d_mass[m + ndof*(k + ndof*j)];
rd2q->Ba3t[i + nqpt*a_3d_mass] = rd2q->Ba3[a_3d_mass + basis_dim3d*i] =
shape_a3(
m);
if (j < ndof-1 && k < ndof-j-1 && m < ndof-j-k-1)
{
// // collapsed 1D access
// d2q->Ga3[i + nqpt*(m + d2q->offset3d[k + (ndof-1)*j])] = shape_Ga3(m);
// collapsed 1D access with forward mapping
const int a_3d_diff = rd2q->forward_map3d_diff[m + (ndof-1)*(k + (ndof-1)*j)];
rd2q->Ga3t[i + nqpt*a_3d_diff] = rd2q->Ga3[a_3d_diff + basis_dim3d_diff*i] =
shape_Ga3(m);
}
}
}
}
}
// stores the mapping from 3D Bernstein multi-index (i,j,k,p-i-j-k) to the
// lexicographic DOF ordering
int p = ndof - 1;
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
for (int k = 0; k < ndof-i-j; k++)
{
int dof = (p+1)*(p+2)*(p+3) / 6;
int tet = (p-k)*(p-k+1)*(p-k+2) / 6;
int tri = (p+1-k-j)*(p+2-k-j)/2;
int multi_idx = dof - tet - tri + i;
rd2q->lex_map[k + ndof*(j + ndof*i)] = multi_idx;
}
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TetrahedronElement::CalcShape(
const int p, const real_t l1, const real_t l2, const real_t l3,
+30
View File
@@ -191,6 +191,21 @@ public:
/// Construct the H1Pos_TriangleElement of order @a p
H1Pos_TriangleElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)/2 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
real_t *shape);
@@ -220,6 +235,21 @@ public:
/// Construct the H1Pos_TetrahedronElement of order @a p
H1Pos_TetrahedronElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)(p+3)/6 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
const real_t z, real_t *shape);
+27
View File
@@ -250,6 +250,14 @@ public:
its GetOrder() method. */
virtual FiniteElementCollection *Clone(int p) const;
/** @brief Return the order parameter used to construct this collection.
* This differs from GetOrder() depending on the collection type. */
virtual int GetConstructorOrder() const
{
MFEM_ABORT("Collection " << Name() << " does not support GetConstructorOrder");
return -1;
}
protected:
const int base_p; ///< Order as returned by GetOrder().
@@ -314,6 +322,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new H1_FECollection(p, dim, b_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~H1_FECollection();
};
@@ -343,6 +354,10 @@ class H1_Trace_FECollection : public H1_FECollection
public:
H1_Trace_FECollection(const int p, const int dim,
const int btype = BasisType::GaussLobatto);
FiniteElementCollection *Clone(int p) const override
{ return new H1_Trace_FECollection(p, dim+1, b_type); }
};
/// Arbitrary order "L2-conforming" discontinuous finite elements.
@@ -396,6 +411,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new L2_FECollection(p, dim, b_type, m_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~L2_FECollection();
};
@@ -456,6 +474,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new RT_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return base_p-1; }
virtual ~RT_FECollection();
};
@@ -536,6 +557,9 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new ND_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return dim>1 ? base_p : base_p+1; }
virtual ~ND_FECollection();
};
@@ -548,6 +572,9 @@ public:
ND_Trace_FECollection(const int p, const int dim,
const int cb_type = BasisType::GaussLobatto,
const int ob_type = BasisType::GaussLegendre);
FiniteElementCollection *Clone(int p) const override
{ return new ND_Trace_FECollection(p, dim+1, cb_type, ob_type); }
};
/// Arbitrary order 3D H(curl)-conforming Nedelec finite elements in 1D.
+1 -2
View File
@@ -4631,9 +4631,8 @@ FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
ElementDofOrdering GetEVectorOrdering(const FiniteElementSpace& fes)
{
return UsesTensorBasis(fes)?
return (UsesTensorBasis(fes) || fes.UsesRaggedTensorBasis()) ?
ElementDofOrdering::LEXICOGRAPHIC:
ElementDofOrdering::NATIVE;
}
} // namespace mfem
+12
View File
@@ -1514,6 +1514,18 @@ public:
return dynamic_cast<const L2_FECollection*>(fec) != NULL;
}
/// @brief Return true if the mesh contains only one topology, the elements are
/// all triangles or tetrahedrons, and the elements are ragged tensor elements
/// i.e. Bernstein/positive basis.
bool UsesRaggedTensorBasis() const
{
bool simplex = this->GetMesh()->IsSimplexMesh();
bool positive =
dynamic_cast<const mfem::H1Pos_TriangleElement *>(this->GetTypicalFE()) ||
dynamic_cast<const mfem::H1Pos_TetrahedronElement *>(this->GetTypicalFE());
return simplex && positive;
}
/** In variable-order spaces on nonconforming (NC) meshes, this function
controls whether strict conformity is enforced in cases where coarse
edges/faces have higher polynomial order than their fine NC neighbors.
+167
View File
@@ -2256,6 +2256,104 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
}
}
void GridFunction::AccumulateAndCountTraceValues(
Coefficient *coeff[], VectorCoefficient *vcoeff,
Array<int> &values_counter)
{
if (vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
"vcoeff vdim != fes VDim");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::VALUE &&
fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR,
"Can only call ProjectTraceCoefficient on scalar value-type "
"trace elements. "
"Use ProjectTraceCoefficientNormal for RT and "
"ProjectTraceCoefficientTangent for ND finite elements.");
}
Array<int> vdofs;
Vector vc;
values_counter.SetSize(Size());
values_counter = 0;
const int vdim = fes->GetVDim();
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
const FiniteElement *fe = fes->GetFaceElement(i);
const int fdof = fe->GetDof();
ElementTransformation *transf = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
fes->GetFaceVDofs(i, vdofs);
for (int j = 0; j < fdof; j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
transf->SetIntPoint(&ip);
if (vcoeff) { vcoeff->Eval(vc, *transf, ip); }
for (int d = 0; d < vdim; d++)
{
if (!vcoeff && !coeff[d]) { continue; }
real_t val = vcoeff ? vc(d) : coeff[d]->Eval(*transf, ip);
int ind = vdofs[fdof*d+j];
if ( ind < 0 )
{
val = -val, ind = -1-ind;
}
if (++values_counter[ind] == 1)
{
(*this)(ind) = val;
}
else
{
(*this)(ind) += val;
}
}
}
}
}
void GridFunction::AccumulateAndCountTraceTangentValues(
VectorCoefficient &vcoeff, Array<int> &values_counter)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()
->GetRangeType() == FiniteElement::VECTOR &&
fes->GetTypicalTraceElement()
->GetMapType() == FiniteElement::H_CURL,
"Not an ND FE space!");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetPhysRangeDim(
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
"vcoeff vdim != PhysRangeDim");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
Vector lvec;
values_counter.SetSize(Size());
values_counter = 0;
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
fes->GetFaceVDofs(i, dofs);
lvec.SetSize(fe->GetDof());
fe->Project(vcoeff, *T, lvec);
accumulate_dofs(dofs, lvec, *this, values_counter);
}
}
void GridFunction::ComputeMeans(AvgType type, Array<int> &zones_per_vdof)
{
switch (type)
@@ -2698,6 +2796,74 @@ void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
}
}
void GridFunction::ProjectTraceCoefficient(Coefficient *coeff[])
{
Array<int> values_counter;
AccumulateAndCountTraceValues(coeff, NULL, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficient(Coefficient &coeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == 1, "ProjectTraceCoefficient(Coefficient&)"
"is only valid for scalar GridFunction");
Coefficient *coeff_p = &coeff;
ProjectTraceCoefficient(&coeff_p);
}
void GridFunction::ProjectTraceCoefficient(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == vcoeff.GetVDim(),
"Incompatible vcoeff vdim and fes vdim");
Array<int> values_counter;
AccumulateAndCountTraceValues(NULL, &vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR &&
fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::INTEGRAL, "Not an RT FE space!");
MFEM_VERIFY(vcoeff.GetVDim() == fes->GetMesh()->SpaceDimension(),
"vcoeff vdim (" << vcoeff.GetVDim()
<< ") != SpaceDimension ("
<< fes->GetMesh()->SpaceDimension() << ")");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
int dim = vcoeff.GetVDim();
Vector vc(dim), nor(dim), lvec;
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
lvec.SetSize(fe->GetDof());
for (int j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
vcoeff.Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
lvec(j) = (vc * nor);
}
fes->GetFaceVDofs(i, dofs);
SetSubVector(dofs, lvec);
}
}
void GridFunction::ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff)
{
Array<int> values_counter;
AccumulateAndCountTraceTangentValues(vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectCoefficientGlobalL2(VectorCoefficient &vcoeff,
real_t rtol, int iter)
{
@@ -5286,6 +5452,7 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
Vector lel, uel;
GetElementBounds(plb, lel, uel, vdim);
+24
View File
@@ -578,6 +578,13 @@ protected:
const Array<int> &bdr_attr,
Array<int> &values_counter);
void AccumulateAndCountTraceValues(Coefficient *coeff[],
VectorCoefficient *vcoeff,
Array<int> &values_counter);
void AccumulateAndCountTraceTangentValues(VectorCoefficient &vcoeff,
Array<int> &values_counter);
// Complete the computation of averages; called e.g. after
// AccumulateAndCountZones().
void ComputeMeans(AvgType type, Array<int> &zones_per_vdof);
@@ -663,6 +670,23 @@ public:
ProjectBdrCoefficient(&coeff_p, attr);
}
/// Project a Coefficient on a GridFunction defined on H1 trace space
void ProjectTraceCoefficient(Coefficient *coeff[]);
void ProjectTraceCoefficient(Coefficient &coeff);
/** @brief Project a VectorCoefficient @a vcoeff on a GridFunction
defined on a Vector H1 trace space. Note that this also works
for a scalar H1 trace space, where only the first component of
@a vcoeff is used. */
void ProjectTraceCoefficient(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an RT trace space */
void ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an ND trace space */
void ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on the GridFunction, modifying only
DOFs on the boundary associated with the boundary attributes marked in
the @a attr array. */
@@ -10,6 +10,7 @@
// CONTRIBUTING.md for details.
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -19,6 +20,13 @@ namespace mfem
DiffusionIntegrator::Kernels::Kernels()
{
// 2D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<2,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<2,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<2,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<2,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,7,6>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<2,1,1>();
DiffusionIntegrator::AddSpecialization<2,2,2>();
@@ -40,7 +48,18 @@ DiffusionIntegrator::Kernels::Kernels()
DiffusionIntegrator::AddSpecialization<2,8,9>();
DiffusionIntegrator::AddSpecialization<2,9,10>();
// others
DiffusionIntegrator::AddSimplexSpecialization<2,2,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,6>();
// 3D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<3,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<3,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<3,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<3,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<3,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<3,7,6>();
DiffusionIntegrator::AddSimplexSpecialization<3,8,7>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<3,1,1>();
DiffusionIntegrator::AddSpecialization<3,2,2>();
+19 -16
View File
@@ -12,7 +12,6 @@
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#include "../kernel_dispatch.hpp"
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
@@ -637,8 +636,8 @@ inline void SmemPADiffusionApply2D(const int NE,
const bool symmetric,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Array<real_t> &bt_,
const Array<real_t> &gt_,
const Array<real_t> &,
const Array<real_t> &,
const Vector &d_,
const Vector &x_,
Vector &y_,
@@ -1218,43 +1217,47 @@ inline void SmemPADiffusionApply3D(const int NE,
namespace
{
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
using ApplySimplexKernelType = DiffusionIntegrator::ApplySimplexKernelType;
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
}
inline
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
{
if (DIM == 2) { return internal::PADiffusionApply2D; }
else if (DIM == 3) { return internal::PADiffusionApply3D; }
if (dim == 2) { return internal::PADiffusionApply2D; }
else if (dim == 3) { return internal::PADiffusionApply3D; }
else { MFEM_ABORT(""); }
}
template<int DIM, int D1D, int Q1D>
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
MFEM_ABORT("");
else { MFEM_ABORT(""); }
return nullptr;
}
inline DiagonalKernelType
DiffusionIntegrator::DiagonalPAKernels::Fallback(int DIM, int, int)
DiffusionIntegrator::DiagonalPAKernels::Fallback(int dim, int, int)
{
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
if (dim == 2) { return internal::PADiffusionDiagonal2D; }
else if (dim == 3) { return internal::PADiffusionDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+31 -2
View File
@@ -15,6 +15,7 @@
#include "../../mesh/nurbs.hpp"
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp"
namespace mfem
{
@@ -68,6 +69,24 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
#endif // MFEM_USE_OCCA
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
return ApplySimplexPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
rmaps->lex_map,
rmaps->forward_map2d_diff,
rmaps->inverse_map2d_diff,
rmaps->forward_map3d_diff,
rmaps->inverse_map3d_diff,
rmaps->Ga1,
rmaps->Ga2,
rmaps->Ga3,
rmaps->Ga1t,
rmaps->Ga2t,
rmaps->Ga3t,
Dv, x, y, dofs1D, quad1D);
}
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
Gt, Dv, x, y, dofs1D, quad1D);
}
@@ -94,7 +113,8 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
fespace = &fes;
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, stroud);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -119,13 +139,22 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
dim = mesh->Dimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
const int sdim = mesh->SpaceDimension();
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
if (MQ) { coeff.ProjectTranspose(*MQ); }
else if (VQ) { coeff.Project(*VQ); }
File diff suppressed because it is too large Load Diff
+3
View File
@@ -10,6 +10,7 @@
// CONTRIBUTING.md for details.
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -39,8 +40,10 @@ MassIntegrator::Kernels::Kernels()
MassIntegrator::AddSpecialization<2,9,10>();
// others
MassIntegrator::AddSpecialization<2,2,4>();
MassIntegrator::AddSpecialization<2,2,5>();
MassIntegrator::AddSpecialization<2,3,6>();
MassIntegrator::AddSpecialization<2,4,6>();
// 3D
// Q=P+1
MassIntegrator::AddSpecialization<3,1,1>();
+23 -17
View File
@@ -1408,51 +1408,57 @@ using ApplyKernelType = MassIntegrator::ApplyKernelType;
using DiagonalKernelType = MassIntegrator::DiagonalKernelType;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3)
{
constexpr int MDQ = T_D1D >= T_Q1D ? T_D1D : T_Q1D;
constexpr int MDQ = D1D >= Q1D ? D1D : Q1D;
// max 64 threads in z limit in cuda and hip
if constexpr (MDQ > 0)
{
return internal::SmemPAMassApply3D<T_D1D, T_Q1D,
return internal::SmemPAMassApply3D<D1D, Q1D,
internal::mass::NBZ3D(MDQ)>;
}
}
MFEM_ABORT("");
else { MFEM_ABORT(""); }
return nullptr;
}
inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
int DIM, int, int)
int dim, int, int)
{
if (DIM == 1) { return internal::PAMassApply1D; }
else if (DIM == 2) { return internal::PAMassApply2D; }
else if (DIM == 3) { return internal::PAMassApply3D; }
if (dim == 1) { return internal::PAMassApply1D; }
else if (dim == 2) { return internal::PAMassApply2D; }
else if (dim == 3) { return internal::PAMassApply3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
template<int DIM, int T_D1D, int T_Q1D>
template<int DIM, int D1D, int Q1D>
DiagonalKernelType MassIntegrator::DiagonalPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
}
inline DiagonalKernelType MassIntegrator::DiagonalPAKernels::Fallback(
int DIM, int, int)
int dim, int, int)
{
if (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (DIM == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (DIM == 3) { return internal::PAMassAssembleDiagonal3D; }
if (dim == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (dim == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (dim == 3) { return internal::PAMassAssembleDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+43 -5
View File
@@ -15,6 +15,7 @@
#include "../qfunction.hpp"
#include "../ceed/integrators/mass/mass.hpp"
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp"
namespace mfem
{
@@ -29,9 +30,11 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
dim = mesh->Dimension();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation *T0 = mesh->GetTypicalElementTransformation();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0);
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0, stroud);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -48,17 +51,25 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
return;
}
int map_type = el.GetMapType();
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::DETERMINANTS, mt);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, mt);
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
{
const int NE = ne;
const int NQ = nq;
@@ -147,9 +158,10 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const int D1D = dofs1D;
const int Q1D = quad1D;
const Vector &D = pa_data;
const Array<real_t> &B = maps->B;
const Array<real_t> &Bt = maps->Bt;
const Vector &D = pa_data;
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
@@ -164,7 +176,31 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
}
#endif // MFEM_USE_OCCA
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
const Array<real_t> &Ba1 = rmaps->Ba1;
const Array<real_t> &Ba2 = rmaps->Ba2;
const Array<real_t> &Ba3 = rmaps->Ba3;
const Array<real_t> &Ba1t = rmaps->Ba1t;
const Array<real_t> &Ba2t = rmaps->Ba2t;
const Array<real_t> &Ba3t = rmaps->Ba3t;
const Array<int> &lex_map = rmaps->lex_map;
const Array<int> &forward_map2d = rmaps->forward_map2d_mass;
const Array<int> &inverse_map2d = rmaps->inverse_map2d_mass;
const Array<int> &forward_map3d = rmaps->forward_map3d_mass;
const Array<int> &inverse_map3d = rmaps->inverse_map3d_mass;
ApplySimplexPAKernels::Run(dim, D1D, Q1D, ne, lex_map, forward_map2d,
inverse_map2d,
forward_map3d, inverse_map3d, Ba1, Ba2, Ba3, Ba1t, Ba2t, Ba3t,
D, x, y, D1D, Q1D);
}
else
{
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
}
}
}
@@ -177,6 +213,8 @@ void MassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
}
else
{
MFEM_VERIFY(!fespace->UsesRaggedTensorBasis(),
"AbsMultPA not implemented for ragged tensor basis");
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absB(maps->B);
File diff suppressed because it is too large Load Diff
+376
View File
@@ -236,6 +236,58 @@ IntegrationRule::ApplyToKnotIntervals(KnotVector const& kv) const
return kvir;
}
IntegrationRule IntegrationRule::Reorder(const Array<int> &ordering) const
{
const int np = GetNPoints();
MFEM_VERIFY(np == ordering.Size(), "Invalid permutation size");
IntegrationRule ir(np);
ir.SetOrder(GetOrder());
for (int i = 0; i < np; i++)
{
IntegrationPoint &ip_new = ir.IntPoint(i);
const IntegrationPoint &ip_old = IntPoint(ordering[i]);
ip_new.Set(ip_old.x, ip_old.y, ip_old.z, ip_old.weight);
}
return ir;
}
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim)
{
IntegrationRule ir_mapped(ir.GetNPoints());
ir_mapped.SetOrder(ir.GetOrder());
if (dim == 2)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
ip_mapped.x = ir.IntPoint(i).x;
ip_mapped.weight = ir.IntPoint(i).weight;
}
return ir_mapped;
}
else if (dim == 3)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
ip_mapped.z = ir.IntPoint(i).z * (1 - ir.IntPoint(i).x) * (1 - ir.IntPoint(
i).y);
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
ip_mapped.x = ir.IntPoint(i).x;
ip_mapped.weight = ir.IntPoint(i).weight;
}
return ir_mapped;
}
else
{
MFEM_ABORT("Duffy transformation not implemented for this dimension!");
}
}
#ifdef MFEM_USE_MPFR
// Class for computing hi-precision (HP) quadrature in 1D
@@ -433,6 +485,142 @@ public:
#endif // MFEM_USE_MPFR
void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
const real_t beta, IntegrationRule* ir)
{
/* The np-point Gauss-Jacobi quadrature rule is exact for polynomials of
degree 2np - 1 with weight function w(x) = (1-x)^alpha * x^beta. The
nodes are the zeros of the Jacobi polynomial P_{np}^{alpha,beta} and
the weights are
w_i = C / [(1 - x_i^2) * P'_{np}^{alpha,beta}(x_i)^2]
C = 2^{alpha + beta + 1} * Gamma(np + alpha + 1) * Gamma(np + beta + 1)
/ [Gamma(np + alpha + beta + 1) * Gamma(np + 1)].
The nodes are computed via nonlinear solve (Newton's method) with an
initial guess corresponding to Gatteschi's asymptotic expansions of the
Jacobi polynomial roots [1].
The current initial guess has been tested and performs well for
np <= 200 and -1 <= alpha, beta <= 4. For larger np, it may be necessary
utilize different initial guesses in the vicinity of x = -1,+1 [2].
[1] Gautschi, W., & Giordano, C. (2008). Luigi Gatteschis work on
asymptotics of special functions and their zeros. Numerical Algorithms,
49, 11-31.
[2] Hale, N., & Townsend, A. (2013). Fast and accurate computation of
Gauss--Legendre and Gauss--Jacobi quadrature nodes and weights.
SIAM Journal on Scientific Computing, 35(2), A652-A674.
*/
ir->SetSize(np);
ir->SetPointIndices();
ir->SetOrder(2*np - 1);
if (alpha <= -1.0 || beta <= -1.0)
{
MFEM_ABORT("Gauss-Jacobi quadrature only defined for alpha > -1 and beta > -1");
}
// Jacobi weight function is undefined whenever alpha <= -1 or beta <= -1
if (alpha > 4.0 || beta > 4.0)
{
MFEM_ABORT("Current Gauss-Jacobi quadrature implementation only tested for alpha <= 4 and beta <= 4");
}
// current asymptotic expansions for initial guess may perform poorly for large alpha, beta
switch (np)
{
case 1:
real_t x = (beta - alpha) / (alpha + beta + 2);
real_t w = pow(2, alpha + beta + 1) * tgamma(alpha + 2) * tgamma(
beta + 2) / (tgamma(alpha + beta + 2));
w = 0.5 * w / pow(2, alpha + beta);
// map weight to to [0,1], with additional 1/(2^(alpha + beta)) factor coming from mapping
// the weight (1-x)^alpha * (1+x)^beta to [0,1] as well.
ir->IntPoint(0).Set1w(0.5 * x + 0.5,
4.0 * w / ((1.0 - x*x) * (alpha + beta + 2) * (alpha + beta + 2)));
return;
}
#ifndef MFEM_USE_MPFR
const int n = np;
// common constants for Jacobi polynomials
real_t ab = alpha + beta;
real_t a2_minus_b2 = (alpha - beta) * (alpha + beta);
// roots of P^(alpha,beta)_n in the interval [-1,1]
for (int i = 1; i <= n; i++)
{
// rather than using Chebyshev points for initial guess, use Gatteschi's asymptotic expansion for roots of Jacobi
// polynomials
real_t n_ab_plus_1 = 2 * n + alpha + beta + 1;
real_t v = (2 * i + alpha - 0.5) * M_PI / n_ab_plus_1;
real_t theta = v + 1.0 / (n_ab_plus_1*n_ab_plus_1) * ((0.25 - alpha*alpha) *
1.0/tan(0.5*v) - (0.25 - beta*beta) * tan(0.5*v));
real_t z = cos(theta);
real_t pp, p1, dz, xi = 0.;
bool done = false;
while (1)
{
real_t p2 = 1;
p1 = ((alpha-beta) + (alpha + beta + 2) * z) / 2;
for (int j = 1; j <= n-1; j++)
{
real_t p3 = p2;
p2 = p1;
real_t jx2_ab = 2 * j + ab;
real_t an = (jx2_ab) * (jx2_ab + 2);
real_t bn = a2_minus_b2;
real_t cn = 2 * (j + alpha) * (j + beta) * (jx2_ab + 2) / (jx2_ab + 1);
real_t D = (jx2_ab + 1) / (2 * (j + 1) * (j + ab + 1) * (jx2_ab));
p1 = ((an * z + bn) * p2 - cn * p3) * D;
}
// p1 is Jacobi polynomial
pp = n * (alpha - beta - (2 * n + ab) * z) * p1 + 2 * (n + alpha) *
(n + beta) * p2;
pp = pp / ((2 * n + ab) * (1 - z*z));
// derivative of the Jacobi polynomial
if (done) { break; }
dz = p1/pp;
#ifdef MFEM_USE_SINGLE
if (std::abs(dz) < 1e-7)
#elif defined MFEM_USE_DOUBLE
if (std::abs(dz) < std::numeric_limits<real_t>::epsilon())
// this seems to cause trouble if we try std::abs(dz) < 1e-16
#else
MFEM_ABORT("Floating point type undefined");
// if (std::abs(dz) < 1e-16)
#endif
{
done = true;
xi = z - dz;
}
z -= dz;
}
real_t c0 = exp(lgamma(n + alpha + 1) - lgamma(n + ab + 1)) * exp(lgamma(
n + beta + 1) - lgamma(n + 1));
// ratio of gamma functions prone to overflow for large n, so compute logarithms
// of Gamma function instead, i.e. Gamma(a)/Gamma(b) = exp(lgamma(a) - lgamma(b))
ir->IntPoint(n-i).x = 0.5 * xi + 0.5;
ir->IntPoint(n-i).weight = 0.5 * c0 * pow(2.0,
ab + 1) / ((1.0 - xi*xi)*pp*pp) / pow(2, ab);
// map nodes and weights to the interval [0,1]
}
#else // MFEM_USE_MPFR is defined
MFEM_ABORT("MPFR implementation of Gauss-Jacobi quadrature not defined yet");
#endif // MFEM_USE_MPFR
}
void QuadratureFunctions1D::GaussLegendre(const int np, IntegrationRule* ir)
{
ir->SetSize(np);
@@ -2362,6 +2550,194 @@ IntegrationRule *IntegrationRules::CubeIntegrationRule(int Order)
return CubeIntRules[Order];
}
StroudIntegrationRules StroudIntRules;
StroudIntegrationRules::StroudIntegrationRules()
{
const MemoryType h_mt = MemoryType::HOST;
SquareStroudIntRules.SetSize(32, h_mt);
SquareStroudIntRules = NULL;
TriangleStroudIntRules.SetSize(32, h_mt);
TriangleStroudIntRules = NULL;
CubeStroudIntRules.SetSize(32, h_mt);
CubeStroudIntRules = NULL;
TetrahedronStroudIntRules.SetSize(32, h_mt);
TetrahedronStroudIntRules = NULL;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
IntRuleLocks.SetSize(Geometry::NUM_GEOMETRIES, h_mt);
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
{
omp_init_lock(&IntRuleLocks[i]);
}
#endif
}
const IntegrationRule &StroudIntegrationRules::Get(int GeomType, int Order)
{
Array<IntegrationRule *> *ir_array = NULL;
switch (GeomType)
{
case Geometry::TRIANGLE: ir_array = &TriangleStroudIntRules; break;
case Geometry::TETRAHEDRON: ir_array = &TetrahedronStroudIntRules; break;
case Geometry::INVALID:
case Geometry::NUM_GEOMETRIES:
MFEM_ABORT("Unknown type of reference element!");
default:
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
}
if (Order < 0)
{
Order = 0;
}
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
omp_set_lock(&IntRuleLocks[GeomType]);
#endif
if (!HaveIntRule(*ir_array, Order))
{
IntegrationRule *ir = GenerateIntegrationRule(GeomType, Order);
#ifdef MFEM_DEBUG
int RealOrder = Order;
while (RealOrder+1 < ir_array->Size() && (*ir_array)[RealOrder+1] == ir)
{
RealOrder++;
}
MFEM_VERIFY(RealOrder == ir->GetOrder(), "internal error");
#else
MFEM_CONTRACT_VAR(ir);
#endif
}
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
omp_unset_lock(&IntRuleLocks[GeomType]);
#endif
return *(*ir_array)[Order];
}
void StroudIntegrationRules::DeleteIntRuleArray(
Array<IntegrationRule *> &ir_array) const
{
// Many of the intrules have multiple contiguous copies in the ir_array
// so we have to be careful to not delete them twice.
IntegrationRule *ir = NULL;
for (int i = 0; i < ir_array.Size(); i++)
{
if (ir_array[i] != NULL && ir_array[i] != ir)
{
ir = ir_array[i];
delete ir;
}
}
}
StroudIntegrationRules::~StroudIntegrationRules()
{
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
{
omp_destroy_lock(&IntRuleLocks[i]);
}
#endif
DeleteIntRuleArray(SquareStroudIntRules);
DeleteIntRuleArray(TriangleStroudIntRules);
DeleteIntRuleArray(CubeStroudIntRules);
DeleteIntRuleArray(TetrahedronStroudIntRules);
}
IntegrationRule *StroudIntegrationRules::GenerateIntegrationRule(int GeomType,
int Order)
{
switch (GeomType)
{
case Geometry::TRIANGLE:
return TriangleStroudIntegrationRule(Order);
case Geometry::TETRAHEDRON:
return TetrahedronStroudIntegrationRule(Order);
case Geometry::INVALID:
case Geometry::NUM_GEOMETRIES:
MFEM_ABORT("Unknown type of reference element!");
default:
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
}
return NULL;
}
/* Integration rule in reference triangle according to tensor product Gauss-Jacobi rule.
The nodes and weights are used in the original form defined on the reference
square to evaluate the component 1D basis functions. Mapping to the reference
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
IntegrationRule *StroudIntegrationRules::TriangleStroudIntegrationRule(
int Order)
{
int RealOrder = GetSegmentRealOrder(Order);
// Order is one of {RealOrder-1,RealOrder}
// if (!HaveIntRule(SegmentIntRules, RealOrder))
// {
// SegmentIntegrationRule(RealOrder);
// }
IntegrationRule ir_0_0;
// Gauss-Jacobi is exact for 2*n-1
int n = RealOrder/2 + 1;
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
IntegrationRule ir_1_0;
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
AllocIntRule(TriangleStroudIntRules, RealOrder); // RealOrder >= Order
// create rule in unit square
TriangleStroudIntRules[RealOrder-1] =
TriangleStroudIntRules[RealOrder] =
new IntegrationRule(ir_1_0, ir_0_0);
// map rule to reference triangle
// TriangleStroudIntRules[RealOrder-1]->DuffyTrans(2);
*TriangleStroudIntRules[RealOrder-1] =
DuffyTrans(*TriangleStroudIntRules[RealOrder-1], 2);
return TriangleStroudIntRules[Order];
}
/* Integration rule in reference tetrahedron according to tensor product Gauss-Jacobi rule.
The nodes and weights are used in the original form defined on the reference
square to evaluate the component 1D basis functions. Mapping to the reference
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
IntegrationRule *StroudIntegrationRules::TetrahedronStroudIntegrationRule(
int Order)
{
int RealOrder = GetSegmentRealOrder(Order);
// Order is one of {RealOrder-1,RealOrder}
IntegrationRule ir_0_0;
int n = RealOrder/2 + 1;
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
IntegrationRule ir_1_0;
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
IntegrationRule ir_2_0;
QuadratureFunctions1D::GaussJacobi(n, 2.0, 0.0, &ir_2_0);
AllocIntRule(TetrahedronStroudIntRules, RealOrder); // RealOrder >= Order
// create rule in unit cube
TetrahedronStroudIntRules[RealOrder-1] =
TetrahedronStroudIntRules[RealOrder] =
new IntegrationRule(ir_2_0, ir_1_0, ir_0_0);
// map rule to reference tetrahedron
// TetrahedronStroudIntRules[RealOrder-1]->DuffyTrans(3);
*TetrahedronStroudIntRules[RealOrder-1] =
DuffyTrans(*TetrahedronStroudIntRules[RealOrder-1], 3);
return TetrahedronStroudIntRules[Order];
}
IntegrationRule& NURBSMeshRules::GetElementRule(const int elem,
const int patch, const int *ijk,
Array<const KnotVector*> const& kv) const
+68
View File
@@ -269,6 +269,13 @@ public:
/// applying this rule on each knot interval.
IntegrationRule* ApplyToKnotIntervals(KnotVector const& kv) const;
/** @brief Returns an integration rule such that the new IntegrationPoints
* are re-ordered based on @a ordering.
*
* @details In the new integration rule, ip_new[i] = ip_old[ordering[i]]
*/
IntegrationRule Reorder(const Array<int> &ordering) const;
/// Destroys an IntegrationRule object
~IntegrationRule() { }
};
@@ -378,6 +385,8 @@ public:
These methods calculate the actual points and weights for the different
types of quadrature rules. */
///@{
static void GaussJacobi(const int np, const real_t alpha, const real_t beta,
IntegrationRule* ir);
static void GaussLegendre(const int np, IntegrationRule* ir);
static void GaussLobatto(const int np, IntegrationRule *ir);
static void OpenUniform(const int np, IntegrationRule *ir);
@@ -487,12 +496,71 @@ public:
~IntegrationRules();
};
/// Container class for integration rules
class StroudIntegrationRules
{
private:
Array<IntegrationRule *> SquareStroudIntRules;
Array<IntegrationRule *> TriangleStroudIntRules;
Array<IntegrationRule *> CubeStroudIntRules;
Array<IntegrationRule *> TetrahedronStroudIntRules;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
Array<omp_lock_t> IntRuleLocks;
#endif
void AllocIntRule(Array<IntegrationRule *> &ir_array, int Order) const
{
if (ir_array.Size() <= Order)
{
ir_array.SetSize(Order + 1, NULL);
}
}
bool HaveIntRule(Array<IntegrationRule *> &ir_array, int Order) const
{
return (ir_array.Size() > Order && ir_array[Order] != NULL);
}
int GetSegmentRealOrder(int Order) const
{
return Order | 1; // valid for all quad_type's
}
void DeleteIntRuleArray(Array<IntegrationRule *> &ir_array) const;
/// The following methods allocate new IntegrationRule objects without
/// checking if they already exist. To avoid memory leaks use
/// IntegrationRules::Get(int GeomType, int Order) instead.
IntegrationRule *GenerateIntegrationRule(int GeomType, int Order);
IntegrationRule *TriangleStroudIntegrationRule(int Order);
IntegrationRule *TetrahedronStroudIntegrationRule(int Order);
public:
/// Sets initial sizes for the integration rule arrays, but rules
/// are defined the first time they are requested with the Get method.
explicit StroudIntegrationRules();
/// Returns a Stroud integration rule for given GeomType and Order.
const IntegrationRule &Get(int GeomType, int Order);
/// Destroys an StroudIntegrationRules object
~StroudIntegrationRules();
};
/// A global object with all integration rules (defined in intrules.cpp)
extern MFEM_EXPORT IntegrationRules IntRules;
/// A global object with all refined integration rules
extern MFEM_EXPORT IntegrationRules RefinedIntRules;
/// A global object with all Stroud integration rules (defined in intrules.cpp)
extern MFEM_EXPORT StroudIntegrationRules StroudIntRules;
/// Duffy Transformation of 2D and 3D tensor product rules of the form
/// $X(t) = \sum_{i=1}^{d+1} \lambda_i(t) * x_i$, where $x_i$ are the vertices
/// of the simplex and $\lambda_i = t_i * (1-\lambda_1-...-\lambda_{i-1})$, with
/// $t$ being the coordinates in the unit square/cube. This function is used only
/// in the partial assembly of Bernstein elements on simplices and does NOT
/// modify the quadrature weights.
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim);
}
#endif
+9 -2
View File
@@ -50,14 +50,21 @@ ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
const FiniteElement *fe = fes.GetFE(e);
auto el_t = dynamic_cast<const TensorBasisElement*>(fe);
auto el_n = dynamic_cast<const NodalFiniteElement*>(fe);
if (el_t || el_n) { continue; }
auto el_p = dynamic_cast<const H1Pos_TriangleElement*>(fe) ||
dynamic_cast<const H1Pos_TetrahedronElement*>(fe);
if (el_t || el_n || el_p) { continue; }
MFEM_ABORT("Finite element not suitable for lexicographic ordering");
}
const FiniteElement *fe = fes.GetTypicalFE();
auto el_t = dynamic_cast<const TensorBasisElement*>(fe);
auto el_n = dynamic_cast<const NodalFiniteElement*>(fe);
auto el_p_tri = dynamic_cast<const H1Pos_TriangleElement*>(fe);
auto el_p_tet = dynamic_cast<const H1Pos_TetrahedronElement*>(fe);
const Array<int> &fe_dof_map =
(el_t) ? el_t->GetDofMap() : el_n->GetLexicographicOrdering();
el_n ? el_n->GetLexicographicOrdering() :
el_t ? el_t->GetDofMap() :
el_p_tri ? el_p_tri->GetDofMap() :
el_p_tet->GetDofMap();
MFEM_VERIFY(fe_dof_map.Size() > 0, "invalid dof map");
dof_map = fe_dof_map.HostRead();
}
+303 -113
View File
@@ -3758,7 +3758,8 @@ void TMOP_Integrator::SetInitialMeshPos(const GridFunction *x0)
TMOP_Integrator::~TMOP_Integrator()
{
delete lim_func;
delete adapt_lim_gf;
for (int i = 0; i < adapt_lim_gf.Size(); i++) { delete adapt_lim_gf[i]; }
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { delete adapt_lim_gf0[i]; }
delete surf_fit_gf;
delete surf_fit_limiter;
delete surf_fit_grad;
@@ -3800,20 +3801,13 @@ void TMOP_Integrator::EnableAdaptiveLimiting(const GridFunction &z0,
AdaptivityEvaluator &ae,
real_t delta_max)
{
MFEM_VERIFY(delta_max > 0.0,
"EnableAdaptiveLimiting requires delta_max > 0.0.");
adapt_lim_gf0 = &z0;
delete adapt_lim_gf;
adapt_lim_gf = new GridFunction(z0);
adapt_lim_coeff = &coeff;
adapt_lim_eval = &ae;
adapt_lim_delta_max = delta_max;
adapt_lim_eval->SetSerialMetaInfo(*z0.FESpace()->GetMesh(),
*z0.FESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
Array<const GridFunction *> z0_arr(1);
Array<Coefficient *> c_arr(1);
Array<real_t> d_arr(1);
z0_arr[0] = &z0;
c_arr[0] = &coeff;
d_arr[0] = delta_max;
EnableAdaptiveLimiting(z0_arr, c_arr, ae, d_arr);
}
#ifdef MFEM_USE_MPI
@@ -3822,21 +3816,111 @@ void TMOP_Integrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
AdaptivityEvaluator &ae,
real_t delta_max)
{
MFEM_VERIFY(delta_max > 0.0,
"EnableAdaptiveLimiting requires delta_max > 0.0.");
Array<const ParGridFunction *> z0_arr(1);
Array<Coefficient *> c_arr(1);
Array<real_t> d_arr(1);
z0_arr[0] = &z0;
c_arr[0] = &coeff;
d_arr[0] = delta_max;
EnableAdaptiveLimiting(z0_arr, c_arr, ae, d_arr);
}
#endif
adapt_lim_gf0 = &z0;
adapt_lim_pgf0 = &z0;
delete adapt_lim_gf;
adapt_lim_gf = new GridFunction(z0);
adapt_lim_coeff = &coeff;
void TMOP_Integrator::
EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(z0.Size() > 0, "Requires at least one field.");
MFEM_VERIFY(z0.Size() == coeff.Size(), "Requires one Coefficient per field.");
MFEM_VERIFY(z0.Size() == delta_max.Size(), "Requires one delta_max per field.");
for (int i = 0; i < delta_max.Size(); i++)
{
MFEM_VERIFY(delta_max[i] > 0.0, "Requires delta_max > 0.0.");
}
// Verify compatibility of input fields.
const FiniteElementSpace *sfes = z0[0]->FESpace();
MFEM_VERIFY(sfes->GetVDim() == 1, "Expects scalar input GridFunctions.");
const int ndofs = sfes->GetVSize();
Mesh *mesh = sfes->GetMesh();
MFEM_VERIFY(mesh->GetNodes(), "EnableAdaptiveLimiting requires mesh Nodes.");
for (int i = 0; i < z0.Size(); i++)
{
MFEM_VERIFY(z0[i], "NULL GridFunction pointer.");
const FiniteElementSpace *fes_i = z0[i]->FESpace();
MFEM_VERIFY(fes_i->GetVDim() == 1, "Expects scalar input GridFunctions.");
MFEM_VERIFY(fes_i->GetVSize() == ndofs,
"All fields must be on the same FE space.");
MFEM_VERIFY(fes_i->GetMesh() == mesh,
"All fields must be on the same Mesh.");
MFEM_VERIFY(coeff[i], "NULL Coefficient pointer.");
}
// Delete previous adaptive limiting data.
for (int i = 0; i < adapt_lim_gf.Size(); i++) { delete adapt_lim_gf[i]; }
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { delete adapt_lim_gf0[i]; }
adapt_lim_coeff.SetSize(coeff.Size());
for (int i = 0; i < coeff.Size(); i++) { adapt_lim_coeff[i] = coeff[i]; }
adapt_lim_eval = &ae;
adapt_lim_delta_max = delta_max;
adapt_lim_init_nodes = *mesh->GetNodes();
adapt_lim_eval->SetParMetaInfo(*z0.ParFESpace()->GetParMesh(),
*z0.ParFESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
// Use one internal vector field (vdim = #fields) so remapping can be done in
// one call and incremental remap state (when provided by the evaluator) is
// preserved across TMOP iterations.
//
// Use Ordering::byNODES for the packed vector field so packing / unpacking
// can be done with contiguous sub-vector copies (device-friendly).
const int nal = z0.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
// Setup the evaluator.
#ifdef MFEM_USE_MPI
if (auto pfes = dynamic_cast<const ParFiniteElementSpace *>(sfes))
{
auto *pm = pfes->GetParMesh();
MFEM_VERIFY(pm, "Invalid ParMesh.");
ParFiniteElementSpace vfes(pm, pfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetParMetaInfo(*pm, vfes);
}
else
#endif
{
FiniteElementSpace vfes(mesh, sfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetSerialMetaInfo(*mesh, vfes);
}
// Copy the initial fields; remapped fields are initialized to the same data.
adapt_lim_gf0.SetSize(z0.Size());
adapt_lim_gf.SetSize(z0.Size());
for (int i = 0; i < z0.Size(); i++)
{
adapt_lim_gf0[i] = new GridFunction(*z0[i]);
adapt_lim_gf[i] = new GridFunction(*z0[i]);
}
// Initialize the evaluator with the packed vector field.
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::
EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
Array<const GridFunction *> z0_base(z0.Size());
for (int i = 0; i < z0.Size(); i++) { z0_base[i] = z0[i]; }
EnableAdaptiveLimiting(z0_base, coeff, ae, delta_max);
}
#endif
@@ -4157,26 +4241,61 @@ void TMOP_Integrator::GetSurfaceFittingErrors(const Vector &d_loc,
void TMOP_Integrator::UpdateAfterMeshTopologyChange()
{
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_gf->Update();
adapt_lim_eval->SetSerialMetaInfo(*adapt_lim_gf->FESpace()->GetMesh(),
*adapt_lim_gf->FESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { adapt_lim_gf0[i]->Update(); }
for (int i = 0; i < adapt_lim_gf.Size(); i++) { adapt_lim_gf[i]->Update(); }
Mesh *mesh = adapt_lim_gf[0]->FESpace()->GetMesh();
// Same setup as in EnableAdaptiveLimiting().
const int nal = adapt_lim_coeff.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
FiniteElementSpace vfes(mesh, adapt_lim_gf[0]->FESpace()->FEColl(), nal,
packed_ord);
adapt_lim_eval->SetSerialMetaInfo(*mesh, vfes);
adapt_lim_init_nodes = *mesh->GetNodes();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::ParUpdateAfterMeshTopologyChange()
{
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_gf->Update();
adapt_lim_eval->SetParMetaInfo(*adapt_lim_pgf0->ParFESpace()->GetParMesh(),
*adapt_lim_pgf0->ParFESpace());
adapt_lim_eval->SetInitialField
(*adapt_lim_gf->FESpace()->GetMesh()->GetNodes(), *adapt_lim_gf);
for (int i = 0; i < adapt_lim_gf0.Size(); i++) { adapt_lim_gf0[i]->Update(); }
for (int i = 0; i < adapt_lim_gf.Size(); i++) { adapt_lim_gf[i]->Update(); }
// Same setup as in EnableAdaptiveLimiting().
auto *pfes = dynamic_cast<ParFiniteElementSpace *>(adapt_lim_gf[0]->FESpace());
MFEM_VERIFY(pfes, "internal error");
ParMesh *pmesh = pfes->GetParMesh();
const int nal = adapt_lim_coeff.Size();
const Ordering::Type packed_ord = Ordering::byNODES;
ParFiniteElementSpace vfes(pmesh, pfes->FEColl(), nal, packed_ord);
adapt_lim_eval->SetParMetaInfo(*pmesh, vfes);
adapt_lim_init_nodes = *pmesh->GetNodes();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector init_field_vec;
init_field_vec.SetSize(nal * ndofs, *adapt_lim_gf0[0]);
init_field_vec.UseDevice(adapt_lim_gf0[0]->UseDevice());
for (int c = 0; c < nal; c++)
{
init_field_vec.SetVector(*adapt_lim_gf0[c], c * ndofs);
}
adapt_lim_eval->SetInitialField(adapt_lim_init_nodes, init_field_vec);
}
}
#endif
@@ -4208,7 +4327,8 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// No adaptive limiting / surface fitting terms if the function is called
// as part of a FD derivative computation (because we include the exact
// derivatives of these terms in FD computations).
const bool adaptive_limiting = (adapt_lim_gf && fd_call_flag == false);
const bool adaptive_limiting = (adapt_lim_gf.Size() > 0 &&
fd_call_flag == false);
const bool surface_fit = (surf_fit_marker && fd_call_flag == false);
DSh.SetSize(dof, dim);
@@ -4271,11 +4391,21 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// the physical coordinates (i.e. changes in 'elfun'), e.g. when the
// coefficient is a ConstantCoefficient or a GridFunctionCoefficient.
const int nal = adapt_lim_coeff.Size();
const int nqp = ir.GetNPoints();
Vector adapt_lim_gf_q, adapt_lim_gf0_q;
if (adaptive_limiting)
{
adapt_lim_gf->GetValues(el_id, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(el_id, ir, adapt_lim_gf0_q);
adapt_lim_gf_q.SetSize(nal * nqp);
adapt_lim_gf0_q.SetSize(nal * nqp);
Vector zc, z0c;
for (int c = 0; c < nal; c++)
{
zc.MakeRef(adapt_lim_gf_q, c * nqp, nqp);
z0c.MakeRef(adapt_lim_gf0_q, c * nqp, nqp);
adapt_lim_gf[c]->GetValues(el_id, ir, zc);
adapt_lim_gf0[c]->GetValues(el_id, ir, z0c);
}
}
for (int i = 0; i < ir.GetNPoints(); i++)
@@ -4307,9 +4437,13 @@ real_t TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
// Contribution from the adaptive limiting term.
if (adaptive_limiting)
{
const real_t diff = (adapt_lim_gf_q(i) - adapt_lim_gf0_q(i)) /
adapt_lim_delta_max;
val += adapt_lim_coeff->Eval(*Tpr, ip) * lim_normal * diff * diff;
for (int c = 0; c < nal; c++)
{
const int idx = c * nqp + i;
const real_t diff = (adapt_lim_gf_q(idx) - adapt_lim_gf0_q(idx)) /
adapt_lim_delta_max[c];
val += adapt_lim_coeff[c]->Eval(*Tpr, ip) * lim_normal * diff * diff;
}
}
energy += weight * val;
@@ -4602,7 +4736,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (metric_coeff || lim_coeff || adapt_lim_gf ||
if (metric_coeff || lim_coeff || adapt_lim_gf.Size() > 0 ||
surf_fit_gf || surf_fit_pos || exact_action)
{
Tpr = new IsoparametricTransformation;
@@ -4700,7 +4834,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
}
}
if (adapt_lim_gf) { AssembleElemVecAdaptLim(el, *Tpr, ir, weights, PMatO); }
if (adapt_lim_gf.Size() > 0) { AssembleElemVecAdaptLim(el, *Tpr, ir, weights, PMatO); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemVecSurfFit(el, *Tpr, PMatO); }
delete Tpr;
@@ -4774,7 +4908,8 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (metric_coeff || lim_coeff || adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (metric_coeff || lim_coeff || adapt_lim_gf.Size() > 0 ||
surf_fit_gf || surf_fit_pos)
{
Tpr = new IsoparametricTransformation;
Tpr->SetFE(&el);
@@ -4829,7 +4964,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
}
}
if (adapt_lim_gf) { AssembleElemGradAdaptLim(el, *Tpr, ir, weights, elmat); }
if (adapt_lim_gf.Size() > 0) { AssembleElemGradAdaptLim(el, *Tpr, ir, weights, elmat); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemGradSurfFit(el, *Tpr, elmat);}
delete Tpr;
@@ -4842,34 +4977,42 @@ void TMOP_Integrator::AssembleElemVecAdaptLim(const FiniteElement &el,
DenseMatrix &mat)
{
const int dof = el.GetDof(), dim = el.GetDim(), nqp = weights.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q, adapt_lim_gf0_q(nqp);
const int nal = adapt_lim_coeff.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q(nqp), adapt_lim_gf0_q(nqp);
Array<int> dofs;
adapt_lim_gf->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
adapt_lim_gf->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf[0]->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
// Project the gradient of adapt_lim_gf in the same space.
// The FE coefficients of the gradient go in adapt_lim_gf_grad_e.
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
el.ProjectGrad(el, Tpr, grad_phys);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
Vector adapt_lim_gf_grad_q(dim);
for (int q = 0; q < nqp; q++)
for (int c = 0; c < nal; c++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
const real_t delta2 = adapt_lim_delta_max[c] * adapt_lim_delta_max[c];
adapt_lim_gf[c]->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
adapt_lim_gf_grad_q *= 2.0 * (adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) /
adapt_lim_delta_max / adapt_lim_delta_max;
adapt_lim_gf_grad_q *= weights(q) * lim_normal * adapt_lim_coeff->Eval(Tpr, ip);
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
AddMultVWt(shape, adapt_lim_gf_grad_q, mat);
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
adapt_lim_gf_grad_q *= 2.0 * (adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) /
delta2;
adapt_lim_gf_grad_q *=
weights(q) * lim_normal * adapt_lim_coeff[c]->Eval(Tpr, ip);
AddMultVWt(shape, adapt_lim_gf_grad_q, mat);
}
}
}
@@ -4880,60 +5023,66 @@ void TMOP_Integrator::AssembleElemGradAdaptLim(const FiniteElement &el,
DenseMatrix &mat)
{
const int dof = el.GetDof(), dim = el.GetDim(), nqp = weights.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q, adapt_lim_gf0_q(nqp);
const int nal = adapt_lim_coeff.Size();
Vector shape(dof), adapt_lim_gf_e, adapt_lim_gf_q(nqp), adapt_lim_gf0_q(nqp);
Array<int> dofs;
adapt_lim_gf->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
adapt_lim_gf->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf[0]->FESpace()->GetElementDofs(Tpr.ElementNo, dofs);
// Project the gradient of adapt_lim_gf in the same space.
// The FE coefficients of the gradient go in adapt_lim_gf_grad_e.
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
el.ProjectGrad(el, Tpr, grad_phys);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
// Project the gradient of each gradient of adapt_lim_gf in the same space.
// The FE coefficients of the second derivatives go in adapt_lim_gf_hess_e.
DenseMatrix adapt_lim_gf_hess_e(dof*dim, dim);
Mult(grad_phys, adapt_lim_gf_grad_e, adapt_lim_gf_hess_e);
// Reshape to be more convenient later (no change in the data).
adapt_lim_gf_hess_e.SetSize(dof, dim*dim);
Vector adapt_lim_gf_grad_q(dim);
DenseMatrix adapt_lim_gf_hess_q(dim, dim);
for (int q = 0; q < nqp; q++)
for (int c = 0; c < nal; c++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
const real_t delta2 = adapt_lim_delta_max[c] * adapt_lim_delta_max[c];
adapt_lim_gf[c]->GetSubVector(dofs, adapt_lim_gf_e);
adapt_lim_gf[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf_q);
adapt_lim_gf0[c]->GetValues(Tpr.ElementNo, ir, adapt_lim_gf0_q);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
Vector gg_ptr(adapt_lim_gf_hess_q.GetData(), dim*dim);
adapt_lim_gf_hess_e.MultTranspose(shape, gg_ptr);
DenseMatrix adapt_lim_gf_grad_e(dof, dim);
Vector grad_ptr(adapt_lim_gf_grad_e.GetData(), dof*dim);
grad_phys.Mult(adapt_lim_gf_e, grad_ptr);
const real_t coeff = adapt_lim_coeff->Eval(Tpr, ip);
const real_t factor =
weights(q) * lim_normal * coeff * 2.0 /
(adapt_lim_delta_max * adapt_lim_delta_max);
// Project the gradient of each gradient of adapt_lim_gf in the same space.
// The FE coefficients of the second derivatives go in adapt_lim_gf_hess_e.
DenseMatrix adapt_lim_gf_hess_e(dof*dim, dim);
Mult(grad_phys, adapt_lim_gf_grad_e, adapt_lim_gf_hess_e);
// Reshape to be more convenient later (no change in the data).
adapt_lim_gf_hess_e.SetSize(dof, dim*dim);
for (int i = 0; i < dof * dim; i++)
for (int q = 0; q < nqp; q++)
{
const int idof = i % dof, idim = i / dof;
for (int j = 0; j <= i; j++)
const IntegrationPoint &ip = ir.IntPoint(q);
el.CalcShape(ip, shape);
adapt_lim_gf_grad_e.MultTranspose(shape, adapt_lim_gf_grad_q);
Vector gg_ptr(adapt_lim_gf_hess_q.GetData(), dim*dim);
adapt_lim_gf_hess_e.MultTranspose(shape, gg_ptr);
const real_t coeff_q = adapt_lim_coeff[c]->Eval(Tpr, ip);
const real_t factor =
weights(q) * lim_normal * coeff_q * 2.0 /
delta2;
for (int i = 0; i < dof * dim; i++)
{
const int jdof = j % dof, jdim = j / dof;
const real_t entry =
factor *
(adapt_lim_gf_grad_q(idim) * shape(idof) *
adapt_lim_gf_grad_q(jdim) * shape(jdof) +
(adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) *
adapt_lim_gf_hess_q(idim, jdim) * shape(idof) * shape(jdof));
mat(i, j) += entry;
if (i != j) { mat(j, i) += entry; }
const int idof = i % dof, idim = i / dof;
for (int j = 0; j <= i; j++)
{
const int jdof = j % dof, jdim = j / dof;
const real_t entry =
factor *
(adapt_lim_gf_grad_q(idim) * shape(idof) *
adapt_lim_gf_grad_q(jdim) * shape(jdof) +
(adapt_lim_gf_q(q) - adapt_lim_gf0_q(q)) *
adapt_lim_gf_hess_q(idim, jdim) * shape(idof) * shape(jdof));
mat(i, j) += entry;
if (i != j) { mat(j, i) += entry; }
}
}
}
}
@@ -5206,7 +5355,7 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
fd_call_flag = false;
// Contributions from adaptive limiting, surface fitting (exact derivatives).
if (adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (adapt_lim_gf.Size() > 0 || surf_fit_gf || surf_fit_pos)
{
const IntegrationRule &ir = ActionIntegrationRule(el);
const int nqp = ir.GetNPoints();
@@ -5230,7 +5379,7 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
}
PMatO.UseExternalData(elvect.GetData(), dof, dim);
if (adapt_lim_gf) { AssembleElemVecAdaptLim(el, Tpr, ir, weights, PMatO); }
if (adapt_lim_gf.Size() > 0) { AssembleElemVecAdaptLim(el, Tpr, ir, weights, PMatO); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemVecSurfFit(el, Tpr, PMatO); }
}
}
@@ -5316,7 +5465,7 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
fd_call_flag = false;
// Contributions from adaptive limiting.
if (adapt_lim_gf || surf_fit_gf || surf_fit_pos)
if (adapt_lim_gf.Size() > 0 || surf_fit_gf || surf_fit_pos)
{
const IntegrationRule &ir = GradientIntegrationRule(el);
const int nqp = ir.GetNPoints();
@@ -5339,7 +5488,7 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
ir.IntPoint(q).weight;
}
if (adapt_lim_gf) { AssembleElemGradAdaptLim(el, Tpr, ir, weights, elmat); }
if (adapt_lim_gf.Size() > 0) { AssembleElemGradAdaptLim(el, Tpr, ir, weights, elmat); }
if (surf_fit_gf || surf_fit_pos) { AssembleElemGradSurfFit(el, Tpr, elmat); }
}
}
@@ -5686,9 +5835,22 @@ UpdateAfterMeshPositionChange(const Vector &d, const FiniteElementSpace &d_fes)
}
// Update adapt_lim_gf if adaptive limiting is enabled.
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
adapt_lim_eval->ComputeAtNewPosition(x_loc, *adapt_lim_gf, ordering);
// All adapt_lim_gf are remapped as a multi-component vector.
const int nal = adapt_lim_coeff.Size();
const int ndofs = adapt_lim_gf0[0]->Size();
Vector new_field_vec;
new_field_vec.SetSize(nal * ndofs, *adapt_lim_gf[0]);
new_field_vec.UseDevice(adapt_lim_gf[0]->UseDevice());
adapt_lim_eval->ComputeAtNewPosition(x_loc, new_field_vec, ordering);
for (int c = 0; c < nal; c++)
{
const real_t *src = new_field_vec.Read() + c * ndofs;
real_t *dst = adapt_lim_gf[c]->Write();
internal::device_copy(dst, src, ndofs);
}
if (PA.enabled)
{
PA.AL_grads_assembled = false;
@@ -5698,9 +5860,17 @@ UpdateAfterMeshPositionChange(const Vector &d, const FiniteElementSpace &d_fes)
// Refresh PA.ALF from the updated adapt_lim_gf.
const ElementDofOrdering ord = ElementDofOrdering::LEXICOGRAPHIC;
const Operator *alf_R =
adapt_lim_gf->FESpace()->GetElementRestriction(ord);
alf_R->Mult(*adapt_lim_gf, PA.ALF);
const FiniteElementSpace *alfes = adapt_lim_gf[0]->FESpace();
const Operator *alf_R = alfes->GetElementRestriction(ord);
const int Esize = alf_R->Height();
Vector ALFc;
for (int c = 0; c < nal; c++)
{
MFEM_VERIFY(adapt_lim_gf[c]->Size() == ndofs, "internal error");
ALFc.MakeRef(PA.ALF, c * Esize, Esize);
alf_R->Mult(*adapt_lim_gf[c], ALFc);
}
// Step 2 of PA.ALFmF0 update: add the new ALF.
PA.ALFmF0 += PA.ALF;
@@ -5917,7 +6087,7 @@ ComputeUntangleMetricQuantiles(const Vector &d, const FiniteElementSpace &fes)
dynamic_cast<const ParFiniteElementSpace *>(&fes);
#endif
if (wcuo && wcuo->GetBarrierType() ==
if (wcuo->GetBarrierType() ==
TMOP_WorstCaseUntangleOptimizer_Metric::BarrierType::Shifted)
{
real_t min_detT = ComputeMinDetT(x_loc, fes);
@@ -5929,7 +6099,7 @@ ComputeUntangleMetricQuantiles(const Vector &d, const FiniteElementSpace &fes)
MPITypeMap<real_t>::mpi_type, MPI_MIN, pfes->GetComm());
}
#endif
if (wcuo) { wcuo->SetMinDetT(min_detT_all); }
wcuo->SetMinDetT(min_detT_all);
}
real_t max_muT = ComputeUntanglerMaxMuBarrier(x_loc, fes);
@@ -5975,6 +6145,16 @@ void TMOPComboIntegrator::EnableAdaptiveLimiting(const GridFunction &z0,
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
void TMOPComboIntegrator::
EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(tmopi.Size() > 0, "No TMOP_Integrators were added.");
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
#ifdef MFEM_USE_MPI
void TMOPComboIntegrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
Coefficient &coeff,
@@ -5985,6 +6165,16 @@ void TMOPComboIntegrator::EnableAdaptiveLimiting(const ParGridFunction &z0,
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
void TMOPComboIntegrator::
EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae, const Array<real_t> &delta_max)
{
MFEM_VERIFY(tmopi.Size() > 0, "No TMOP_Integrators were added.");
tmopi[0]->EnableAdaptiveLimiting(z0, coeff, ae, delta_max);
}
#endif
void TMOPComboIntegrator::SetLimitingNodes(const GridFunction &n0)
+34 -11
View File
@@ -2038,14 +2038,17 @@ protected:
real_t lim_normal;
// Adaptive limiting.
const GridFunction *adapt_lim_gf0; // Not owned.
#ifdef MFEM_USE_MPI
const ParGridFunction *adapt_lim_pgf0;
#endif
GridFunction *adapt_lim_gf; // Owned. Updated by adapt_lim_eval.
Coefficient *adapt_lim_coeff; // Not owned.
AdaptivityEvaluator *adapt_lim_eval; // Not owned.
real_t adapt_lim_delta_max = 1.0;
// Adaptive limiting fields. Each field adds a term to the integral:
// int [ c_k (z_k(x) - z_k0(x0))^2 / delta_max_k^2 ] dx
// with one Coefficient per field. The fields z_k(x) are remapped from their
// initial values z_k0(x0) through a single AdaptivityEvaluator instance.
// All GridFunctions must use the same FE space.
Array<GridFunction *> adapt_lim_gf0; // Owned. Initial fields z_k0(x0).
Array<GridFunction *> adapt_lim_gf; // Owned. Remapped fields z_k(x).
Vector adapt_lim_init_nodes; // Owned. Initial mesh nodes (ldofs).
Array<Coefficient *> adapt_lim_coeff; // Not owned, one per field.
AdaptivityEvaluator *adapt_lim_eval; // Not owned. Used for all fields.
Array<real_t> adapt_lim_delta_max; // Per-field delta_max_k (>0).
// Surface fitting.
const Array<bool> *surf_fit_marker; // Not owned. Nodes to fit.
@@ -2141,13 +2144,13 @@ protected:
{
bool enabled;
int dim, ne, nq;
int nal = 0; // number of adaptive limiting fields
mutable DenseTensor Jtr;
mutable bool Jtr_needs_update;
mutable bool Jtr_debug_grad;
mutable Vector E, O, X0, XL, H, C0, LD, H0, MC, ALC,
ALF, ALFmF0, ALFG, ALFH;
ALF, ALFmF0, ALFG, ALFH, ALD;
mutable bool AL_grads_assembled;
real_t al_delta;
const DofToQuad *maps;
const DofToQuad *maps_lim = nullptr;
const DofToQuad *maps_nodes = nullptr;
@@ -2314,7 +2317,6 @@ public:
integ_order(-1), metric_coeff(NULL), metric_normal(1.0),
lim_nodes0(NULL), lim_coeff(NULL),
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
adapt_lim_gf0(NULL), adapt_lim_gf(NULL), adapt_lim_coeff(NULL),
adapt_lim_eval(NULL),
surf_fit_marker(NULL), surf_fit_coeff(NULL),
surf_fit_gf(NULL), surf_fit_eval(NULL),
@@ -2403,10 +2405,21 @@ public:
Smaller values activate the term faster. */
void EnableAdaptiveLimiting(const GridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field adaptive limiting with per-field delta_max values. All
/// GridFunctions must be on the same FiniteElementSpace.
void EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#ifdef MFEM_USE_MPI
/// Parallel support for adaptive limiting.
void EnableAdaptiveLimiting(const ParGridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field parallel adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#endif
/** @brief Fitting of certain DOFs to the zero level set of a function.
@@ -2632,10 +2645,20 @@ public:
/// Adds the adaptive limiting term to the first integrator.
void EnableAdaptiveLimiting(const GridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const GridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#ifdef MFEM_USE_MPI
/// Parallel support for adaptive limiting.
void EnableAdaptiveLimiting(const ParGridFunction &z0, Coefficient &coeff,
AdaptivityEvaluator &ae, real_t delta_max = 1.0);
/// Multi-field parallel adaptive limiting with per-field delta_max values.
void EnableAdaptiveLimiting(const Array<const ParGridFunction *> &z0,
const Array<Coefficient *> &coeff,
AdaptivityEvaluator &ae,
const Array<real_t> &delta_max);
#endif
+31 -14
View File
@@ -119,15 +119,14 @@ void TMOP_AssembleDiagPA_AdaptLim_2D(const real_t lim_normal,
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t weight = W(qx, qy) * detJtr;
const real_t coeff = const_coeff ? ALC(0, 0, 0) : ALC(qx, qy, e);
const real_t coeff = const_coeff ? ALC(0,0,0) : ALC(qx, qy, e);
const real_t factor = weight * coeff * normal_inv_delta_sq;
const real_t diff = alf_quad(qy, qx);
const real_t grad_v = ALF_grad(v, qx, qy, e);
const real_t hess_vv = ALF_hess(v, v, qx, qy, e);
const real_t hdiag = factor * (grad_v * grad_v + diff * hess_vv);
QD(qx, dy) += bb * hdiag;
QD(qx, dy) += bb * factor * (grad_v*grad_v + diff * hess_vv);
}
}
}
@@ -176,27 +175,45 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagAdaptLim2D);
void TMOP_Integrator::AssembleDiagonalPA_AdaptLim_2D(Vector &diagonal) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto *B = PA.maps->B.Read();
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 2, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiagAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const int ALFH_stride = 2 * 2 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 2, 2, q, q, NE);
TMOPAssembleDiagAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
}
}
} // namespace mfem
+32 -13
View File
@@ -183,15 +183,15 @@ void TMOP_AssembleDiagPA_AdaptLim_3D(const real_t lim_normal,
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t weight = W(qx, qy, qz) * detJtr;
const real_t coeff = const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t coeff =
const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t factor = weight * coeff * normal_inv_delta_sq;
const real_t diff = alf_quad(qz, qy, qx);
const real_t grad_v = ALF_grad(v, qx, qy, qz, e);
const real_t hess_vv = ALF_hess(v, v, qx, qy, qz, e);
const real_t hdiag = factor * (grad_v * grad_v + diff * hess_vv);
u += bb * hdiag;
u += bb * factor * (grad_v * grad_v + diff * hess_vv);
}
r0[dz][qy][qx] = u;
}
@@ -265,26 +265,45 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagAdaptLim3D);
void TMOP_Integrator::AssembleDiagonalPA_AdaptLim_3D(Vector &diagonal) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto *B = PA.maps->B.Read();
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 3, q, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiagAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 3 * nqp_el * NE;
const int ALFH_stride = 3 * 3 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 3, q, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 3, 3, q, q, q, NE);
TMOPAssembleDiagAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE,
J, W, B, ALF_grad, ALF_hess, ALFmF0, D, d, q);
}
}
} // namespace mfem
+22 -7
View File
@@ -113,7 +113,7 @@ void TMOP_AssembleGradPA_C0_2D(const real_t lim_normal,
});
}
// Assemble gradient and Hessian of ALF field at quadrature points for AdaptLim (2D)
// Assemble gradient and Hessian of ALF field at quad points for AdaptLim (2D).
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_AdaptLim_2D(const int NE,
const real_t *B_nodes,
@@ -185,7 +185,7 @@ void TMOP_AssembleGradPA_AdaptLim_2D(const int NE,
}
MFEM_SYNC_THREAD;
// Compute/interpolate gradient and Hessian one vector component at a time.
// Compute/interpolate gradient and Hessian, one component at a time.
for (int c = 0; c < 2; c++)
{
kernels::internal::s_regs2d_t<MD1> rgrad_nodes, ddalf_dx_n, ddalf_dy_n;
@@ -326,16 +326,31 @@ void TMOP_Integrator::AssembleGradPA_AdaptLim_2D(const Vector &x) const
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const auto *B_nodes = PA.maps_nodes->B.Read(),
*G_nodes = PA.maps_nodes->G.Read();
const auto *B = PA.maps->B.Read();
const auto X = Reshape(x.Read(), d, d, 2, NE);
const auto ALF = Reshape(PA.ALF.Read(), d, d, NE);
auto ALF_grad = Reshape(PA.ALFG.Write(), 2, q, q, NE);
auto ALF_hess = Reshape(PA.ALFH.Write(), 2, 2, q, q, NE);
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const int ALFH_stride = 2 * 2 * nqp_el * NE;
TMOPAssembleGradAdaptLim2D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
const real_t *ALF_all = PA.ALF.Read();
real_t *ALFG_all = PA.ALFG.Write();
real_t *ALFH_all = PA.ALFH.Write();
for (int c = 0; c < nal; c++)
{
const auto ALF = Reshape(ALF_all + c * ALF_stride, d, d, NE);
auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 2, 2, q, q, NE);
TMOPAssembleGradAdaptLim2D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
}
PA.AL_grads_assembled = true;
}
+25 -8
View File
@@ -164,7 +164,7 @@ void TMOP_Integrator::AssembleGradPA_C0_3D(const Vector &x) const
J, W, b, bld, XL, X, H0, exp_lim, d, q);
}
// Assemble gradient and Hessian of ALF field at quadrature points for AdaptLim (3D)
// Assemble gradient and Hessian of ALF field at quadr points for AdaptLim (3D)
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_AdaptLim_3D(const int NE,
const real_t *B_nodes,
@@ -202,7 +202,8 @@ void TMOP_AssembleGradPA_AdaptLim_3D(const int NE,
kernels::internal::Grad3d(D1D, D1D, smem.d, sB_nodes, sG_nodes, r_X, r_J);
// Compute the reference derivatives of ALF at DOF nodes.
kernels::internal::s_regs3d_t<MD1> alf_n, dalf_dxi_n, dalf_deta_n, dalf_dzeta_n;
kernels::internal::s_regs3d_t<MD1> alf_n;
kernels::internal::s_regs3d_t<MD1> dalf_dxi_n, dalf_deta_n, dalf_dzeta_n;
kernels::internal::LoadDofs3d(e, D1D, ALF, alf_n);
kernels::internal::Contract3d<false, MD1>(D1D, D1D, smem.d,
sG_nodes, sB_nodes, sB_nodes,
@@ -222,7 +223,8 @@ void TMOP_AssembleGradPA_AdaptLim_3D(const int NE,
// Compute/interpolate gradient and Hessian one vector component at a time.
for (int c = 0; c < 3; c++)
{
kernels::internal::s_regs3d_t<MD1> rgrad_nodes, dd_dxi_n, dd_deta_n, dd_dzeta_n;
kernels::internal::s_regs3d_t<MD1> rgrad_nodes;
kernels::internal::s_regs3d_t<MD1> dd_dxi_n, dd_deta_n, dd_dzeta_n;
// Precompute the inverse of the physical Jacobian.
kernels::internal::vd_regs3d_t<3, 3, MD1> Jpr_inv;
@@ -399,16 +401,31 @@ void TMOP_Integrator::AssembleGradPA_AdaptLim_3D(const Vector &x) const
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const auto *B_nodes = PA.maps_nodes->B.Read(),
*G_nodes = PA.maps_nodes->G.Read();
const auto *B = PA.maps->B.Read();
const auto X = Reshape(x.Read(), d, d, d, 3, NE);
const auto ALF = Reshape(PA.ALF.Read(), d, d, d, NE);
auto ALF_grad = Reshape(PA.ALFG.Write(), 3, q, q, q, NE);
auto ALF_hess = Reshape(PA.ALFH.Write(), 3, 3, q, q, q, NE);
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 3 * nqp_el * NE;
const int ALFH_stride = 3 * 3 * nqp_el * NE;
TMOPAssembleGradAdaptLim3D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
const real_t *ALF_all = PA.ALF.Read();
real_t *ALFG_all = PA.ALFG.Write();
real_t *ALFH_all = PA.ALFH.Write();
for (int c = 0; c < nal; c++)
{
const auto ALF = Reshape(ALF_all + c * ALF_stride, d, d, d, NE);
auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 3, q, q, q, NE);
auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 3, 3, q, q, q, NE);
TMOPAssembleGradAdaptLim3D::Run(d, q, NE, B_nodes, G_nodes, B, X, ALF,
ALF_grad, ALF_hess, d, q);
}
PA.AL_grads_assembled = true;
}
+29 -10
View File
@@ -182,27 +182,46 @@ void TMOP_Integrator::AddMultGradPA_AdaptLim_2D(const Vector &R,
Vector &C) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto *B = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto RR = Reshape(R.Read(), d, d, 2, NE);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 2, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 2, 2, q, q, NE);
auto Y = Reshape(C.ReadWrite(), d, d, 2, NE);
TMOPMultGradAdaptLim::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, B,
RR, ALF_grad, ALF_hess, ALFmF0, Y, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const int ALFH_stride = 2 * 2 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 2, 2, q, q, NE);
TMOPMultGradAdaptLim::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, B,
RR, ALF_grad, ALF_hess, ALFmF0, Y, d, q);
}
}
} // namespace mfem
+40 -16
View File
@@ -169,10 +169,12 @@ void TMOP_AddMultGradPA_AdaptLim_3D(const real_t lim_normal,
// Hessian action:
// H = factor * (grad x grad + (gf - gf0) * hess)
const real_t coeff = const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t coeff =
const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t factor = weight * coeff * normal_inv_delta_sq;
const real_t grad_dot_R =
grad_alf[0] * R_q[0] + grad_alf[1] * R_q[1] + grad_alf[2] * R_q[2];
const real_t grad_dot_R = grad_alf[0] * R_q[0] +
grad_alf[1] * R_q[1] +
grad_alf[2] * R_q[2];
real_t hess_R[3];
hess_R[0] =
ALF_hess(0, 0, qx, qy, qz, e) * R_q[0] +
@@ -187,9 +189,12 @@ void TMOP_AddMultGradPA_AdaptLim_3D(const real_t lim_normal,
ALF_hess(2, 1, qx, qy, qz, e) * R_q[1] +
ALF_hess(2, 2, qx, qy, qz, e) * R_q[2];
r00(0, qz, qy, qx) = factor * (grad_alf[0] * grad_dot_R + diff * hess_R[0]);
r00(1, qz, qy, qx) = factor * (grad_alf[1] * grad_dot_R + diff * hess_R[1]);
r00(2, qz, qy, qx) = factor * (grad_alf[2] * grad_dot_R + diff * hess_R[2]);
r00(0, qz, qy, qx) = factor * (grad_alf[0] * grad_dot_R +
diff * hess_R[0]);
r00(1, qz, qy, qx) = factor * (grad_alf[1] * grad_dot_R +
diff * hess_R[1]);
r00(2, qz, qy, qx) = factor * (grad_alf[2] * grad_dot_R +
diff * hess_R[2]);
}
}
}
@@ -206,27 +211,46 @@ void TMOP_Integrator::AddMultGradPA_AdaptLim_3D(const Vector &R,
Vector &C) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto *B = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto RR = Reshape(R.Read(), d, d, d, 3, NE);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 3, q, q, q, NE);
const auto ALF_hess = Reshape(PA.ALFH.Read(), 3, 3, q, q, q, NE);
auto Y = Reshape(C.ReadWrite(), d, d, d, 3, NE);
TMOPMultGradAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, B,
RR, ALF_grad, ALF_hess, ALFmF0, Y, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 3 * nqp_el * NE;
const int ALFH_stride = 3 * 3 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
const real_t *ALFH_all = PA.ALFH.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 3, q, q, q, NE);
const auto ALF_hess = Reshape(ALFH_all + c * ALFH_stride, 3, 3, q, q, q, NE);
TMOPMultGradAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J,
W, B, RR, ALF_grad, ALF_hess, ALFmF0, Y, d, q);
}
}
} // namespace mfem
+25 -9
View File
@@ -205,25 +205,41 @@ void TMOP_Integrator::AddMultPA_AdaptLim_2D([[maybe_unused]] const Vector &x,
Vector &y) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto *B = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 2, q, q, NE);
auto Y = Reshape(y.ReadWrite(), d, d, 2, NE);
TMOPMultAdaptLim::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W,
B, ALF_grad, ALFmF0, Y, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 2 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 2, q, q, NE);
TMOPMultAdaptLim::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W,
B, ALF_grad, ALFmF0, Y, d, q);
}
}
} // namespace mfem
+28 -12
View File
@@ -76,7 +76,7 @@ void TMOP_AddMultPA_C0_3D(const real_t lim_normal,
r11(1, qz, qy, qx),
r11(2, qz, qy, qx)
};
const real_t coeff0 = const_c0 ? C0(0, 0, 0, 0) : C0(qx, qy, qz, e);
const real_t coeff0 = const_c0 ? C0(0,0,0,0) : C0(qx, qy, qz, e);
real_t d1[3];
// Eval_d1 (Quadratic Limiter)
@@ -193,7 +193,8 @@ void TMOP_AddMultPA_AdaptLim_3D(const real_t lim_normal,
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t weight = W(qx, qy, qz) * detJtr;
const real_t coeff = const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t coeff =
const_coeff ? ALC(0, 0, 0, 0) : ALC(qx, qy, qz, e);
const real_t factor = weight * coeff * normal_inv_delta_sq *
alf_quad(qz, qy, qx);
@@ -217,26 +218,41 @@ void TMOP_Integrator::AddMultPA_AdaptLim_3D([[maybe_unused]] const Vector &x,
Vector &y) const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto *B = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, d, NE);
const auto ALF_grad = Reshape(PA.ALFG.Read(), 3, q, q, q, NE);
auto Y = Reshape(y.ReadWrite(), d, d, d, 3, NE);
TMOPMultAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W,
B, ALF_grad, ALFmF0, Y, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
const real_t *ALD = PA.ALD.HostRead();
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const int ALFG_stride = 3 * nqp_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
const real_t *ALFG_all = PA.ALFG.Read();
for (int c = 0; c < nal; c++)
{
const real_t delta_max = ALD[c];
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, d, NE);
const auto ALF_grad = Reshape(ALFG_all + c * ALFG_stride, 3, q, q, q, NE);
TMOPMultAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W,
B, ALF_grad, ALFmF0, Y, d, q);
}
}
} // namespace mfem
+99 -41
View File
@@ -46,14 +46,14 @@ void TMOP_Integrator::AssembleGradPA(const Vector &de,
{
AssembleGradPA_2D(xe);
if (lim_coeff) { AssembleGradPA_C0_2D(xe); }
if (adapt_lim_gf) { AssembleGradPA_AdaptLim_2D(xe); }
if (adapt_lim_gf.Size() > 0) { AssembleGradPA_AdaptLim_2D(xe); }
}
if (PA.dim == 3)
{
AssembleGradPA_3D(xe);
if (lim_coeff) { AssembleGradPA_C0_3D(xe); }
if (adapt_lim_gf) { AssembleGradPA_AdaptLim_3D(xe); }
if (adapt_lim_gf.Size() > 0) { AssembleGradPA_AdaptLim_3D(xe); }
}
}
@@ -201,12 +201,15 @@ void TMOP_Integrator::UpdateCoefficientsPA(const Vector &d_loc)
// All are constant or not specified.
if (PA.MC.Size() == 1 && PA.C0.Size() <= 1 && PA.ALC.Size() <= 1) { return; }
const int nal = PA.nal;
const bool alc_is_qvec =
(nal > 0) ? (PA.ALC.Size() == nal * PA.nq * PA.ne) : false;
if (PA.MC.Size() == 1 && PA.C0.Size() <= 1 && !alc_is_qvec) { return; }
// Coefficients are always evaluated on the CPU for now.
PA.MC.HostWrite();
PA.C0.HostWrite();
PA.ALC.HostWrite();
if (alc_is_qvec) { PA.ALC.HostWrite(); }
const IntegrationRule &ir = *PA.ir;
auto T = new IsoparametricTransformation;
@@ -231,11 +234,17 @@ void TMOP_Integrator::UpdateCoefficientsPA(const Vector &d_loc)
}
}
if (PA.ALC.Size() > 1)
if (alc_is_qvec)
{
for (int q = 0; q < PA.nq; ++q)
MFEM_VERIFY(nal == adapt_lim_coeff.Size(), "internal error");
for (int c = 0; c < nal; c++)
{
PA.ALC(q + e * PA.nq) = adapt_lim_coeff->Eval(*T, ir.IntPoint(q));
real_t *ALC_c = PA.ALC.HostWrite() + c * PA.nq * PA.ne;
for (int q = 0; q < PA.nq; ++q)
{
ALC_c[q + e * PA.nq] =
adapt_lim_coeff[c]->Eval(*T, ir.IntPoint(q));
}
}
}
}
@@ -336,37 +345,67 @@ void TMOP_Integrator::AssemblePA(const FiniteElementSpace &fes)
if (lim_coeff) { AssemblePA_Limiting(); }
// Adaptive limiting: adapt_lim_coeff -> PA.ALC, adapt_lim_gf -> PA.ALF,
// adapt_lim_gf0 -> PA.ALF0, adapt_lim_delta_max -> PA.ALD
if (adapt_lim_gf) { AssemblePA_AdaptLim(); }
if (adapt_lim_gf.Size() > 0) { AssemblePA_AdaptLim(); }
}
void TMOP_Integrator::AssemblePA_AdaptLim()
{
const FiniteElementSpace *alfes = adapt_lim_gf->FESpace();
const int nal = adapt_lim_coeff.Size();
MFEM_VERIFY(nal > 0, "internal error");
MFEM_VERIFY(adapt_lim_gf.Size() == nal && adapt_lim_gf0.Size() == nal,
"internal error");
const FiniteElementSpace *alfes = adapt_lim_gf[0]->FESpace();
MFEM_VERIFY(alfes && alfes->GetVDim() == 1, "internal error");
MFEM_VERIFY(strcmp(alfes->FEColl()->Name(), PA.fes->FEColl()->Name()) == 0 &&
alfes->FEColl()->GetOrder() == PA.fes->FEColl()->GetOrder(),
"The PA code assumes the same FE spaces for mesh and limiting.");
PA.AL_grads_assembled = false;
PA.nal = nal;
// adapt_lim_coeff -> PA.ALC (Q-vector).
PA.ALC.UseDevice(true);
if (auto *cQ = dynamic_cast<ConstantCoefficient *>(adapt_lim_coeff))
// adapt_lim_coeff -> PA.ALC
// Keep the ConstantCoefficient fast-path: when all coefficients are
// constant, store one scalar per adaptive-limiting term.
bool all_const = true;
for (int c = 0; c < nal; c++)
{
PA.ALC.SetSize(1, Device::GetMemoryType());
PA.ALC.HostWrite();
PA.ALC(0) = cQ->constant;
if (!dynamic_cast<ConstantCoefficient *>(adapt_lim_coeff[c]))
{
all_const = false;
break;
}
}
PA.ALC.UseDevice(true);
if (all_const)
{
PA.ALC.SetSize(nal, Device::GetMemoryType());
real_t *ALC_all = PA.ALC.HostWrite();
for (int c = 0; c < nal; c++)
{
auto *cc = dynamic_cast<ConstantCoefficient *>(adapt_lim_coeff[c]);
MFEM_VERIFY(cc, "internal error");
ALC_all[c] = cc->constant;
}
}
else
{
PA.ALC.SetSize(PA.nq * PA.ne, Device::GetMemoryType());
auto ALC = Reshape(PA.ALC.HostWrite(), PA.nq, PA.ne);
for (int e = 0; e < PA.ne; ++e)
// If one Coefficient is not constant, we allocate the full size for
// all Coefficients. Could be optimized in the future.
PA.ALC.SetSize(nal * PA.nq * PA.ne, Device::GetMemoryType());
real_t *ALC_all = PA.ALC.HostWrite();
for (int c = 0; c < nal; c++)
{
ElementTransformation &T = *PA.fes->GetElementTransformation(e);
for (int q = 0; q < PA.ir->GetNPoints(); ++q)
real_t *ALC_c = ALC_all + c * PA.nq * PA.ne;
for (int e = 0; e < PA.ne; ++e)
{
ALC(q, e) = adapt_lim_coeff->Eval(T, PA.ir->IntPoint(q));
ElementTransformation &T = *PA.fes->GetElementTransformation(e);
for (int q = 0; q < PA.ir->GetNPoints(); ++q)
{
ALC_c[q + e * PA.nq] =
adapt_lim_coeff[c]->Eval(T, PA.ir->IntPoint(q));
}
}
}
}
@@ -396,29 +435,46 @@ void TMOP_Integrator::AssemblePA_AdaptLim()
PA.maps_nodes = &fe_n->GetDofToQuad(lex_nodes, DofToQuad::TENSOR);
}
// adapt_lim_gf -> PA.ALF (E-vector, same pattern as LD).
const FiniteElement &fe = *alfes->GetTypicalFE();
PA.ALF.SetSize(PA.ne * fe.GetDof(), Device::GetMemoryType());
PA.ALF.UseDevice(true);
// Restrict each adaptive limiting field into separate contiguous E-vectors
// (one block per adaptive limiting term).
const Operator *alf_R = alfes->GetElementRestriction(ordering);
alf_R->Mult(*adapt_lim_gf, PA.ALF);
// adapt_lim_gf - adapt_lim_gf0 -> PA.ALFmF0
PA.ALFmF0.SetSize(PA.ne * fe.GetDof(), Device::GetMemoryType());
const int ndofs = alfes->GetVSize();
const int Esize = alf_R->Height();
PA.ALF.SetSize(nal * Esize, Device::GetMemoryType());
PA.ALF.UseDevice(true);
PA.ALFmF0.SetSize(nal * Esize, Device::GetMemoryType());
PA.ALFmF0.UseDevice(true);
alf_R->Mult(*adapt_lim_gf0, PA.ALFmF0);
Vector ALFc, ALF0c;
for (int c = 0; c < nal; c++)
{
ALFc.MakeRef(PA.ALF, c * Esize, Esize);
ALF0c.MakeRef(PA.ALFmF0, c * Esize, Esize);
MFEM_VERIFY(adapt_lim_gf[c]->Size() == ndofs, "internal error");
MFEM_VERIFY(adapt_lim_gf0[c]->Size() == ndofs, "internal error");
alf_R->Mult(*adapt_lim_gf[c], ALFc);
alf_R->Mult(*adapt_lim_gf0[c], ALF0c);
}
// Build differences in-place: ALFmF0 = ALF - ALF0.
PA.ALFmF0 *= -1.0;
PA.ALFmF0 += PA.ALF;
// adapt_lim_delta_max -> PA.al_delta.
PA.al_delta = adapt_lim_delta_max;
// Per-field delta_max values.
MFEM_VERIFY(adapt_lim_delta_max.Size() == nal, "internal error");
PA.ALD.SetSize(nal);
PA.ALD.HostWrite();
for (int c = 0; c < nal; c++) { PA.ALD(c) = adapt_lim_delta_max[c]; }
// Allocate storage for gradient and Hessian of ALF at quadrature points
// These will be filled during AssembleGradPA
const int dim = PA.dim;
PA.ALFG.UseDevice(true);
PA.ALFG.SetSize(dim * PA.nq * PA.ne, Device::GetMemoryType());
PA.ALFG.SetSize(nal * dim * PA.nq * PA.ne, Device::GetMemoryType());
PA.ALFH.UseDevice(true);
PA.ALFH.SetSize(dim * dim * PA.nq * PA.ne, Device::GetMemoryType());
PA.ALFH.SetSize(nal * dim * dim * PA.nq * PA.ne, Device::GetMemoryType());
}
@@ -439,14 +495,14 @@ void TMOP_Integrator::AssembleGradDiagonalPA(Vector &de) const
{
AssembleDiagonalPA_2D(de);
if (lim_coeff) { AssembleDiagonalPA_C0_2D(de); }
if (adapt_lim_gf) { AssembleDiagonalPA_AdaptLim_2D(de); }
if (adapt_lim_gf.Size() > 0) { AssembleDiagonalPA_AdaptLim_2D(de); }
}
if (PA.dim == 3)
{
AssembleDiagonalPA_3D(de);
if (lim_coeff) { AssembleDiagonalPA_C0_3D(de); }
if (adapt_lim_gf) { AssembleDiagonalPA_AdaptLim_3D(de); }
if (adapt_lim_gf.Size() > 0) { AssembleDiagonalPA_AdaptLim_3D(de); }
}
}
@@ -473,7 +529,7 @@ void TMOP_Integrator::AddMultPA(const Vector &de, Vector &ye) const
{
AddMultPA_2D(xe, ye);
if (lim_coeff) { AddMultPA_C0_2D(xe, ye); }
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
// AddMultPA_AdaptLim_2D uses the precomputed AdaptLim field gradient
// at quadrature points (PA.ALFG). Ensure it is up-to-date for the
@@ -488,7 +544,7 @@ void TMOP_Integrator::AddMultPA(const Vector &de, Vector &ye) const
{
AddMultPA_3D(xe, ye);
if (lim_coeff) { AddMultPA_C0_3D(xe, ye); }
if (adapt_lim_gf)
if (adapt_lim_gf.Size() > 0)
{
AssembleGradPA_AdaptLim_3D(xe);
AddMultPA_AdaptLim_3D(xe, ye);
@@ -513,14 +569,14 @@ void TMOP_Integrator::AddMultGradPA(const Vector &re, Vector &ce) const
{
AddMultGradPA_2D(re, ce);
if (lim_coeff) { AddMultGradPA_C0_2D(re, ce); }
if (adapt_lim_gf) { AddMultGradPA_AdaptLim_2D(re, ce); }
if (adapt_lim_gf.Size() > 0) { AddMultGradPA_AdaptLim_2D(re, ce); }
}
if (PA.dim == 3)
{
AddMultGradPA_3D(re, ce);
if (lim_coeff) { AddMultGradPA_C0_3D(re, ce); }
if (adapt_lim_gf) { AddMultGradPA_AdaptLim_3D(re, ce); }
if (adapt_lim_gf.Size() > 0) { AddMultGradPA_AdaptLim_3D(re, ce); }
}
}
@@ -549,14 +605,16 @@ real_t TMOP_Integrator::GetLocalStateEnergyPA(const Vector &de) const
{
GetLocalStateEnergyPA_2D(xe, energy);
if (lim_coeff) { energy += GetLocalStateEnergyPA_C0_2D(xe); }
if (adapt_lim_gf) { energy += GetLocalStateEnergyPA_AdaptLim_2D(); }
if (adapt_lim_gf.Size() > 0)
{ energy += GetLocalStateEnergyPA_AdaptLim_2D(); }
}
if (PA.dim == 3)
{
GetLocalStateEnergyPA_3D(xe, energy);
if (lim_coeff) { energy += GetLocalStateEnergyPA_C0_3D(xe); }
if (adapt_lim_gf) { energy += GetLocalStateEnergyPA_AdaptLim_3D(); }
if (adapt_lim_gf.Size() > 0)
{ energy += GetLocalStateEnergyPA_AdaptLim_3D(); }
}
return energy;
+26 -9
View File
@@ -182,26 +182,43 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPEnergyAdaptLim2D);
real_t TMOP_Integrator::GetLocalStateEnergyPA_AdaptLim_2D() const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto *b = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, NE);
auto E = Reshape(PA.E.Write(), q, q, NE);
TMOPEnergyAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, b,
ALFmF0, E, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
MFEM_VERIFY(PA.ALD.Size() == nal, "internal error");
PA.ALD.HostRead();
return PA.E * PA.O;
const int ndof_el = d * d;
const int nqp_el = q * q;
const int ALF_stride = ndof_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
real_t energy = 0.0;
for (int c = 0; c < nal; c++)
{
const real_t delta_max = PA.ALD(c);
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, NE);
TMOPEnergyAdaptLim2D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, b,
ALFmF0, E, d, q);
energy += PA.E * PA.O;
}
return energy;
}
} // namespace mfem
+26 -9
View File
@@ -198,26 +198,43 @@ MFEM_TMOP_MDQ_SPECIALIZE(TMOPEnergyAdaptLim3D);
real_t TMOP_Integrator::GetLocalStateEnergyPA_AdaptLim_3D() const
{
const real_t ln = lim_normal;
const real_t delta_max = PA.al_delta;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const bool const_coeff = PA.ALC.Size() == 1;
const auto ALC = const_coeff
? Reshape(PA.ALC.Read(), 1, 1, 1, 1)
: Reshape(PA.ALC.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto *b = PA.maps->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto ALFmF0 = Reshape(PA.ALFmF0.Read(), d, d, d, NE);
auto E = Reshape(PA.E.Write(), q, q, q, NE);
TMOPEnergyAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, b,
ALFmF0, E, d, q);
const int nal = PA.nal;
MFEM_VERIFY(nal > 0, "internal error");
MFEM_VERIFY(PA.ALD.Size() == nal, "internal error");
PA.ALD.HostRead();
return PA.E * PA.O;
const int ndof_el = d * d * d;
const int nqp_el = q * q * q;
const int ALF_stride = ndof_el * NE;
const bool const_coeff = (PA.ALC.Size() == nal);
const int ALC_stride = const_coeff ? 1 : (nqp_el * NE);
const real_t *ALC_all = PA.ALC.Read();
const real_t *ALFmF0_all = PA.ALFmF0.Read();
real_t energy = 0.0;
for (int c = 0; c < nal; c++)
{
const real_t delta_max = PA.ALD(c);
const auto ALC = const_coeff
? Reshape(ALC_all + c, 1, 1, 1, 1)
: Reshape(ALC_all + c * ALC_stride, q, q, q, NE);
const auto ALFmF0 = Reshape(ALFmF0_all + c * ALF_stride, d, d, d, NE);
TMOPEnergyAdaptLim3D::Run(d, q, ln, delta_max, const_coeff, ALC, NE, J, W, b,
ALFmF0, E, d, q);
energy += PA.E * PA.O;
}
return energy;
}
} // namespace mfem
+2
View File
@@ -759,6 +759,7 @@ void TMOPHRSolver::Update()
gridfuncarr[i]->SetTrueVector();
gridfuncarr[i]->SetFromTrueVector();
}
tmopns->UpdateDeterminantBoundGridFunction();
// Update Discrete Indicator for all the TMOP_Integrators in NonLinearForm
Array<NonlinearFormIntegrator*> &integs = *(nlf->GetDNFI());
@@ -806,6 +807,7 @@ void TMOPHRSolver::ParUpdate()
pgridfuncarr[i]->SetTrueVector();
pgridfuncarr[i]->SetFromTrueVector();
}
tmopns->UpdateDeterminantBoundGridFunction();
// Update Discrete Indicator
Array<NonlinearFormIntegrator*> &integs = *(nlf->GetDNFI());
+105 -7
View File
@@ -68,6 +68,9 @@ void AdvectorCG::ComputeAtNewPosition(const Vector &new_mesh_nodes,
}
}
// Without this, the next remap would start from the initial mesh, i.e.,
// every consecutive remap would be more expensive, as it would have to
// transport the solution through bigger displacements.
field0 = new_field;
nodes0 = new_mesh_nodes;
}
@@ -305,8 +308,14 @@ void ParAdvectorCGOper::Mult(const Vector &ind, Vector &di_dt) const
M.BilinearForm::operator=(0.0);
M.Assemble();
HypreParVector *RHS = rhs.ParallelAssemble();
HypreParVector X(K.ParFESpace());
Vector RHS;
RHS.SetSize(M.ParFESpace()->GetTrueVSize(), ind);
RHS.UseDevice(ind.UseDevice());
rhs.ParallelAssemble(RHS);
Vector X;
X.SetSize(M.ParFESpace()->GetTrueVSize(), ind);
X.UseDevice(ind.UseDevice());
X = 0.0;
OperatorHandle Mop;
@@ -335,10 +344,8 @@ void ParAdvectorCGOper::Mult(const Vector &ind, Vector &di_dt) const
lin_solver.SetRelTol(rtol); lin_solver.SetAbsTol(0.0);
lin_solver.SetMaxIter(100);
lin_solver.SetPrintLevel(0);
lin_solver.Mult(*RHS, X);
lin_solver.Mult(RHS, X);
K.ParFESpace()->GetProlongationMatrix()->Mult(X, di_dt);
delete RHS;
delete prec;
}
#endif
@@ -493,7 +500,10 @@ real_t TMOPNewtonSolver::ComputeScalingFactor(const Vector &d_in,
// Check if the starting mesh (given by x) is inverted. Note that x hasn't
// been modified by the Newton update yet.
const real_t min_detT_in = ComputeMinDet(d_loc, *fes);
const real_t min_detT_in =
detJpr_pos_bound ? ComputeDetJptLowerBound(d_loc, *fes)
/* */ : ComputeMinDet(d_loc, *fes);
const bool untangling = (min_detT_in <= 0.0) ? true : false;
const real_t untangle_factor = 1.5;
if (untangling)
@@ -544,7 +554,10 @@ real_t TMOPNewtonSolver::ComputeScalingFactor(const Vector &d_in,
#endif
// Check the changes in detJ.
min_detT_out = ComputeMinDet(d_loc, *fes);
min_detT_out =
detJpr_pos_bound ? ComputeDetJptLowerBound(d_loc, *fes)
/* */ : ComputeMinDet(d_loc, *fes);
if (untangling == false && min_detT_out <= min_detJ_limit)
{
// No untangling, and detJ got negative (or small) -- no good.
@@ -969,6 +982,37 @@ void TMOPNewtonSolver::ProcessNewState(const Vector &dx) const
}
}
void TMOPNewtonSolver::EnsurePositiveDeterminantBound(
Mesh &mesh, int ref_factor, int max_recursion_depth)
{
#ifdef MFEM_USE_MPI
if (ParMesh *pmesh = dynamic_cast<ParMesh *>(&mesh))
{
det_gf = pmesh->GetJacobianDeterminantGF();
}
else
#endif
{
det_gf = mesh.GetJacobianDeterminantGF();
}
// setup the PLBound object for estimating the minima.
// note: this must be updated if the mesh is p-refined.
int max_order = det_gf->FESpace()->GetMaxElementOrder();
det_plb = std::make_unique<PLBound>(det_gf->FESpace(),
ref_factor*(max_order+1));
plb_rec_depth = max_recursion_depth;
detJpr_pos_bound = true;
}
void TMOPNewtonSolver::UpdateDeterminantBoundGridFunction()
{
if (!det_gf) { return; }
det_gf->FESpace()->Update();
det_gf->Update();
}
real_t TMOPNewtonSolver::ComputeMinDet(const Vector &d_loc,
const FiniteElementSpace &fes) const
{
@@ -1028,6 +1072,60 @@ real_t TMOPNewtonSolver::ComputeMinDet(const Vector &d_loc,
return min_detJ;
}
real_t TMOPNewtonSolver::ComputeDetJptLowerBound(const Vector &d_loc,
const FiniteElementSpace &fes) const
{
MFEM_VERIFY(det_gf != nullptr && det_plb != nullptr,
"Determinant bounding has not been setup.");
FiniteElementSpace *det_fes = det_gf->FESpace();
MFEM_VERIFY(!det_fes->IsVariableOrder() && UsesTensorBasis(*det_fes),
"Determinant lower bounds require a fixed-order tensor-product "
"determinant space.");
Array<int> dofs, xdofs;
DenseMatrix dshape, Jpr, pos;
Vector d_loc_el, detvals;
for (int e = 0; e < fes.GetNE(); e++)
{
const FiniteElement *fe = fes.GetFE(e);
const int dof = fe->GetDof(), dim = fe->GetDim();
dshape.SetSize(dof, dim);
Jpr.SetSize(dim);
pos.SetSize(dof, dim);
Vector posV(pos.Data(), dof * dim);
x_0.GetElementDofValues(e, posV);
if (periodic)
{
auto n_el = dynamic_cast<const NodalFiniteElement *>(fe);
n_el->ReorderLexToNative(dim, posV);
}
fes.GetElementVDofs(e, xdofs);
d_loc.GetSubVector(xdofs, d_loc_el);
posV += d_loc_el;
const IntegrationRule &irule = det_fes->GetFE(e)->GetNodes();
const int nsp = irule.GetNPoints();
detvals.SetSize(nsp);
det_fes->GetElementDofs(e, dofs);
for (int q = 0; q < nsp; q++)
{
fe->CalcDShape(irule.IntPoint(q), dshape);
MultAtB(pos, dshape, Jpr);
detvals(q) = Jpr.Det();
}
det_gf->SetSubVector(dofs, detvals);
}
auto minbounds = det_gf->EstimateFunctionMinimum(0, *det_plb,
plb_rec_depth, 1e-5);
const DenseMatrix &Wideal =
Geometries.GetGeomToPerfGeomJac(fes.GetMesh()->GetTypicalElementGeometry());
return minbounds.first/Wideal.Det();
}
#ifdef MFEM_USE_MPI
// Metric values are visualized by creating an L2 finite element functions and
// computing the metric values at the nodes.
+32
View File
@@ -204,6 +204,11 @@ protected:
// These fields are relevant for mixed meshes.
IntegrationRules *IntegRules;
int integ_order;
// Determinant lower-bound data used by the line search.
bool detJpr_pos_bound = false;
std::unique_ptr<GridFunction> det_gf;
std::unique_ptr<PLBound> det_plb;
int plb_rec_depth = 0;
MemoryType temp_mt = MemoryType::DEFAULT;
@@ -216,9 +221,16 @@ protected:
return ir;
}
/// Compute the minimum det(Jpt) of the trial mesh at quadrature points
/// (computes det(Jpr) and scales by the det of ideal target element).
real_t ComputeMinDet(const Vector &d_loc,
const FiniteElementSpace &fes) const;
/// Compute a lower bound for det(Jpt) of the trial mesh,
/// (computes det(Jpr) and scales by the det of ideal target element).
real_t ComputeDetJptLowerBound(const Vector &d_loc,
const FiniteElementSpace &fes) const;
real_t MinDetJpr_2D(const FiniteElementSpace *, const Vector &) const;
real_t MinDetJpr_3D(const FiniteElementSpace *, const Vector &) const;
@@ -261,6 +273,26 @@ public:
void SetMinDetPtr(real_t *md_ptr) { min_det_ptr = md_ptr; }
/** @brief Ensure a positive lower bound for the Jacobian determinant in
tensor-product elements during line-search.
@note The solver creates and updates its own determinant GridFunction
from @a mesh while testing trial mesh positions. When @a mesh is a
ParMesh, the internal determinant field is a ParGridFunction. The
@a ref_factor controls the number of control points used by the PLBound
object, and @a max_recursion_depth controls the depth used by the
minimum-value estimator.
The determinant is represented by a high-order GridFunction computed
at the mesh nodes. The order is chosen s.t. interpolating the det at
some quad point would be equivalent to computing the det directly at the
same quad point using the mesh positions.
*/
void EnsurePositiveDeterminantBound(Mesh &mesh, int ref_factor,
int max_recursion_depth = 0);
/// Update internal determinant GridFunction after a mesh topology change.
void UpdateDeterminantBoundGridFunction();
/// Set the memory type for temporary memory allocations.
void SetTempMemoryType(MemoryType mt) { temp_mt = mt; }
+501 -95
View File
@@ -1030,12 +1030,42 @@ void L2ProjectionGridTransfer::L2ProjectionL2Space::EAProlongateTranspose(
BatchedLinAlg::MultTranspose(P_dt, x, y);
}
L2ProjectionGridTransfer::L2ProjectionH1Space::H1ConsistentMassOperator::
H1ConsistentMassOperator(const Operator &M_LH_, const Solver &M_L_solver_)
: Operator(M_LH_.Height(), M_LH_.Width()),
M_LH(M_LH_),
M_L_solver(M_L_solver_)
{
MFEM_VERIFY(M_LH.Height() == M_L_solver.Height() &&
M_LH.Height() == M_L_solver.Width(),
"incompatible consistent mass operator dimensions");
}
void L2ProjectionGridTransfer::L2ProjectionH1Space::H1ConsistentMassOperator::
Mult(const Vector &x, Vector &y) const
{
Vector tmp(M_LH.Height());
M_LH.Mult(x, tmp);
M_L_solver.Mult(tmp, y);
}
void L2ProjectionGridTransfer::L2ProjectionH1Space::H1ConsistentMassOperator::
MultTranspose(const Vector &x, Vector &y) const
{
Vector tmp(M_LH.Height());
M_L_solver.Mult(x, tmp);
M_LH.MultTranspose(tmp, y);
}
L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
const FiniteElementSpace& fes_ho_, const FiniteElementSpace& fes_lor_,
const bool use_ea_, MemoryType d_mt_)
const bool use_ea_, const bool use_consistent_mass_, MemoryType d_mt_)
: L2Projection(fes_ho_, fes_lor_, d_mt_),
use_ea(use_ea_)
use_ea(use_ea_),
use_consistent_mass(use_consistent_mass_)
{
MFEM_VERIFY(!(use_ea && use_consistent_mass),
"consistent mass is not supported with element assembly");
// need scalar to keep dimensions matching (operators are built to apply
// individually on each vdim)
@@ -1053,7 +1083,7 @@ L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
std::unique_ptr<SparseMatrix> R_mat, M_LH_mat;
std::tie(R_mat, M_LH_mat) = ComputeSparseRAndM_LH();
std::tie(R_mat, M_LH_mat) = ComputeSparseRAndM_LH(!use_consistent_mass);
const SparseMatrix *P_ho = fes_ho_scalar->GetConformingProlongation();
const SparseMatrix *P_lor = fes_lor_scalar->GetConformingProlongation();
@@ -1062,40 +1092,71 @@ L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
{
if (P_ho && P_lor)
{
R_mat.reset(RAP(*P_lor, *R_mat, *P_ho));
if (R_mat) { R_mat.reset(RAP(*P_lor, *R_mat, *P_ho)); }
M_LH_mat.reset(RAP(*P_lor, *M_LH_mat, *P_ho));
}
else if (P_ho)
{
R_mat.reset(mfem::Mult(*R_mat, *P_ho));
if (R_mat) { R_mat.reset(mfem::Mult(*R_mat, *P_ho)); }
M_LH_mat.reset(mfem::Mult(*M_LH_mat, *P_ho));
}
else // P_lor != nullptr
{
R_mat.reset(mfem::Mult(*P_lor, *R_mat));
if (R_mat) { R_mat.reset(mfem::Mult(*P_lor, *R_mat)); }
M_LH_mat.reset(mfem::Mult(*P_lor, *M_LH_mat));
}
}
SparseMatrix *RTxM_LH_mat = TransposeMult(*R_mat, *M_LH_mat);
precon.reset(new DSmoother(*RTxM_LH_mat));
if (use_consistent_mass)
{
BilinearForm M_lor(fes_lor_scalar.get());
M_lor.AddDomainIntegrator(new MassIntegrator);
M_lor.Assemble();
M_lor.Finalize();
SparseMatrix *M_L_mat = M_lor.LoseMat();
// Set ownership
RTxM_LH.reset(RTxM_LH_mat);
R = std::move(R_mat);
M_LH = std::move(M_LH_mat);
ML_precon.reset(new DSmoother(*M_L_mat));
ML_pcg.SetPrintLevel(0);
ML_pcg.SetMaxIter(1000);
ML_pcg.SetRelTol(1e-13);
ML_pcg.SetAbsTol(1e-13);
ML_pcg.SetPreconditioner(*ML_precon);
ML_pcg.SetOperator(*M_L_mat);
// Start each solve from zero so repeated Operator::Mult() calls do not
// depend on the output vector contents supplied by the caller.
ML_pcg.iterative_mode = false;
SetupPCG();
M_L.reset(M_L_mat);
M_LH = std::move(M_LH_mat);
R.reset(new H1ConsistentMassOperator(*M_LH, ML_pcg));
}
else
{
SparseMatrix *RTxM_LH_mat = TransposeMult(*R_mat, *M_LH_mat);
precon.reset(new DSmoother(*RTxM_LH_mat));
// Set ownership
RTxM_LH.reset(RTxM_LH_mat);
R = std::move(R_mat);
M_LH = std::move(M_LH_mat);
SetupPCG();
}
}
#ifdef MFEM_USE_MPI
L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
const ParFiniteElementSpace& pfes_ho, const ParFiniteElementSpace& pfes_lor,
const bool use_ea_, MemoryType d_mt_)
const bool use_ea_, const bool use_consistent_mass_, MemoryType d_mt_)
: L2Projection(pfes_ho, pfes_lor, d_mt_),
use_ea(use_ea_), pcg(pfes_ho.GetComm())
use_ea(use_ea_),
use_consistent_mass(use_consistent_mass_),
ML_pcg(pfes_ho.GetComm()),
pcg(pfes_ho.GetComm())
{
MFEM_VERIFY(!(use_ea && use_consistent_mass),
"consistent mass is not supported with element assembly");
// need scalar to keep dimensions matching (operators are built to apply
// individually on each vdim)
@@ -1111,8 +1172,42 @@ L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
return;
}
std::tie(R, M_LH) = ComputeSparseRAndM_LH();
std::tie(R, M_LH) = ComputeSparseRAndM_LH(!use_consistent_mass);
HypreParMatrix M_LH_local = HypreParMatrix(pfes_ho.GetComm(),
pfes_lor_scalar->GlobalVSize(),
pfes_ho_scalar->GlobalVSize(),
pfes_lor_scalar->GetDofOffsets(),
pfes_ho_scalar->GetDofOffsets(),
static_cast<SparseMatrix*>(M_LH.get()));
HypreParMatrix *M_LH_mat = RAP(pfes_lor_scalar->Dof_TrueDof_Matrix(),
&M_LH_local, pfes_ho_scalar->Dof_TrueDof_Matrix());
if (use_consistent_mass)
{
ParBilinearForm M_lor(pfes_lor_scalar.get());
M_lor.AddDomainIntegrator(new MassIntegrator);
M_lor.Assemble();
M_lor.Finalize();
HypreParMatrix *M_L_mat = M_lor.ParallelAssemble();
M_L.reset(M_L_mat);
M_LH.reset(M_LH_mat);
HypreDiagScale *ML_hypre_precon = new HypreDiagScale(*M_L_mat);
HyprePCG *ML_hypre_pcg = new HyprePCG(*M_L_mat);
ML_hypre_pcg->SetPrintLevel(0);
ML_hypre_pcg->SetMaxIter(1000);
ML_hypre_pcg->SetTol(1e-13);
ML_hypre_pcg->SetAbsTol(1e-13);
ML_hypre_pcg->SetPreconditioner(*ML_hypre_precon);
// Start each solve from zero so repeated Operator::Mult() calls do not
// depend on the output vector contents supplied by the caller.
ML_hypre_pcg->SetZeroInitialIterate();
ML_precon.reset(ML_hypre_precon);
ML_solver.reset(ML_hypre_pcg);
R.reset(new H1ConsistentMassOperator(*M_LH, *ML_solver));
return;
}
HypreParMatrix R_local = HypreParMatrix(pfes_ho.GetComm(),
pfes_lor_scalar->GlobalVSize(),
@@ -1120,17 +1215,9 @@ L2ProjectionGridTransfer::L2ProjectionH1Space::L2ProjectionH1Space(
pfes_lor_scalar->GetDofOffsets(),
pfes_ho_scalar->GetDofOffsets(),
static_cast<SparseMatrix*>(R.get()));
HypreParMatrix M_LH_local = HypreParMatrix(pfes_ho.GetComm(),
pfes_lor_scalar->GlobalVSize(),
pfes_ho_scalar->GlobalVSize(),
pfes_lor_scalar->GetDofOffsets(),
pfes_ho_scalar->GetDofOffsets(),
static_cast<SparseMatrix*>(M_LH.get()));
HypreParMatrix *R_mat = RAP(pfes_lor_scalar->Dof_TrueDof_Matrix(),
&R_local, pfes_ho_scalar->Dof_TrueDof_Matrix());
HypreParMatrix *M_LH_mat = RAP(pfes_lor_scalar->Dof_TrueDof_Matrix(),
&M_LH_local, pfes_ho_scalar->Dof_TrueDof_Matrix());
std::unique_ptr<HypreParMatrix> R_T(R_mat->Transpose());
HypreParMatrix *RTxM_LH_mat = ParMult(R_T.get(), M_LH_mat, true);
@@ -1438,6 +1525,8 @@ void L2ProjectionGridTransfer::L2ProjectionH1Space::MultTranspose(
void L2ProjectionGridTransfer::L2ProjectionH1Space::Prolongate(
const Vector& x, Vector& y) const
{
MFEM_VERIFY(!use_consistent_mass,
"BackwardOperator is not supported with consistent mass");
Vector X(fes_lor.GetTrueVSize());
Vector X_dim(M_LH->Height());
@@ -1469,6 +1558,9 @@ void L2ProjectionGridTransfer::L2ProjectionH1Space::Prolongate(
void L2ProjectionGridTransfer::L2ProjectionH1Space::ProlongateTranspose(
const Vector& x, Vector& y) const
{
MFEM_VERIFY(!use_consistent_mass,
"BackwardOperator is not supported with consistent mass");
Vector X(fes_ho.GetTrueVSize());
Vector X_dim(pcg.Width());
Vector Xbar(pcg.Height());
@@ -1499,17 +1591,34 @@ void L2ProjectionGridTransfer::L2ProjectionH1Space::ProlongateTranspose(
void L2ProjectionGridTransfer::L2ProjectionH1Space::SetRelTol(real_t p_rtol_)
{
pcg.SetRelTol(p_rtol_);
ML_pcg.SetRelTol(p_rtol_);
#ifdef MFEM_USE_MPI
if (ML_solver)
{
HyprePCG *hypre_pcg = dynamic_cast<HyprePCG*>(ML_solver.get());
if (hypre_pcg) { hypre_pcg->SetTol(p_rtol_); }
}
#endif
}
void L2ProjectionGridTransfer::L2ProjectionH1Space::SetAbsTol(real_t p_atol_)
{
pcg.SetAbsTol(p_atol_);
ML_pcg.SetAbsTol(p_atol_);
#ifdef MFEM_USE_MPI
if (ML_solver)
{
HyprePCG *hypre_pcg = dynamic_cast<HyprePCG*>(ML_solver.get());
if (hypre_pcg) { hypre_pcg->SetAbsTol(p_atol_); }
}
#endif
}
std::pair<
std::unique_ptr<SparseMatrix>,
std::unique_ptr<SparseMatrix>>
L2ProjectionGridTransfer::L2ProjectionH1Space::ComputeSparseRAndM_LH()
L2ProjectionGridTransfer::L2ProjectionH1Space::ComputeSparseRAndM_LH(
bool build_R)
{
std::pair<std::unique_ptr<SparseMatrix>,
std::unique_ptr<SparseMatrix>> r_and_mlh;
@@ -1523,10 +1632,10 @@ std::unique_ptr<SparseMatrix>>
// If the local mesh is empty, skip all computations
if (nel_ho == 0)
{
return std::make_pair(
std::unique_ptr<SparseMatrix>(new SparseMatrix),
std::unique_ptr<SparseMatrix>(new SparseMatrix)
);
std::unique_ptr<SparseMatrix> R_empty;
if (build_R) { R_empty.reset(new SparseMatrix); }
std::unique_ptr<SparseMatrix> M_LH_empty(new SparseMatrix);
return std::make_pair(std::move(R_empty), std::move(M_LH_empty));
}
const CoarseFineTransformations& cf_tr = mesh_lor->GetRefinementTransforms();
@@ -1542,69 +1651,76 @@ std::unique_ptr<SparseMatrix>>
BuildHo2Lor(nel_ho, nel_lor, cf_tr);
// ML_inv contains the inverse lumped (row sum) mass matrix. Note that the
// method will also work with a full (consistent) mass matrix, though this is
// not implemented here. L refers to the low-order refined mesh
Vector ML_inv(ndof_lor);
ML_inv = 0.0;
// Compute ML_inv
for (int iho = 0; iho < nel_ho; ++iho)
if (build_R)
{
Array<int> lor_els;
ho2lor.GetRow(iho, lor_els);
int nref = ho2lor.RowSize(iho);
// ML_inv contains the inverse lumped (row sum) mass matrix. L refers to
// the low-order refined mesh.
ML_inv = 0.0;
Geometry::Type geom = mesh_ho->GetElementBaseGeometry(iho);
const FiniteElement& fe_lor = *fes_lor.GetFE(lor_els[0]);
int nedof_lor = fe_lor.GetDof();
// Instead of using a MassIntegrator, manually loop over integration
// points so we can row sum and store the diagonal as a Vector.
Vector ML_el(nedof_lor);
Vector shape_lor(nedof_lor);
Array<int> dofs_lor(nedof_lor);
for (int iref = 0; iref < nref; ++iref)
// Compute ML_inv
for (int iho = 0; iho < nel_ho; ++iho)
{
int ilor = lor_els[iref];
ElementTransformation* el_tr = fes_lor.GetElementTransformation(ilor);
Array<int> lor_els;
ho2lor.GetRow(iho, lor_els);
int nref = ho2lor.RowSize(iho);
int order = 2 * fe_lor.GetOrder() + el_tr->OrderW();
const IntegrationRule* ir = &IntRules.Get(geom, order);
ML_el = 0.0;
for (int i = 0; i < ir->GetNPoints(); ++i)
Geometry::Type geom = mesh_ho->GetElementBaseGeometry(iho);
const FiniteElement& fe_lor = *fes_lor.GetFE(lor_els[0]);
int nedof_lor = fe_lor.GetDof();
// Instead of using a MassIntegrator, manually loop over integration
// points so we can row sum and store the diagonal as a Vector.
Vector ML_el(nedof_lor);
Vector shape_lor(nedof_lor);
Array<int> dofs_lor(nedof_lor);
for (int iref = 0; iref < nref; ++iref)
{
const IntegrationPoint& ip_lor = ir->IntPoint(i);
fe_lor.CalcShape(ip_lor, shape_lor);
el_tr->SetIntPoint(&ip_lor);
ML_el += (shape_lor *= (el_tr->Weight() * ip_lor.weight));
int ilor = lor_els[iref];
ElementTransformation* el_tr = fes_lor.GetElementTransformation(ilor);
int order = 2 * fe_lor.GetOrder() + el_tr->OrderW();
const IntegrationRule* ir = &IntRules.Get(geom, order);
ML_el = 0.0;
for (int i = 0; i < ir->GetNPoints(); ++i)
{
const IntegrationPoint& ip_lor = ir->IntPoint(i);
fe_lor.CalcShape(ip_lor, shape_lor);
el_tr->SetIntPoint(&ip_lor);
ML_el += (shape_lor *= (el_tr->Weight() * ip_lor.weight));
}
fes_lor.GetElementDofs(ilor, dofs_lor);
ML_inv.AddElementVector(dofs_lor, ML_el);
}
fes_lor.GetElementDofs(ilor, dofs_lor);
ML_inv.AddElementVector(dofs_lor, ML_el);
}
// DOF by DOF inverse of non-zero entries
LumpedMassInverse(ML_inv);
}
// DOF by DOF inverse of non-zero entries
LumpedMassInverse(ML_inv);
// Compute sparsity pattern for R = M_L^(-1) M_LH and allocate
r_and_mlh.first = AllocR();
std::unique_ptr<SparseMatrix> pattern = AllocR();
if (build_R)
{
r_and_mlh.first = std::move(pattern);
}
// Allocate M_LH (same sparsity pattern as R)
// L refers to the low-order refined mesh (DOFs correspond to rows)
// H refers to the higher-order mesh (DOFs correspond to columns)
Memory<int> I(r_and_mlh.first->Height() + 1);
for (int icol = 0; icol < r_and_mlh.first->Height() + 1; ++icol)
SparseMatrix &pattern_mat = build_R ? *r_and_mlh.first : *pattern;
Memory<int> I(pattern_mat.Height() + 1);
for (int icol = 0; icol < pattern_mat.Height() + 1; ++icol)
{
I[icol] = r_and_mlh.first->GetI()[icol];
I[icol] = pattern_mat.GetI()[icol];
}
Memory<int> J(r_and_mlh.first->NumNonZeroElems());
for (int jcol = 0; jcol < r_and_mlh.first->NumNonZeroElems(); ++jcol)
Memory<int> J(pattern_mat.NumNonZeroElems());
for (int jcol = 0; jcol < pattern_mat.NumNonZeroElems(); ++jcol)
{
J[jcol] = r_and_mlh.first->GetJ()[jcol];
J[jcol] = pattern_mat.GetJ()[jcol];
}
r_and_mlh.second = std::unique_ptr<SparseMatrix>(
new SparseMatrix(I, J, NULL, r_and_mlh.first->Height(),
r_and_mlh.first->Width(), true, true, true));
new SparseMatrix(I, J, NULL, pattern_mat.Height(),
pattern_mat.Width(), true, true, true));
IntegrationPointTransformation ip_tr;
IsoparametricTransformation& emb_tr = ip_tr.Transf;
@@ -1647,15 +1763,21 @@ std::unique_ptr<SparseMatrix>>
Array<int> dofs_lor(nedof_lor);
fes_lor.GetElementDofs(ilor, dofs_lor);
Vector R_row;
for (int i = 0; i < nedof_lor; ++i)
if (build_R)
{
M_LH_el.GetRow(i, R_row);
R_el.SetRow(i, R_row.Set(ML_inv[dofs_lor[i]], R_row));
for (int i = 0; i < nedof_lor; ++i)
{
M_LH_el.GetRow(i, R_row);
R_el.SetRow(i, R_row.Set(ML_inv[dofs_lor[i]], R_row));
}
}
Array<int> dofs_ho(nedof_ho);
fes_ho.GetElementDofs(iho, dofs_ho);
r_and_mlh.second->AddSubMatrix(dofs_lor, dofs_ho, M_LH_el);
r_and_mlh.first->AddSubMatrix(dofs_lor, dofs_ho, R_el);
if (build_R)
{
r_and_mlh.first->AddSubMatrix(dofs_lor, dofs_ho, R_el);
}
}
}
@@ -2009,6 +2131,8 @@ const Operator &L2ProjectionGridTransfer::ForwardOperator()
const Operator &L2ProjectionGridTransfer::BackwardOperator()
{
MFEM_VERIFY(!UsesH1ConsistentMass(),
"BackwardOperator is not supported with consistent mass");
if (!B)
{
if (!F) { BuildF(); }
@@ -2017,15 +2141,30 @@ const Operator &L2ProjectionGridTransfer::BackwardOperator()
return *B;
}
void L2ProjectionGridTransfer::UseConsistentMass(bool use_consistent_mass_)
{
MFEM_VERIFY(!F && !B,
"UseConsistentMass must be called before constructing operators");
use_consistent_mass = use_consistent_mass_;
}
bool L2ProjectionGridTransfer::UsesH1ConsistentMass() const
{
return use_consistent_mass && !force_l2_space &&
dom_fes.FEColl()->GetContType() == FiniteElementCollection::CONTINUOUS;
}
void L2ProjectionGridTransfer::BuildF()
{
if (!force_l2_space &&
dom_fes.FEColl()->GetContType() == FiniteElementCollection::CONTINUOUS)
{
MFEM_VERIFY(!(use_ea && use_consistent_mass),
"consistent mass is not supported with element assembly");
if (!Parallel())
{
F = new L2ProjectionH1Space(dom_fes, ran_fes,
use_ea, d_mt);
use_ea, use_consistent_mass, d_mt);
}
else
{
@@ -2035,7 +2174,7 @@ void L2ProjectionGridTransfer::BuildF()
const mfem::ParFiniteElementSpace& ran_pfes =
static_cast<mfem::ParFiniteElementSpace&>(ran_fes);
F = new L2ProjectionH1Space(dom_pfes, ran_pfes,
use_ea, d_mt);
use_ea, use_consistent_mass, d_mt);
#endif
}
}
@@ -2048,6 +2187,7 @@ void L2ProjectionGridTransfer::BuildF()
bool L2ProjectionGridTransfer::SupportsBackwardsOperator() const
{
if (UsesH1ConsistentMass()) { return false; }
return ran_fes.GetTrueVSize() >= dom_fes.GetTrueVSize();
}
@@ -2057,6 +2197,10 @@ TransferOperator::TransferOperator(const FiniteElementSpace& lFESpace_,
: Operator(hFESpace_.GetVSize(), lFESpace_.GetVSize())
{
bool isvar_order = lFESpace_.IsVariableOrder() || hFESpace_.IsVariableOrder();
bool is_trace_space =
(dynamic_cast<const H1_Trace_FECollection*>(lFESpace_.FEColl()) ||
dynamic_cast<const ND_Trace_FECollection*>(lFESpace_.FEColl()) ||
dynamic_cast<const RT_Trace_FECollection*>(lFESpace_.FEColl()));
if (lFESpace_.FEColl() == hFESpace_.FEColl() && !isvar_order)
{
OperatorPtr P(Operator::ANY_TYPE);
@@ -2066,6 +2210,7 @@ TransferOperator::TransferOperator(const FiniteElementSpace& lFESpace_,
}
else if (lFESpace_.GetVDim() == 1
&& hFESpace_.GetVDim() == 1
&& !is_trace_space
&& dynamic_cast<const TensorBasisElement*>(lFESpace_.GetTypicalFE())
&& dynamic_cast<const TensorBasisElement*>(hFESpace_.GetTypicalFE())
&& !isvar_order
@@ -2096,15 +2241,245 @@ void TransferOperator::MultTranspose(const Vector& x, Vector& y) const
PRefinementTransferOperator::PRefinementTransferOperator(
const FiniteElementSpace& lFESpace_, const FiniteElementSpace& hFESpace_)
const FiniteElementSpace& lFESpace_, const FiniteElementSpace& hFESpace_,
bool assemble_matrix)
: Operator(hFESpace_.GetVSize(), lFESpace_.GetVSize()), lFESpace(lFESpace_),
hFESpace(hFESpace_)
{
isvar_order = lFESpace_.IsVariableOrder() || hFESpace_.IsVariableOrder();
MFEM_VERIFY(lFESpace.FEColl()->GetContType() ==
hFESpace.FEColl()->GetContType(),
"Incompatible finite element space continuity types.");
is_trace_space =
(dynamic_cast<const H1_Trace_FECollection*>(lFESpace.FEColl()) ||
dynamic_cast<const ND_Trace_FECollection*>(lFESpace.FEColl()) ||
dynamic_cast<const RT_Trace_FECollection*>(lFESpace.FEColl()));
if (assemble_matrix) { AssembleMatrix(); }
}
void PRefinementTransferOperator::AssembleMatrix()
{
Mesh* mesh = hFESpace.GetMesh();
const int nL = lFESpace.GetVSize();
const int nH = hFESpace.GetVSize();
P.reset(new SparseMatrix(nH, nL));
Array<int> l_dofs, h_dofs, l_vdofs, h_vdofs;
DenseMatrix loc_prol;
Geometry::Type cached_geom = Geometry::INVALID;
const FiniteElement* h_fe = nullptr;
const FiniteElement* l_fe = nullptr;
IsoparametricTransformation T;
int vdim = lFESpace.GetVDim();
const int iend = (is_trace_space) ? mesh->GetNumFaces() : mesh->GetNE();
DofTransformation doftrans_h, doftrans_l;
Vector w(nH); w = 0.0;
for (int i = 0; i < iend; i++)
{
if (is_trace_space)
{
hFESpace.GetFaceDofs(i, h_dofs);
lFESpace.GetFaceDofs(i, l_dofs);
}
else
{
hFESpace.GetElementDofs(i, h_dofs, doftrans_h);
lFESpace.GetElementDofs(i, l_dofs, doftrans_l);
}
const Geometry::Type geom = (is_trace_space) ? mesh->GetFaceGeometry(i)
: mesh->GetElementBaseGeometry(i);
if (geom != cached_geom || isvar_order)
{
h_fe = (is_trace_space) ? hFESpace.GetFaceElement(i) : hFESpace.GetFE(i);
l_fe = (is_trace_space) ? lFESpace.GetFaceElement(i) : lFESpace.GetFE(i);
T.SetIdentityTransformation(h_fe->GetGeomType());
h_fe->GetTransferMatrix(*l_fe, T, loc_prol);
cached_geom = geom;
}
DenseMatrix Aeff(loc_prol);
TransformPrimal(doftrans_h, doftrans_l, Aeff);
for (int vd = 0; vd < vdim; vd++)
{
DenseMatrix temp_Aeff(Aeff);
l_dofs.Copy(l_vdofs);
lFESpace.DofsToVDofs(vd, l_vdofs);
h_dofs.Copy(h_vdofs);
hFESpace.DofsToVDofs(vd, h_vdofs);
temp_Aeff.AdjustDofDirection(h_vdofs, l_vdofs);
P->AddSubMatrix(h_vdofs, l_vdofs, temp_Aeff);
for (int rr = 0; rr < h_vdofs.Size(); rr++)
{
w(h_vdofs[rr]) += 1.0;
}
}
}
P->Finalize();
Vector inv_w(nH);
for (int i = 0; i < nH; i++)
{
inv_w(i) = (w(i) > 0.0) ? (1.0 / w(i)) : 1.0;
}
P->ScaleRows(inv_w);
assembled = true;
}
std::unique_ptr<SparseMatrix>
PRefinementTransferOperator::BuildConformingTransferMatrix() const
{
MFEM_VERIFY(assembled && P, "Matrix path requires assembled P.");
const SparseMatrix *Pl = lFESpace.GetConformingProlongation();
const SparseMatrix *Rh = hFESpace.GetRestrictionMatrix();
if (Pl && Rh)
{
SparseMatrix *RhP = mfem::Mult(*Rh, *P);
SparseMatrix *RhPPl = mfem::Mult(*RhP, *Pl);
delete RhP;
return std::unique_ptr<SparseMatrix>(RhPPl);
}
else if (Pl)
{
return std::unique_ptr<SparseMatrix>(mfem::Mult(*P, *Pl));
}
else if (Rh)
{
return std::unique_ptr<SparseMatrix>(mfem::Mult(*Rh, *P));
}
else
{
return std::make_unique<SparseMatrix>(*P);
}
}
std::unique_ptr<Operator>
PRefinementTransferOperator::BuildConformingTransferOperator() const
{
const Operator *Pl = lFESpace.GetProlongationMatrix();
const Operator *Rh = hFESpace.GetRestrictionOperator();
if (Pl && Rh)
{
return std::make_unique<TripleProductOperator>(Rh,
const_cast<PRefinementTransferOperator*>(this), Pl,
false, false, false);
}
else if (Pl)
{
return std::make_unique<ProductOperator>
(const_cast<PRefinementTransferOperator*>(this), Pl,
false, false);
}
else if (Rh)
{
return std::make_unique<ProductOperator>(Rh,
const_cast<PRefinementTransferOperator*>(this),
false, false);
}
else
{
// return nullptr to mean "identity/no-op wrapper", i.e. use `this`
return nullptr;
}
}
Operator *
PRefinementTransferOperator::GetTrueTransferOperator()
{
if (tP) { return tP.get(); }
#ifdef MFEM_USE_MPI
const ParFiniteElementSpace* lpfes = dynamic_cast<const ParFiniteElementSpace*>
(&lFESpace);
const ParFiniteElementSpace* hpfes = dynamic_cast<const ParFiniteElementSpace*>
(&hFESpace);
bool parallel = (lpfes) && (hpfes);
if (parallel)
{
if (assembled)
{
HypreParMatrix * Pl = lpfes->Dof_TrueDof_Matrix();
const SparseMatrix * Rh = hpfes->GetRestrictionMatrix();
// Rh * P
SparseMatrix * RhP = mfem::Mult(*Rh, *P);
HypreParMatrix * RhPh = new HypreParMatrix(hpfes->GetComm(),
hpfes->GlobalTrueVSize(), lpfes->GlobalVSize(),
hpfes->GetTrueDofOffsets(), lpfes->GetDofOffsets(), RhP);
HypreStealOwnership(*RhPh, *RhP);
delete RhP;
HypreParMatrix * tmp = ParMult(RhPh, Pl, true);
delete RhPh;
tP.reset(tmp);
return tP.get();
}
else
{
auto Pl = lpfes->GetProlongationMatrix();
auto Rh = hpfes->GetRestrictionOperator();
tP = std::make_unique<TripleProductOperator>(Rh, this, Pl, false, false, false);
return tP.get();
}
}
else
{
if (assembled)
{
auto M = BuildConformingTransferMatrix();
tP.reset(M.release());
return tP.get();
}
else
{
tP = BuildConformingTransferOperator();
return tP ? tP.get() : this;
}
}
#else
{
if (assembled)
{
auto M = BuildConformingTransferMatrix();
tP.reset(M.release());
return tP.get();
}
else
{
tP = BuildConformingTransferOperator();
return tP ? tP.get() : this;
}
}
#endif
}
void PRefinementTransferOperator::Mult(const Vector& x, Vector& y) const
{
y = 0.0;
if (assembled) { P->Mult(x, y); return; }
Mesh* mesh = hFESpace.GetMesh();
Array<int> l_dofs, h_dofs, l_vdofs, h_vdofs;
DenseMatrix loc_prol;
@@ -2117,19 +2492,31 @@ void PRefinementTransferOperator::Mult(const Vector& x, Vector& y) const
int vdim = lFESpace.GetVDim();
y = 0.0;
DofTransformation doftrans_h, doftrans_l;
for (int i = 0; i < mesh->GetNE(); i++)
{
hFESpace.GetElementDofs(i, h_dofs, doftrans_h);
lFESpace.GetElementDofs(i, l_dofs, doftrans_l);
const Geometry::Type geom = mesh->GetElementBaseGeometry(i);
const int iend = (is_trace_space) ? mesh->GetNumFaces() : mesh->GetNE();
for (int i = 0; i < iend; i++)
{
if (is_trace_space)
{
hFESpace.GetFaceDofs(i, h_dofs);
lFESpace.GetFaceDofs(i, l_dofs);
}
else
{
hFESpace.GetElementDofs(i, h_dofs, doftrans_h);
lFESpace.GetElementDofs(i, l_dofs, doftrans_l);
}
const Geometry::Type geom = (is_trace_space) ? mesh->GetFaceGeometry(i)
: mesh->GetElementBaseGeometry(i);
if (geom != cached_geom || isvar_order)
{
h_fe = hFESpace.GetFE(i);
l_fe = lFESpace.GetFE(i);
h_fe = (is_trace_space) ? hFESpace.GetFaceElement(i) : hFESpace.GetFE(i);
l_fe = (is_trace_space) ? lFESpace.GetFaceElement(i) : lFESpace.GetFE(i);
T.SetIdentityTransformation(h_fe->GetGeomType());
h_fe->GetTransferMatrix(*l_fe, T, loc_prol);
subY.SetSize(loc_prol.Height());
@@ -2144,6 +2531,7 @@ void PRefinementTransferOperator::Mult(const Vector& x, Vector& y) const
hFESpace.DofsToVDofs(vd, h_vdofs);
x.GetSubVector(l_vdofs, subX);
doftrans_l.InvTransformPrimal(subX);
loc_prol.Mult(subX, subY);
doftrans_h.TransformPrimal(subY);
y.SetSubVector(h_vdofs, subY);
@@ -2156,6 +2544,12 @@ void PRefinementTransferOperator::MultTranspose(const Vector& x,
{
y = 0.0;
if (assembled)
{
P->MultTranspose(x, y);
return;
}
Mesh* mesh = hFESpace.GetMesh();
Array<int> l_dofs, h_dofs, l_vdofs, h_vdofs;
DenseMatrix loc_prol;
@@ -2173,16 +2567,28 @@ void PRefinementTransferOperator::MultTranspose(const Vector& x,
DofTransformation doftrans_h, doftrans_l;
for (int i = 0; i < mesh->GetNE(); i++)
{
hFESpace.GetElementDofs(i, h_dofs, doftrans_h);
lFESpace.GetElementDofs(i, l_dofs, doftrans_l);
int iend = (is_trace_space) ? mesh->GetNumFaces() : mesh->GetNE();
for (int i = 0; i < iend; i++)
{
if (is_trace_space)
{
hFESpace.GetFaceDofs(i, h_dofs);
lFESpace.GetFaceDofs(i, l_dofs);
}
else
{
hFESpace.GetElementDofs(i, h_dofs, doftrans_h);
lFESpace.GetElementDofs(i, l_dofs, doftrans_l);
}
const Geometry::Type geom = (is_trace_space) ? mesh->GetFaceGeometry(i)
: mesh->GetElementBaseGeometry(i);
const Geometry::Type geom = mesh->GetElementBaseGeometry(i);
if (geom != cached_geom || isvar_order)
{
h_fe = hFESpace.GetFE(i);
l_fe = lFESpace.GetFE(i);
h_fe = (is_trace_space) ? hFESpace.GetFaceElement(i) : hFESpace.GetFE(i);
l_fe = (is_trace_space) ? lFESpace.GetFaceElement(i) : lFESpace.GetFE(i);
T.SetIdentityTransformation(h_fe->GetGeomType());
h_fe->GetTransferMatrix(*l_fe, T, loc_prol);
loc_prol.Transpose();
+92 -20
View File
@@ -169,10 +169,12 @@ public:
is the forward transfer matrix, and M_f is the mass matrix on the coarse
element. For L2 spaces, M_f is the mass matrix on the union of all fine
elements comprising the coarse element. For H1 spaces, M_f is a diagonal
(lumped) mass matrix computed through row-summation. Note that the backward
transfer operator, B, is a left inverse of the forward transfer operator, F,
i.e. B F = I. Both F and B are defined in physical space and, generally for
L2 spaces, vary between different mesh elements.
(lumped) mass matrix computed through row-summation, unless
UseConsistentMass() is enabled for the forward H1 operator. When the
backward transfer operator, B, is supported, it is a left inverse of the
forward transfer operator, F, i.e. B F = I. Both F and B are defined in
physical space and, generally for L2 spaces, vary between different mesh
elements.
This class supports H1 and L2 finite element spaces. Fine meshes are a
uniform refinement of the coarse mesh, usually created through
@@ -352,16 +354,21 @@ public:
class L2ProjectionH1Space : public L2Projection
{
const bool use_ea;
/// Use the consistent low-order mass matrix in non-EA H1 Mult() and
/// MultTranspose().
const bool use_consistent_mass;
public:
L2ProjectionH1Space(const FiniteElementSpace &fes_ho_,
const FiniteElementSpace &fes_lor_,
const bool use_ea_,
const bool use_consistent_mass_,
MemoryType d_mt_ = Device::GetHostMemoryType());
#ifdef MFEM_USE_MPI
L2ProjectionH1Space(const ParFiniteElementSpace &pfes_ho_,
const ParFiniteElementSpace &pfes_lor_,
const bool use_ea_,
const bool use_consistent_mass_,
MemoryType d_mt_ = Device::GetHostMemoryType());
#endif
/// Same as above but assembles action of R through 4 parts:
@@ -417,13 +424,33 @@ public:
void SetAbsTol(real_t p_atol_) override;
protected:
/// Applies the H1 transfer R = M_L^{-1} M_LH and its transpose, where
/// M_L is the consistent low-order mass matrix.
class H1ConsistentMassOperator : public Operator
{
private:
const Operator &M_LH;
const Solver &M_L_solver;
public:
H1ConsistentMassOperator(const Operator &M_LH_,
const Solver &M_L_solver_);
void Mult(const Vector &x, Vector &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override;
};
/// Sets up the PCG solver (sets parameters, operator, and preconditioner)
void SetupPCG();
/// @brief Computes on-rank R and M_LH matrices. If true, computes mixed mass and/or
/// inverse lumped mass matrix error when compared to device implementation.
/** @brief Computes on-rank R and M_LH matrices.
If build_R is true, the returned pair contains both R and M_LH. If
build_R is false, the first pointer is null and only M_LH is built. */
std::pair<std::unique_ptr<SparseMatrix>,
std::unique_ptr<SparseMatrix>> ComputeSparseRAndM_LH();
std::unique_ptr<SparseMatrix>> ComputeSparseRAndM_LH(
bool build_R = true);
/// @brief Recovers vector of tdofs given a vector of dofs and a finite
/// element space
@@ -453,20 +480,30 @@ public:
/// elements and refined LOR elements.
std::unique_ptr<SparseMatrix> AllocR();
CGSolver pcg;
std::unique_ptr<Solver> precon;
/// Consistent low-order mass matrix used when use_consistent_mass is true.
std::unique_ptr<Operator> M_L;
// Used to compute P = (RT*M_LH)^(-1) M_LH^T
std::unique_ptr<Operator> M_LH;
// Lumped M_L inverse operator built via EA. Wrapped with restriction maps
// to multiply with scalar TDof LOR vectors.
std::unique_ptr<Operator> ML_inv_vea;
/// Preconditioner for applying the inverse consistent low-order mass
/// matrix.
std::unique_ptr<Solver> ML_precon;
/// Serial PCG solver for applying the inverse consistent low-order mass
/// matrix in H1 Mult() and MultTranspose().
CGSolver ML_pcg;
/// Solver used by H1ConsistentMassOperator to apply M_L^{-1}.
std::unique_ptr<Solver> ML_solver;
// The restriction operator is represented as an Operator R. The
// prolongation operator is a dense matrix computed as the inverse of (R^T
// M_L R), and hence, is not stored.
// If element assembly is enabled
std::unique_ptr<Operator> R;
// Used to compute P = (RT*M_LH)^(-1) M_LH^T
std::unique_ptr<Operator> M_LH;
// Inverted operator in P = (RT*M_LH)^(-1) M_LH^T. Used to compute P via PCG.
std::unique_ptr<Operator> RTxM_LH;
// Lumped M_L inverse operator built via EA. Wrapped with restriction maps
// to multiply with scalar TDof LOR vectors.
std::unique_ptr<Operator> ML_inv_vea;
std::unique_ptr<Solver> precon;
CGSolver pcg;
// LDof Mixed mass operator built via EA. Wrapped with restriction maps to send
// scalar LDof HO vectors to LDof LOR vectors.
Operator *M_LH_local_op;
@@ -478,7 +515,6 @@ public:
Vector M_LH_ea;
// Element Assembled lumped M_L inverse built via EA. Stores diagonal as a Ldof vector.
Vector ML_inv_ea;
#ifdef MFEM_USE_MPI
std::unique_ptr<ParFiniteElementSpace> pfes_ho_scalar;
std::unique_ptr<ParFiniteElementSpace> pfes_lor_scalar;
@@ -511,6 +547,9 @@ public:
L2Projection *F; ///< Forward, coarse-to-fine, operator
L2Prolongation *B; ///< Backward, fine-to-coarse, operator
bool force_l2_space;
/// Use the consistent low-order mass matrix for non-EA H1 Mult() and
/// MultTranspose().
bool use_consistent_mass;
public:
L2ProjectionGridTransfer(FiniteElementSpace &coarse_fes_,
@@ -518,16 +557,26 @@ public:
bool force_l2_space_ = false,
MemoryType d_mt_ = Device::GetHostMemoryType()) // move to method
: GridTransfer(coarse_fes_, fine_fes_),
F(NULL), B(NULL), force_l2_space(force_l2_space_)
F(NULL), B(NULL), force_l2_space(force_l2_space_),
use_consistent_mass(false)
{ }
virtual ~L2ProjectionGridTransfer();
/** @brief Use the consistent low-order mass matrix in H1 non-EA Mult() and
MultTranspose().
This option must be set before constructing the transfer operators. It
only affects H1 transfer, is not supported with element assembly, and
disables BackwardOperator(). */
void UseConsistentMass(bool use_consistent_mass_ = true);
const Operator &ForwardOperator() override;
const Operator &BackwardOperator() override;
bool SupportsBackwardsOperator() const override;
private:
bool UsesH1ConsistentMass() const;
void BuildF();
};
@@ -569,15 +618,38 @@ private:
const FiniteElementSpace& lFESpace;
const FiniteElementSpace& hFESpace;
bool isvar_order;
bool is_trace_space;
bool assembled = false;
std::unique_ptr<SparseMatrix> P;
std::unique_ptr<Operator> tP;
std::unique_ptr<SparseMatrix> BuildConformingTransferMatrix() const;
std::unique_ptr<Operator> BuildConformingTransferOperator() const;
void AssembleMatrix();
public:
/// @brief Constructs a transfer operator from \p lFESpace to \p hFESpace
/// which have different FE collections.
/** No matrices are assembled, only the action to a vector is being computed.
The underlying finite elements need to implement the GetTransferMatrix
methods. */
/** By default no matrices are assembled, only the action to a vector is
being computed. The underlying finite elements need to implement
the GetTransferMatrix methods. */
PRefinementTransferOperator(const FiniteElementSpace& lFESpace_,
const FiniteElementSpace& hFESpace_);
const FiniteElementSpace& hFESpace_,
bool assemble_matrix = false);
/** @brief Return the true-dof transfer operator.
The returned pointer is non-owning; the operator is either this object
or a cached operator owned by this PRefinementTransferOperator. The
pointer remains valid until this PRefinementTransferOperator is
destroyed and must not be deleted by the caller. */
Operator * GetTrueTransferOperator();
const Operator * GetTrueTransferOperator() const
{
return const_cast<PRefinementTransferOperator*>(this)
->GetTrueTransferOperator();
}
/// Destructor
virtual ~PRefinementTransferOperator() { }
+12
View File
@@ -46,6 +46,8 @@ struct DofQuadLimits_CUDA
{
static constexpr int MAX_D1D = 14;
static constexpr int MAX_Q1D = 14;
static constexpr int MAX_D1D_SIMPLEX = 14;
static constexpr int MAX_Q1D_SIMPLEX = 14;
static constexpr int MAX_T1D = 32;
static constexpr int HCURL_MAX_D1D = 5;
static constexpr int HCURL_MAX_Q1D = 6;
@@ -59,6 +61,8 @@ struct DofQuadLimits_HIP
{
static constexpr int MAX_D1D = 10;
static constexpr int MAX_Q1D = 10;
static constexpr int MAX_D1D_SIMPLEX = 9;
static constexpr int MAX_Q1D_SIMPLEX = 9;
static constexpr int MAX_T1D = 32;
static constexpr int HCURL_MAX_D1D = 5;
static constexpr int HCURL_MAX_Q1D = 5;
@@ -73,9 +77,13 @@ struct DofQuadLimits_CPU
#ifndef _WIN32
static constexpr int MAX_D1D = 24;
static constexpr int MAX_Q1D = 24;
static constexpr int MAX_D1D_SIMPLEX = 24;
static constexpr int MAX_Q1D_SIMPLEX = 24;
#else
static constexpr int MAX_D1D = 14;
static constexpr int MAX_Q1D = 14;
static constexpr int MAX_D1D_SIMPLEX = 14;
static constexpr int MAX_Q1D_SIMPLEX = 14;
#endif
static constexpr int MAX_T1D = 32;
static constexpr int HCURL_MAX_D1D = 10;
@@ -117,6 +125,8 @@ struct DeviceDofQuadLimits
{
int MAX_D1D; ///< Maximum number of 1D nodal points.
int MAX_Q1D; ///< Maximum number of 1D quadrature points.
int MAX_D1D_SIMPLEX; ///< Maximum number of 1D nodal points for simplices.
int MAX_Q1D_SIMPLEX; ///< Maximum number of 1D quadrature points for simplices.
int HCURL_MAX_D1D; ///< Maximum number of 1D nodal points for H(curl).
int HCURL_MAX_Q1D; ///< Maximum number of 1D quadrature points for H(curl).
int HDIV_MAX_D1D; ///< Maximum number of 1D nodal points for H(div).
@@ -148,6 +158,8 @@ private:
{
MAX_D1D = T::MAX_D1D;
MAX_Q1D = T::MAX_Q1D;
MAX_D1D_SIMPLEX = T::MAX_D1D_SIMPLEX;
MAX_Q1D_SIMPLEX = T::MAX_Q1D_SIMPLEX;
HCURL_MAX_D1D = T::HCURL_MAX_D1D;
HCURL_MAX_Q1D = T::HCURL_MAX_Q1D;
HDIV_MAX_D1D = T::HDIV_MAX_D1D;
+28
View File
@@ -14,6 +14,7 @@
#include "blockvector.hpp"
#include "blockoperator.hpp"
namespace mfem
{
@@ -129,6 +130,33 @@ void BlockOperator::MultTranspose(const Vector &x, Vector &y) const
}
}
#ifdef MFEM_USE_MPI
HypreParMatrix * BlockOperator::GetMonolithicHypreParMatrix() const
{
Array2D<const HypreParMatrix*> blocks(nRowBlocks, nColBlocks);
for (int i = 0; i < nRowBlocks; ++i)
{
for (int j = 0; j < nColBlocks; ++j)
{
if (IsZeroBlock(i, j))
{
blocks(i, j) = nullptr;
}
else
{
auto mat = dynamic_cast<const HypreParMatrix*>(&GetBlock(i, j));
MFEM_VERIFY(mat,"BlockOperator block (" << i << "," << j
<< ") is not a HypreParMatrix.");
blocks(i, j) = mat;
}
}
}
Array2D<real_t> coef_mut = coef; // make a non-const copy
return HypreParMatrixFromBlocks(blocks, &coef_mut);
}
#endif
BlockOperator::~BlockOperator()
{
if (owns_blocks)
+12
View File
@@ -16,6 +16,9 @@
#include "../general/array.hpp"
#include "operator.hpp"
#include "blockvector.hpp"
#ifdef MFEM_USE_MPI
#include "hypre.hpp"
#endif
namespace mfem
{
@@ -105,6 +108,15 @@ public:
/// Action of the transpose operator
void MultTranspose (const Vector & x, Vector & y) const override;
#ifdef MFEM_USE_MPI
/** @brief Returns a monolithic HypreParMatrix formed by merging the blocks of
this BlockOperator, assuming every block is a HypreParMatrix.
The returned matrix is newly allocated and owned by the caller, who is
responsible for deleting it. */
HypreParMatrix * GetMonolithicHypreParMatrix() const;
#endif
~BlockOperator();
//! Controls the ownership of the blocks: if nonzero, BlockOperator will
+48
View File
@@ -10,6 +10,9 @@
// CONTRIBUTING.md for details.
#include "complex_operator.hpp"
#ifdef MFEM_USE_MPI
#include "blockoperator.hpp"
#endif
#include <set>
#include <map>
@@ -164,6 +167,51 @@ void ComplexOperator::MultTranspose(const Vector &x_r, const Vector &x_i,
}
}
#ifdef MFEM_USE_MPI
ComplexHypreParMatrix * ComplexOperator::AsComplexHypreParMatrix() const
{
HypreParMatrix *Ar = nullptr;
HypreParMatrix *Ai = nullptr;
bool own_r = false;
bool own_i = false;
if (auto *Ahr = dynamic_cast<const HypreParMatrix*>(&real()))
{
Ar = const_cast<HypreParMatrix*>(Ahr);
}
else if (auto *Br = dynamic_cast<const BlockOperator*>(&real()))
{
Ar = Br->GetMonolithicHypreParMatrix();
own_r = true;
}
else
{
MFEM_ABORT("Real part is neither HypreParMatrix nor BlockOperator.");
}
if (auto *Ahi = dynamic_cast<const HypreParMatrix*>(&imag()))
{
Ai = const_cast<HypreParMatrix*>(Ahi);
}
else if (auto *Bi = dynamic_cast<const BlockOperator*>(&imag()))
{
Ai = Bi->GetMonolithicHypreParMatrix();
own_i = true;
}
else
{
MFEM_ABORT("Imag part is neither HypreParMatrix nor BlockOperator.");
}
return new ComplexHypreParMatrix(Ar, Ai, own_r, own_i, GetConvention());
}
#endif
SparseMatrix & ComplexSparseMatrix::real()
{
+17
View File
@@ -24,6 +24,9 @@
namespace mfem
{
#ifdef MFEM_USE_MPI
class ComplexHypreParMatrix; // forward declaration
#endif
/** @brief Mimic the action of a complex operator using two real operators.
@@ -118,6 +121,20 @@ public:
Convention GetConvention() const { return convention_; }
#ifdef MFEM_USE_MPI
/** @brief Return a newly allocated ComplexHypreParMatrix representation.
If the real and imaginary parts are HypreParMatrix objects, the returned
object borrows them and they must outlive the returned
ComplexHypreParMatrix. If they are BlockOperator objects with
HypreParMatrix blocks, they are first merged into monolithic matrices
owned by the returned ComplexHypreParMatrix.
The returned ComplexHypreParMatrix is owned by the caller, who is
responsible for deleting it. */
ComplexHypreParMatrix *AsComplexHypreParMatrix() const;
#endif
protected:
// Let this be hidden from the public interface since the implementation
// depends on internal members
+41 -2
View File
@@ -2158,7 +2158,7 @@ void DenseMatrix::GetFromVector(int offset, const Vector &v)
}
}
void DenseMatrix::AdjustDofDirection(Array<int> &dofs)
void DenseMatrix::AdjustDofDirection(const Array<int> &dofs)
{
const int n = Height();
@@ -2169,7 +2169,7 @@ void DenseMatrix::AdjustDofDirection(Array<int> &dofs)
}
#endif
int *dof = dofs;
const int *dof = dofs;
for (int i = 0; i < n-1; i++)
{
const int s = (dof[i] < 0) ? (-1) : (1);
@@ -2185,6 +2185,45 @@ void DenseMatrix::AdjustDofDirection(Array<int> &dofs)
}
}
void DenseMatrix::AdjustDofDirection(Array<int> &row_dofs,
Array<int> &col_dofs)
{
const int nr = row_dofs.Size();
const int nc = col_dofs.Size();
MFEM_VERIFY(Height() == nr && Width() == nc,
"DenseMatrix::AdjustDofDirection: size mismatch.");
// Extract signs and convert to unsigned indices
Vector rsign(nr), csign(nc);
for (int i = 0; i < nr; i++)
{
const int d = row_dofs[i];
if (d >= 0) { rsign(i) = 1.0; }
else { rsign(i) = -1.0; row_dofs[i] = -d - 1; continue; }
row_dofs[i] = d;
}
for (int j = 0; j < nc; j++)
{
const int d = col_dofs[j];
if (d >= 0) { csign(j) = 1.0; }
else { csign(j) = -1.0; col_dofs[j] = -d - 1; continue; }
col_dofs[j] = d;
}
// Apply row/column signs
for (int i = 0; i < nr; i++)
{
const real_t rs = rsign(i);
for (int j = 0; j < nc; j++)
{
(*this)(i,j) *= rs * csign(j);
}
}
}
void DenseMatrix::SetRow(int row, real_t value)
{
for (int j = 0; j < Width(); j++)
+7 -1
View File
@@ -474,7 +474,13 @@ public:
void GetFromVector(int offset, const Vector &v);
/** If (dofs[i] < 0 and dofs[j] >= 0) or (dofs[i] >= 0 and dofs[j] < 0)
then (*this)(i,j) = -(*this)(i,j). */
void AdjustDofDirection(Array<int> &dofs);
void AdjustDofDirection(const Array<int> &dofs);
/** If (row_dofs[i] < 0) xor (col_dofs[j] < 0) then
(*this)(i,j) = -(*this)(i,j). This method also converts
row_dofs/col_dofs to unsigned indices (d -> -d-1). */
void AdjustDofDirection(Array<int> &row_dofs,
Array<int> &col_dofs);
/// Replace small entries, abs(a_ij) <= eps, with zero.
void Threshold(real_t eps);
+107 -64
View File
@@ -46,6 +46,19 @@
#define MFEM_GPUSPARSE_ALG HIPSPARSE_CSRMV_ALG1
#endif // defined(MFEM_USE_CUDA)
#ifdef MFEM_USE_CUDA_OR_HIP
#define MFEM_CHECK_SPARSE(call) \
do { \
auto status = (call); \
if (status != MFEM_CU_or_HIP(SPARSE_STATUS_SUCCESS)) \
{ \
MFEM_VERIFY(status == MFEM_CU_or_HIP(SPARSE_STATUS_SUCCESS), \
MFEM_cu_or_hip(sparseGetErrorString)(status)); \
} \
} while (0)
#endif // MFEM_USE_CUDA_OR_HIP
#if defined(MFEM_USE_SINGLE)
#define MFEM_CUDA_or_HIP_REAL_T MFEM_CUDA_or_HIP(_R_32F)
#elif defined(MFEM_USE_DOUBLE)
@@ -69,13 +82,16 @@ void * SparseMatrix::dBuffer = nullptr;
#endif
#endif // MFEM_USE_CUDA_OR_HIP
bool SparseMatrix::use_gpu_vendor_sparse_if_available = true;
void SparseMatrix::InitGPUSparse()
{
// Initialize cuSPARSE/hipSPARSE library
#ifdef MFEM_USE_CUDA_OR_HIP
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK))
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK) &&
use_gpu_vendor_sparse_if_available)
{
if (!handle) { MFEM_cu_or_hip(sparseCreate)(&handle); }
if (!handle) { MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreate)(&handle)); }
useGPUSparse=true;
SparseMatrixCount++;
}
@@ -92,9 +108,9 @@ void SparseMatrix::ClearGPUSparse()
if (initBuffers)
{
#if CUDA_VERSION >= 10010 || defined(MFEM_USE_HIP)
MFEM_cu_or_hip(sparseDestroySpMat)(matA_descr);
MFEM_cu_or_hip(sparseDestroyDnVec)(vecX_descr);
MFEM_cu_or_hip(sparseDestroyDnVec)(vecY_descr);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroySpMat)(matA_descr));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroyDnVec)(vecX_descr));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroyDnVec)(vecY_descr));
#else
cusparseDestroyMatDescr(matA_descr);
#endif // CUDA_VERSION >= 10010 || defined(MFEM_USE_HIP)
@@ -472,7 +488,8 @@ void SparseMatrix::SortColumnIndices()
}
#ifdef MFEM_USE_CUDA_OR_HIP
if (Device::Allows(Backend::CUDA_MASK) || Device::Allows(Backend::HIP_MASK))
if ((Device::Allows(Backend::CUDA_MASK) || Device::Allows(Backend::HIP_MASK)) &&
useGPUSparse)
{
const int m = Height();
const int n = Width();
@@ -483,14 +500,15 @@ void SparseMatrix::SortColumnIndices()
// Get size of temporary buffer needed to sort the column indices,
// allocate the temporary buffer.
size_t pBufferSizeInBytes;
MFEM_cu_or_hip(sparseXcsrsort_bufferSizeExt)(handle, m, n, nnzA, d_ia,
d_ja, &pBufferSizeInBytes);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseXcsrsort_bufferSizeExt)(handle, m, n,
nnzA, d_ia,
d_ja, &pBufferSizeInBytes));
void *pBuffer = MFEM_Cu_or_Hip(MemAlloc)(&pBuffer, pBufferSizeInBytes);
// Create matrix descriptor, will have default values
// CUSPARSE_INDEX_BASE_ZERO and CUSPARSE_MATRIX_TYPE_GENERAL.
MFEM_cu_or_hip(sparseMatDescr_t) matA_descr;
MFEM_cu_or_hip(sparseCreateMatDescr)(&matA_descr);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateMatDescr)(&matA_descr));
// Initialize permutation to identity
Array<int> P(nnzA);
@@ -499,8 +517,9 @@ void SparseMatrix::SortColumnIndices()
// Sort the column indices. The array d_ja will now be sorted. The
// permutation required to sort the values will be returned in d_P.
MFEM_cu_or_hip(sparseXcsrsort)(handle, m, n, nnzA, matA_descr, d_ia, d_ja,
d_P, pBuffer);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseXcsrsort)(handle, m, n, nnzA, matA_descr,
d_ia, d_ja,
d_P, pBuffer));
// Create a copy of the unsorted matrix values.
real_t *d_a = ReadWriteData();
@@ -510,26 +529,28 @@ void SparseMatrix::SortColumnIndices()
// Create the (input) dense vector with the unsorted values.
MFEM_cu_or_hip(sparseDnVecDescr_t) d_a_dense;
MFEM_cu_or_hip(sparseCreateDnVec)(&d_a_dense, nnzA, d_a_unsorted,
MFEM_CUDA_or_HIP_REAL_T);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateDnVec)(&d_a_dense, nnzA,
d_a_unsorted,
MFEM_CUDA_or_HIP_REAL_T));
// Create the (output) sparse vector that will have the sorted values.
MFEM_cu_or_hip(sparseSpVecDescr_t) d_a_sparse;
MFEM_cu_or_hip(sparseCreateSpVec)(&d_a_sparse, nnzA, nnzA, d_P, d_a,
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
MFEM_CUDA_or_HIP_REAL_T);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateSpVec)(&d_a_sparse, nnzA, nnzA,
d_P, d_a,
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
MFEM_CUDA_or_HIP_REAL_T));
// Sort the matrix values using the permutation vector.
MFEM_cu_or_hip(sparseGather)(handle, d_a_dense, d_a_sparse);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseGather)(handle, d_a_dense, d_a_sparse));
// The above calls may be asynchronous, so we need to wait for them to
// finish before we can free memory.
MFEM_STREAM_SYNC;
MFEM_cu_or_hip(sparseDestroyDnVec)(d_a_dense);
MFEM_cu_or_hip(sparseDestroySpVec)(d_a_sparse);
MFEM_cu_or_hip(sparseDestroyMatDescr)(matA_descr);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroyDnVec)(d_a_dense));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroySpVec)(d_a_sparse));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroyMatDescr)(matA_descr));
MFEM_Cu_or_Hip(MemFree)(d_a_unsorted);
MFEM_Cu_or_Hip(MemFree)(pBuffer);
@@ -777,25 +798,25 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
{
#if CUDA_VERSION >= 10010 || defined(MFEM_USE_HIP)
// Setup matrix descriptor
MFEM_cu_or_hip(sparseCreateCsr)(
&matA_descr,Height(),
Width(),
J.Capacity(),
const_cast<int *>(d_I),
const_cast<int *>(d_J),
const_cast<real_t *>(d_A),
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
MFEM_CUDA_or_HIP_REAL_T);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateCsr)(
&matA_descr,Height(),
Width(),
J.Capacity(),
const_cast<int *>(d_I),
const_cast<int *>(d_J),
const_cast<real_t *>(d_A),
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_32I),
MFEM_CU_or_HIP(SPARSE_INDEX_BASE_ZERO),
MFEM_CUDA_or_HIP_REAL_T));
// Create handles for input/output vectors
MFEM_cu_or_hip(sparseCreateDnVec)(&vecX_descr,
x.Size(),
const_cast<real_t *>(d_x),
MFEM_CUDA_or_HIP_REAL_T);
MFEM_cu_or_hip(sparseCreateDnVec)(&vecY_descr, y.Size(), d_y,
MFEM_CUDA_or_HIP_REAL_T);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateDnVec)(&vecX_descr,
x.Size(),
const_cast<real_t *>(d_x),
MFEM_CUDA_or_HIP_REAL_T));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseCreateDnVec)(&vecY_descr, y.Size(), d_y,
MFEM_CUDA_or_HIP_REAL_T));
#else
cusparseCreateMatDescr(&matA_descr);
cusparseSetMatIndexBase(matA_descr, CUSPARSE_INDEX_BASE_ZERO);
@@ -806,17 +827,17 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
// Allocate kernel space. Buffer is shared between different sparsemats
size_t newBufferSize = 0;
MFEM_cu_or_hip(sparseSpMV_bufferSize)(
handle,
MFEM_CU_or_HIP(SPARSE_OPERATION_NON_TRANSPOSE),
&alpha,
matA_descr,
vecX_descr,
&beta,
vecY_descr,
MFEM_CUDA_or_HIP_REAL_T,
MFEM_GPUSPARSE_ALG,
&newBufferSize);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseSpMV_bufferSize)(
handle,
MFEM_CU_or_HIP(SPARSE_OPERATION_NON_TRANSPOSE),
&alpha,
matA_descr,
vecX_descr,
&beta,
vecY_descr,
MFEM_CUDA_or_HIP_REAL_T,
MFEM_GPUSPARSE_ALG,
&newBufferSize));
// Check if we need to resize
if (newBufferSize > bufferSize)
@@ -826,24 +847,46 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
MFEM_Cu_or_Hip(MemAlloc)(&dBuffer, bufferSize);
}
// With ROCm 7, rocsparse (used by hipsparse) requires an explicit analysis call before the spmv otherwise you get errors like:
// invalid stage, the stage rocsparse_v2_spmv_stage_analysis must be executed before the stage rocsparse_v2_spmv_stage_compute
//
// It's not clear if this is supposed to be necessary or not but as of ROCm 7.2.1 it is still required to run without issues
//
// "This step is optional but if used may results in better performance."
// https://rocm.docs.amd.com/projects/hipSPARSE/en/docs-7.2.1/reference/generic.html#hipsparsespmv-preprocess
#if HIP_VERSION_MAJOR >= 7
MFEM_CHECK_SPARSE(hipsparseSpMV_preprocess(
handle,
MFEM_CU_or_HIP(SPARSE_OPERATION_NON_TRANSPOSE),
&alpha,
matA_descr,
vecX_descr,
&beta,
vecY_descr,
MFEM_CUDA_or_HIP_REAL_T,
MFEM_GPUSPARSE_ALG,
dBuffer));
#endif
#if CUDA_VERSION >= 10010 || defined(MFEM_USE_HIP)
// Update input/output vectors
MFEM_cu_or_hip(sparseDnVecSetValues)(vecX_descr,
const_cast<real_t *>(d_x));
MFEM_cu_or_hip(sparseDnVecSetValues)(vecY_descr, d_y);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDnVecSetValues)(vecX_descr,
const_cast<real_t *>(d_x)));
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDnVecSetValues)(vecY_descr, d_y));
// Y = alpha A * X + beta * Y
MFEM_cu_or_hip(sparseSpMV)(
handle,
MFEM_CU_or_HIP(SPARSE_OPERATION_NON_TRANSPOSE),
&alpha,
matA_descr,
vecX_descr,
&beta,
vecY_descr,
MFEM_CUDA_or_HIP_REAL_T,
MFEM_GPUSPARSE_ALG,
dBuffer);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseSpMV)(
handle,
MFEM_CU_or_HIP(SPARSE_OPERATION_NON_TRANSPOSE),
&alpha,
matA_descr,
vecX_descr,
&beta,
vecY_descr,
MFEM_CUDA_or_HIP_REAL_T,
MFEM_GPUSPARSE_ALG,
dBuffer));
#else
#ifdef MFEM_USE_SINGLE
cusparseScsrmv(handle,
@@ -4335,7 +4378,7 @@ SparseMatrix::~SparseMatrix()
{
if (handle)
{
MFEM_cu_or_hip(sparseDestroy)(handle);
MFEM_CHECK_SPARSE(MFEM_cu_or_hip(sparseDestroy)(handle));
handle = nullptr;
}
#ifndef MFEM_CUDA_1897_WORKAROUND
+8
View File
@@ -49,6 +49,14 @@ public:
/// Data type sparse matrix
class SparseMatrix : public AbstractSparseMatrix
{
public:
/** @brief Use the GPU vendor sparse library (cusparse/hipsparse), if
available, for sparse matrix operations. True by default. */
/** Performance is expected to be worse when set to false on the GPU, only
use false for debugging. This has no effect on CPUs.
*/
static bool use_gpu_vendor_sparse_if_available;
protected:
/// @name Arrays used by the CSR storage format.
/** */
+1 -1
View File
@@ -547,7 +547,7 @@ public:
template <typename T>
inline T ZeroSubnormal(T val)
{
return (std::fpclassify(val) == FP_SUBNORMAL) ? 0.0 : val;
return (std::fpclassify(val) == FP_SUBNORMAL) ? T{} : val;
}
inline bool IsFinite(const real_t &val)
+110 -3
View File
@@ -2831,20 +2831,25 @@ void Mesh::ReorderElements(const Array<int> &ordering, bool reorder_vertices)
// - elements - reorder of the pointers and the vertex ids if reordering
// the vertices
// - vertices - if reordering the vertices
// - boundary - update the vertex ids, if reordering the vertices
// - boundary - update the vertex ids if reordering the vertices; reorder
// the array (Dim > 1) by face index so the result matches
// what GenerateBoundaryElements would produce on a mesh that
// was originally stored in the new element order
// - faces - regenerate
// - faces_info - regenerate
// Deleted by DeleteTables():
// - el_to_edge - rebuild in 2D and 3D only
// - el_to_face - rebuild in 3D only
// - bel_to_edge - rebuild in 3D only
// - bel_to_edge - rebuild in 3D only; rows then permuted to match the new
// boundary element ordering
// - el_to_el - no need to rebuild
// - face_edge - no need to rebuild
// - edge_vertex - no need to rebuild
// - geom_factors - no need to rebuild
// - be_to_face
// - be_to_face - rebuild (Dim > 1); then permuted to match the new
// boundary element ordering
// - Nodes
@@ -2942,6 +2947,66 @@ void Mesh::ReorderElements(const Array<int> &ordering, bool reorder_vertices)
// Update faces and faces_info
GenerateFaces();
// Reorder boundary elements
if (Dim > 1)
{
// Build a sort permutation: boundary element i goes to position
// bdr_perm[i]. Sort by face index (be_to_face[i]) rather than just
// adjacent element index: after GetElementToFaceTable face indices are
// assigned in element order, so be_to_face encodes both the adjacent
// element and its local face position within that element. This makes
// the result identical to what GenerateBoundaryElements would produce on
// a mesh that was originally written in Hilbert element order.
Array<int> bdr_perm(NumOfBdrElements);
for (int i = 0; i < NumOfBdrElements; ++i) { bdr_perm[i] = i; }
bdr_perm.Sort([this](int a, int b)
{
return be_to_face[a] < be_to_face[b];
});
// Apply permutation to the boundary element array and be_to_face
Array<Element *> new_boundary(NumOfBdrElements);
Array<int> new_be_to_face(NumOfBdrElements);
for (int new_i = 0; new_i < NumOfBdrElements; ++new_i)
{
new_boundary[new_i] = boundary[bdr_perm[new_i]];
new_be_to_face[new_i] = be_to_face[bdr_perm[new_i]];
}
mfem::Swap(boundary, new_boundary);
new_boundary.DeleteAll(); // pointers are now owned by boundary; just free container
mfem::Swap(be_to_face, new_be_to_face);
// For 3D meshes bel_to_edge maps boundary element index -> edges.
// Permute its rows so the mapping stays consistent with the new boundary
// element ordering.
if (Dim == 3 && bel_to_edge)
{
int total_nnz = 0;
for (int new_i = 0; new_i < NumOfBdrElements; ++new_i)
{
total_nnz += bel_to_edge->RowSize(bdr_perm[new_i]);
}
Table *new_bel_to_edge = new Table;
new_bel_to_edge->SetDims(NumOfBdrElements, total_nnz);
int *new_I = new_bel_to_edge->GetI();
int *new_J = new_bel_to_edge->GetJ();
new_I[0] = 0;
for (int new_i = 0; new_i < NumOfBdrElements; ++new_i)
{
const int old_i = bdr_perm[new_i];
const int nrow = bel_to_edge->RowSize(old_i);
const int *old_J = bel_to_edge->GetRow(old_i);
for (int k = 0; k < nrow; ++k)
{
new_J[new_I[new_i] + k] = old_J[k];
}
new_I[new_i + 1] = new_I[new_i] + nrow;
}
delete bel_to_edge;
bel_to_edge = new_bel_to_edge;
}
}
// Build the nodes from the saved locations if they were around before
if (Nodes)
{
@@ -7098,6 +7163,48 @@ void Mesh::SetVerticesFromNodes(const GridFunction *nodes)
}
}
void Mesh::UpdateJacobianDeterminantGF(GridFunction &detgf) const
{
const FiniteElementSpace *fespace_det = detgf.FESpace();
Array<int> dofs;
IsoparametricTransformation transf;
for (int e = 0; e < GetNE(); e++)
{
const FiniteElement *fe = fespace_det->GetFE(e);
const IntegrationRule ir = fe->GetNodes();
GetElementTransformation(e, &transf);
DenseMatrix Jac(spaceDim, Dim);
Vector detvals(ir.GetNPoints());
for (int q = 0; q < ir.GetNPoints(); q++)
{
IntegrationPoint ip = ir.IntPoint(q);
transf.SetIntPoint(&ip);
Jac = transf.Jacobian();
detvals(q) = Jac.Weight();
}
fespace_det->GetElementDofs(e, dofs);
detgf.SetSubVector(dofs, detvals);
}
}
std::unique_ptr<GridFunction> Mesh::GetJacobianDeterminantGF() const
{
int mesh_poly_deg =
Nodes != NULL ? Nodes->FESpace()->GetMaxElementOrder() : 1;
// determinant order is d*p-1 for tensor product elements and
// d*(p-1) for simplices. We use the former here for simplicity.
int det_order = Dim*mesh_poly_deg-1;
L2_FECollection *fec_det = new L2_FECollection(det_order, Dim,
BasisType::GaussLobatto);
FiniteElementSpace *fespace_det =
new FiniteElementSpace(const_cast<Mesh *>(this), fec_det);
auto detgf = std::make_unique<GridFunction>(fespace_det);
detgf->MakeOwner(fec_det);
UpdateJacobianDeterminantGF(*detgf.get());
return detgf;
}
int Mesh::GetNumFaces() const
{
switch (Dim)
+18
View File
@@ -1362,6 +1362,11 @@ public:
/// A mixed mesh is one where there are multiple types of element geometries.
bool IsMixedMesh() const;
/// @brief Returns true if the mesh is a simplex mesh, false otherwise.
///
/// A simplex mesh is one where all the elements are simplices.
bool IsSimplexMesh() const { return (MeshGenerator() == 1); }
/// Returns the minimum and maximum corners of the mesh bounding box.
/** For high-order meshes, the geometry is first refined @a ref times. */
void GetBoundingBox(Vector &min, Vector &max, int ref = 2);
@@ -2204,6 +2209,13 @@ public:
by Mesh::GetFaceElements() and Mesh::GetFaceInfos(). */
FaceInformation GetFaceInformation(int f) const;
/// @brief Return the indices of the elements sharing face @a Face.
///
/// @param[in] Face Index of the face.
/// @param[out] Elem1 Index of the first element.
/// @param[out] Elem2 Index of the second neighboring element.
///
/// @sa GetFaceInfos(), GetFaceInformation(), FaceInfo
void GetFaceElements (int Face, int *Elem1, int *Elem2) const;
void GetFaceInfos (int Face, int *Inf1, int *Inf2) const;
void GetFaceInfos (int Face, int *Inf1, int *Inf2, int *NCFace) const;
@@ -2419,6 +2431,12 @@ public:
/// @}
/// Create a GridFunction representing the Jacobian determinant
std::unique_ptr<GridFunction> GetJacobianDeterminantGF() const;
/// Update Jacobian determinant values in a given gridfunction
void UpdateJacobianDeterminantGF(GridFunction &detgf) const;
/// @name Methods related to mesh refinement
/// @{
+17
View File
@@ -2014,6 +2014,23 @@ void ParMesh::DeleteFaceNbrData()
send_face_nbr_vertices.Clear();
}
std::unique_ptr<ParGridFunction> ParMesh::GetJacobianDeterminantGF() const
{
int mesh_poly_deg =
Nodes != NULL ? Nodes->FESpace()->GetMaxElementOrder() : 1;
// determinant order is d*p-1 for tensor product elements and
// d*(p-1) for simplices. We use the former here for simplicity.
int det_order = Dim*mesh_poly_deg-1;
L2_FECollection *fec_det = new L2_FECollection(det_order, Dim,
BasisType::GaussLobatto);
ParFiniteElementSpace *fespace_det =
new ParFiniteElementSpace(const_cast<ParMesh *>(this), fec_det);
auto detgf = std::make_unique<ParGridFunction>(fespace_det);
detgf->MakeOwner(fec_det);
Mesh::UpdateJacobianDeterminantGF(*detgf.get());
return detgf;
}
void ParMesh::SetCurvature(int order, bool discont, int space_dim, int ordering)
{
DeleteFaceNbrData();
+3
View File
@@ -28,6 +28,7 @@ namespace mfem
#ifdef MFEM_USE_PUMI
class ParPumiMesh;
#endif
class ParGridFunction;
/// Class for parallel meshes
class ParMesh : public Mesh
@@ -564,6 +565,8 @@ public:
void SetCurvature(int order, bool discont = false, int space_dim = -1,
int ordering = 1) override;
std::unique_ptr<ParGridFunction> GetJacobianDeterminantGF() const;
/** Replace the internal node GridFunction with a new GridFunction defined on
the given FiniteElementSpace. The new node coordinates are projected
(derived) from the current nodes/vertices. */
+4 -2
View File
@@ -14,14 +14,16 @@ util/weakform.cpp
util/complexweakform.cpp
util/blockstaticcond.cpp
util/complexstaticcond.cpp
util/pml.cpp)
util/pml.cpp
util/preconditioners.cpp)
list(APPEND DPG_HEADERS
util/weakform.hpp
util/complexweakform.hpp
util/blockstaticcond.hpp
util/complexstaticcond.hpp
util/pml.hpp)
util/pml.hpp
util/preconditioners.hpp)
if (MFEM_USE_MPI)
list(APPEND DPG_SOURCES
+3 -3
View File
@@ -20,12 +20,12 @@ CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
DPG_REAL_SEQ_SRC = util/weakform.cpp util/blockstaticcond.cpp
DPG_REAL_SEQ_SRC = util/weakform.cpp util/blockstaticcond.cpp util/preconditioners.cpp
DPG_REAL_PAR_SRC = $(DPG_REAL_SEQ_SRC) util/pweakform.cpp
DPG_REAL_OBJ = $(DPG_REAL_PAR_SRC:.cpp=.o)
DPG_COMPLEX_SEQ_SRC = util/complexweakform.cpp util/complexstaticcond.cpp util/pml.cpp
DPG_COMPLEX_PAR_SRC = $(DPG_COMPLEX_SEQ_SRC) util/pcomplexweakform.cpp
DPG_COMPLEX_SEQ_SRC = util/complexweakform.cpp util/complexstaticcond.cpp util/pml.cpp util/preconditioners.cpp
DPG_COMPLEX_PAR_SRC = $(DPG_COMPLEX_SEQ_SRC) util/pcomplexweakform.cpp
DPG_COMPLEX_OBJ = $(DPG_COMPLEX_PAR_SRC:.cpp=.o)
DIFFUSION_SRC = diffusion.cpp $(DPG_REAL_SEQ_SRC)
+62 -71
View File
@@ -16,10 +16,13 @@
// sample runs
// mpirun -np 4 pacoustics -o 3 -m ../../data/star.mesh -sref 1 -pref 2 -rnum 1.9 -sc -prob 0
// mpirun -np 4 pacoustics -o 3 -m ../../data/star.mesh -sref 1 -pref 2 -rnum 1.9 -sc -prob 0 -pmg
// mpirun -np 4 pacoustics -o 3 -m ../../data/inline-quad.mesh -sref 1 -pref 2 -rnum 5.2 -sc -prob 1
// mpirun -np 4 pacoustics -o 4 -m ../../data/inline-tri.mesh -sref 1 -pref 2 -rnum 7.1 -sc -prob 1
// mpirun -np 4 pacoustics -o 2 -m ../../data/inline-hex.mesh -sref 0 -pref 1 -rnum 1.9 -sc -prob 0
// mpirun -np 4 pacoustics -o 2 -m ../../data/inline-hex.mesh -sref 0 -pref 1 -rnum 1.9 -sc -prob 0 -pmg
// mpirun -np 4 pacoustics -o 3 -m ../../data/inline-quad.mesh -sref 2 -pref 1 -rnum 7.1 -sc -prob 2
// mpirun -np 4 pacoustics -o 3 -m ../../data/inline-quad.mesh -sref 2 -pref 1 -rnum 7.1 -sc -prob 2 -pmg
// mpirun -np 4 pacoustics -o 2 -m ../../data/inline-hex.mesh -sref 0 -pref 1 -rnum 4.1 -sc -prob 2
// mpirun -np 4 pacoustics -o 3 -m meshes/scatter.mesh -sref 1 -pref 1 -rnum 7.1 -sc -prob 3
// mpirun -np 4 pacoustics -o 4 -m meshes/scatter.mesh -sref 1 -pref 1 -rnum 10.1 -sc -prob 4
@@ -121,6 +124,7 @@
#include "mfem.hpp"
#include "util/pcomplexweakform.hpp"
#include "util/pml.hpp"
#include "util/preconditioners.hpp"
#include "../common/mfem-common.hpp"
#include <fstream>
#include <iostream>
@@ -192,6 +196,9 @@ int main(int argc, char *argv[])
int iprob = 0;
int sr = 0;
int pr = 0;
bool pmg = false;
int pmg_levels = -1;
real_t relax_factor = 2.0/3;
int visport = 19916;
bool exact_known = false;
bool with_pml = false;
@@ -216,6 +223,12 @@ int main(int argc, char *argv[])
"Number of parallel refinements.");
args.AddOption(&pr, "-pref", "--parallel-ref",
"Number of parallel refinements.");
args.AddOption(&pmg, "-pmg", "--p-refinement-multigrid", "-no-pmg",
"--no-p-refinement-multigrid", "Enable P-Refinement Multigrid.");
args.AddOption(&pmg_levels, "-pmgl","--p-refinement-multigrid-levels",
"Number of levels for P-Refinement Multigrid.");
args.AddOption(&relax_factor, "-rf", "--relaxation-factor",
"Relaxation factor for the p-multigrid smoother.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
@@ -618,91 +631,69 @@ int main(int argc, char *argv[])
a->FormLinearSystem(ess_tdof_list,x,Ah, X,B);
ComplexOperator * Ahc = Ah.As<ComplexOperator>();
BlockOperator * BlockA_r = dynamic_cast<BlockOperator *>(&Ahc->real());
BlockOperator * BlockA_i = dynamic_cast<BlockOperator *>(&Ahc->imag());
int num_blocks = BlockA_r->NumRowBlocks();
Array<int> tdof_offsets(2*num_blocks+1);
tdof_offsets[0] = 0;
int skip = (static_cond) ? 0 : 2;
int k = (static_cond) ? 2 : 0;
for (int i=0; i<num_blocks; i++)
Array<ParFiniteElementSpace *> prec_fes;
if (static_cond)
{
tdof_offsets[i+1] = trial_fes[i+k]->GetTrueVSize();
tdof_offsets[num_blocks+i+1] = trial_fes[i+k]->GetTrueVSize();
}
tdof_offsets.PartialSum();
BlockOperator blockA(tdof_offsets);
for (int i = 0; i<num_blocks; i++)
{
for (int j = 0; j<num_blocks; j++)
{
blockA.SetBlock(i,j,&BlockA_r->GetBlock(i,j));
blockA.SetBlock(i,j+num_blocks,&BlockA_i->GetBlock(i,j), -1.0);
blockA.SetBlock(i+num_blocks,j+num_blocks,&BlockA_r->GetBlock(i,j));
blockA.SetBlock(i+num_blocks,j,&BlockA_i->GetBlock(i,j));
}
}
X = 0.;
BlockDiagonalPreconditioner M(tdof_offsets);
M.owns_blocks=0;
if (!static_cond)
{
HypreBoomerAMG * solver_p = new HypreBoomerAMG((HypreParMatrix &)
BlockA_r->GetBlock(0,0));
solver_p->SetPrintLevel(0);
solver_p->SetSystemsOptions(dim);
HypreBoomerAMG * solver_u = new HypreBoomerAMG((HypreParMatrix &)
BlockA_r->GetBlock(1,1));
solver_u->SetPrintLevel(0);
solver_u->SetSystemsOptions(dim);
M.SetDiagonalBlock(0,solver_p);
M.SetDiagonalBlock(1,solver_u);
M.SetDiagonalBlock(num_blocks,solver_p);
M.SetDiagonalBlock(num_blocks+1,solver_u);
}
HypreBoomerAMG * solver_hatp = new HypreBoomerAMG((HypreParMatrix &)
BlockA_r->GetBlock(skip,skip));
solver_hatp->SetPrintLevel(0);
HypreSolver * solver_hatu = nullptr;
if (dim == 2)
{
// AMS preconditioner for 2D H(div) (trace) space
solver_hatu = new HypreAMS((HypreParMatrix &)BlockA_r->GetBlock(skip+1,skip+1),
hatu_fes);
dynamic_cast<HypreAMS*>(solver_hatu)->SetPrintLevel(0);
a->GetTraceFESpaces(prec_fes);
}
else
{
// ADS preconditioner for 3D H(div) (trace) space
solver_hatu = new HypreADS((HypreParMatrix &)BlockA_r->GetBlock(skip+1,skip+1),
hatu_fes);
dynamic_cast<HypreADS*>(solver_hatu)->SetPrintLevel(0);
prec_fes = trial_fes;
}
Solver * cprec = nullptr;
if (pmg)
{
#ifdef MFEM_USE_MUMPS
bool mumps_coarse_solver = true;
#else
bool mumps_coarse_solver = false;
#endif
std::vector<Array<int>> ess_bdr_marker(prec_fes.Size());
for (int b = 0; b<prec_fes.Size(); b++)
{
if (pmesh.bdr_attributes.Size())
{
ess_bdr_marker[b].SetSize(pmesh.bdr_attributes.Max());
int ess_block = (static_cond) ? 0 : 2;
if (b == ess_block) // hatp
{
ess_bdr_marker[b] = ess_bdr;
}
else
{
ess_bdr_marker[b] = 0;
}
}
}
cprec = new ComplexPRefinementMultigrid(prec_fes, ess_bdr_marker, *Ahc,
pmg_levels, relax_factor, mumps_coarse_solver );
}
else
{
BlockDiagonalPreconditioner * real_prec = new BlockDiagonalPreconditioner(
BlockA_r->RowOffsets());
real_prec->owns_blocks = 1;
for (int i = 0; i<BlockA_r->NumRowBlocks(); i++)
{
auto prec = MakeFESpaceDefaultSolver(prec_fes[i],0);
prec->SetOperator(BlockA_r->GetBlock(i,i));
real_prec->SetDiagonalBlock(i,prec);
}
cprec = new ComplexPreconditioner(real_prec, true);
}
M.SetDiagonalBlock(skip,solver_hatp);
M.SetDiagonalBlock(skip+1,solver_hatu);
M.SetDiagonalBlock(skip+num_blocks,solver_hatp);
M.SetDiagonalBlock(skip+num_blocks+1,solver_hatu);
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-6);
cg.SetMaxIter(10000);
cg.SetPrintLevel(0);
cg.SetPreconditioner(M);
cg.SetOperator(blockA);
cg.SetOperator(*Ahc);
cg.SetPreconditioner(*cprec);
cg.Mult(B, X);
for (int i = 0; i<num_blocks; i++)
{
delete &M.GetDiagonalBlock(i);
}
delete cprec;
int num_iter = cg.GetNumIterations();
+62 -26
View File
@@ -16,6 +16,7 @@
// sample runs
// mpirun -np 4 pconvection-diffusion -o 2 -ref 3 -prob 0 -eps 1e-1 -beta '4 2' -theta 0.0
// mpirun -np 4 pconvection-diffusion -o 3 -ref 3 -prob 0 -eps 1e-2 -beta '2 3' -theta 0.0
// mpirun -np 4 pconvection-diffusion -o 3 -ref 3 -prob 0 -eps 1e-2 -beta '2 3' -theta 0.0 -pmg
// mpirun -np 4 pconvection-diffusion -m ../../data/inline-hex.mesh -o 2 -ref 1 -prob 0 -sc -eps 1e-1 -theta 0.0
// AMR runs
@@ -66,6 +67,7 @@
#include "mfem.hpp"
#include "util/pweakform.hpp"
#include "util/preconditioners.hpp"
#include "../common/mfem-common.hpp"
#include <fstream>
#include <iostream>
@@ -121,6 +123,9 @@ int main(int argc, char *argv[])
real_t theta = 0.7;
bool static_cond = false;
epsilon = 1e0;
bool pmg = false;
int pmg_levels = -1;
real_t relax_factor = 2.0/3;
bool visualization = true;
int visport = 19916;
@@ -145,6 +150,12 @@ int main(int argc, char *argv[])
"Vector Coefficient beta");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pmg, "-pmg", "--p-refinement-multigrid", "-no-pmg",
"--no-p-refinement-multigrid", "Enable P-Refinement Multigrid.");
args.AddOption(&pmg_levels, "-pmgl","--p-refinement-multigrid-levels",
"Number of levels for P-Refinement Multigrid.");
args.AddOption(&relax_factor, "-rf", "--relaxation-factor",
"Relaxation factor for the p-multigrid smoother.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -453,44 +464,69 @@ int main(int argc, char *argv[])
BlockOperator * A = Ah.As<BlockOperator>();
BlockDiagonalPreconditioner M(A->RowOffsets());
M.owns_blocks = 1;
int skip = 0;
if (!static_cond)
Solver * preconditioner = nullptr;
Array<ParFiniteElementSpace *> prec_fes;
if (static_cond)
{
HypreBoomerAMG * amg0 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(0,0));
HypreBoomerAMG * amg1 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(1,1));
amg0->SetPrintLevel(0);
amg1->SetPrintLevel(0);
M.SetDiagonalBlock(0,amg0);
M.SetDiagonalBlock(1,amg1);
skip = 2;
}
HypreBoomerAMG * amg2 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(skip,
skip));
amg2->SetPrintLevel(0);
M.SetDiagonalBlock(skip,amg2);
HypreSolver * prec;
if (dim == 2)
{
// AMS preconditioner for 2D H(div) (trace) space
prec = new HypreAMS((HypreParMatrix &)A->GetBlock(skip+1,skip+1), hatf_fes);
a->GetTraceFESpaces(prec_fes);
}
else
{
// ADS preconditioner for 3D H(div) (trace) space
prec = new HypreADS((HypreParMatrix &)A->GetBlock(skip+1,skip+1), hatf_fes);
prec_fes = trial_fes;
}
if (pmg)
{
#ifdef MFEM_USE_MUMPS
bool mumps_coarse_solver = true;
#else
bool mumps_coarse_solver = false;
#endif
std::vector<Array<int>> ess_bdr_marker(prec_fes.Size());
for (int b = 0; b<prec_fes.Size(); b++)
{
if (pmesh.bdr_attributes.Size())
{
ess_bdr_marker[b].SetSize(pmesh.bdr_attributes.Max());
int ess_block = (static_cond) ? 0 : 2;
if (b == ess_block) // hatu space has essential bdr conditions
{
ess_bdr_marker[b] = ess_bdr_uhat;
}
else if (b == ess_block+1) // hatf space has essential bdr conditions
{
ess_bdr_marker[b] = ess_bdr_fhat;
}
else
{
ess_bdr_marker[b] = 0;
}
}
}
preconditioner = new PRefinementMultigrid(prec_fes, ess_bdr_marker, *A,
pmg_levels, relax_factor, mumps_coarse_solver);
}
else
{
preconditioner = new BlockDiagonalPreconditioner(A->RowOffsets());
auto block_diag = dynamic_cast<BlockDiagonalPreconditioner*>(preconditioner);
block_diag->owns_blocks = 1;
for (int i = 0; i<A->NumRowBlocks(); i++)
{
auto prec = MakeFESpaceDefaultSolver(prec_fes[i],0);
prec->SetOperator(A->GetBlock(i,i));
block_diag->SetDiagonalBlock(i,prec);
}
}
M.SetDiagonalBlock(skip+1,prec);
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(0);
cg.SetPreconditioner(M);
cg.SetOperator(*A);
cg.SetPreconditioner(*preconditioner);
cg.Mult(B, X);
delete preconditioner;
int num_iter = cg.GetNumIterations();
a->RecoverFEMSolution(X,x);
+55 -26
View File
@@ -15,6 +15,7 @@
//
// Sample runs
// mpirun -np 4 pdiffusion -m ../../data/inline-quad.mesh -o 3 -sref 1 -pref 2 -theta 0.0 -prob 0
// mpirun -np 4 pdiffusion -m ../../data/inline-quad.mesh -o 3 -sref 1 -pref 2 -theta 0.0 -prob 0 -pmg
// mpirun -np 4 pdiffusion -m ../../data/inline-hex.mesh -o 2 -sref 0 -pref 1 -theta 0.0 -prob 0 -sc
// mpirun -np 4 pdiffusion -m ../../data/beam-tet.mesh -o 3 -sref 0 -pref 2 -theta 0.0 -prob 0 -sc
@@ -70,6 +71,7 @@
#include "mfem.hpp"
#include "util/pweakform.hpp"
#include "util/preconditioners.hpp"
#include "../common/mfem-common.hpp"
#include <fstream>
#include <iostream>
@@ -114,6 +116,9 @@ int main(int argc, char *argv[])
int sref = 0; // initial uniform mesh refinements
int pref = 0; // parallel mesh refinements for AMR
int iprob = 0;
bool pmg = false;
int pmg_levels = -1;
real_t relax_factor = 2.0/3;
bool static_cond = false;
real_t theta = 0.7;
bool visualization = true;
@@ -137,6 +142,12 @@ int main(int argc, char *argv[])
" 0: manufactured, 1: L-shape");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pmg, "-pmg", "--p-refinement-multigrid", "-no-pmg",
"--no-p-refinement-multigrid", "Enable P-Refinement Multigrid.");
args.AddOption(&pmg_levels, "-pmgl","--p-refinement-multigrid-levels",
"Number of levels for P-Refinement Multigrid.");
args.AddOption(&relax_factor, "-rf", "--relaxation-factor",
"Relaxation factor for the p-multigrid smoother.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -186,7 +197,6 @@ int main(int argc, char *argv[])
}
}
for (int i = 0; i<sref; i++)
{
mesh.UniformRefinement();
@@ -378,44 +388,63 @@ int main(int argc, char *argv[])
BlockOperator * A = Ah.As<BlockOperator>();
BlockDiagonalPreconditioner M(A->RowOffsets());
M.owns_blocks = 1;
int skip = 0;
if (!static_cond)
Solver * preconditioner = nullptr;
Array<ParFiniteElementSpace *> prec_fes;
if (static_cond)
{
HypreBoomerAMG * amg0 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(0,0));
HypreBoomerAMG * amg1 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(1,1));
amg0->SetPrintLevel(0);
amg1->SetPrintLevel(0);
M.SetDiagonalBlock(0,amg0);
M.SetDiagonalBlock(1,amg1);
skip=2;
}
HypreBoomerAMG * amg2 = new HypreBoomerAMG((HypreParMatrix &)A->GetBlock(skip,
skip));
amg2->SetPrintLevel(0);
M.SetDiagonalBlock(skip,amg2);
HypreSolver * prec;
if (dim == 2)
{
// AMS preconditioner for 2D H(div) (trace) space
prec = new HypreAMS((HypreParMatrix &)A->GetBlock(skip+1,skip+1), hatsigma_fes);
a->GetTraceFESpaces(prec_fes);
}
else
{
// ADS preconditioner for 3D H(div) (trace) space
prec = new HypreADS((HypreParMatrix &)A->GetBlock(skip+1,skip+1), hatsigma_fes);
prec_fes = trial_fes;
}
if (pmg)
{
#ifdef MFEM_USE_MUMPS
bool mumps_coarse_solver = true;
#else
bool mumps_coarse_solver = false;
#endif
std::vector<Array<int>> ess_bdr_marker(prec_fes.Size());
for (int b = 0; b<prec_fes.Size(); b++)
{
ess_bdr_marker[b].SetSize(pmesh.bdr_attributes.Max());
int ess_block = (static_cond) ? 0 : 2;
if (b == ess_block)
{
ess_bdr_marker[b] = ess_bdr;
}
else
{
ess_bdr_marker[b] = 0;
}
}
preconditioner = new PRefinementMultigrid(prec_fes, ess_bdr_marker, *A,
pmg_levels, relax_factor, mumps_coarse_solver);
}
else
{
preconditioner = new BlockDiagonalPreconditioner(A->RowOffsets());
auto block_diag = dynamic_cast<BlockDiagonalPreconditioner*>(preconditioner);
block_diag->owns_blocks = 1;
for (int i = 0; i<A->NumRowBlocks(); i++)
{
auto prec = MakeFESpaceDefaultSolver(prec_fes[i],0);
prec->SetOperator(A->GetBlock(i,i));
block_diag->SetDiagonalBlock(i,prec);
}
}
M.SetDiagonalBlock(skip+1,prec);
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(0);
cg.SetPreconditioner(M);
cg.SetOperator(*A);
cg.SetPreconditioner(*preconditioner);
cg.Mult(B, X);
delete preconditioner;
a->RecoverFEMSolution(X,x);
Vector & residuals = a->ComputeResidual(x);
+66 -76
View File
@@ -18,6 +18,7 @@
// mpirun -np 4 pmaxwell -m ../../data/inline-quad.mesh -o 3 -sref 0 -pref 3 -rnum 4.8 -sc -prob 0
// mpirun -np 4 pmaxwell -m ../../data/inline-hex.mesh -o 2 -sref 0 -pref 1 -rnum 0.8 -sc -prob 0
// mpirun -np 4 pmaxwell -m ../../data/inline-quad.mesh -o 3 -sref 1 -pref 3 -rnum 4.8 -sc -prob 2
// mpirun -np 4 pmaxwell -m ../../data/inline-quad.mesh -o 3 -sref 1 -pref 3 -rnum 4.8 -sc -prob 2 -pmg
// mpirun -np 4 pmaxwell -o 3 -sref 1 -pref 2 -rnum 11.8 -sc -prob 3
// mpirun -np 4 pmaxwell -o 3 -sref 1 -pref 2 -rnum 9.8 -sc -prob 4
@@ -44,7 +45,7 @@
// The DPG UW deals with the First Order System
// i ω μ H + ∇ × E = 0, in Ω
// -i ω ϵ E + ∇ × H = J, in Ω
// E × n = E_0, on ∂Ω
// E × n = E, on ∂Ω
// Note: Ĵ = -iωJ
// The ultraweak-DPG formulation is obtained by integration by parts of both
@@ -71,7 +72,7 @@
// in 3D
// E,H ∈ (L^2(Ω))³
// Ê ∈ H_0^1/2(Ω)(curl, Γₕ), Ĥ ∈ H^-1/2(curl, Γₕ)
// Ê ∈ H\_0^1/2(Ω)(curl, Γₕ), Ĥ ∈ H^-1/2(curl, Γₕ)
// i ω μ (H,F) + (E,∇ × F) + < Ê, F × n > = 0, ∀ F ∈ H(curl,Ω)
// -i ω ϵ (E,G) + (H,∇ × G) + < Ĥ, G × n > = (J,G) ∀ G ∈ H(curl,Ω)
// Ê × n = E₀ on ∂Ω
@@ -106,7 +107,7 @@
// in 2D
// E ∈ (L²(Ω))² , H ∈ L²(Ω)
// Ê ∈ H^-1/2(Ω)(Γₕ), Ĥ ∈ H^1/2(Γₕ)
// i ω μ (α⁻¹ H,F) + (E, ∇ × F) + < AÊ, F > = 0, ∀ F ∈ H¹
// i ω μ (α⁻¹ H,F) + (E, ∇ × F) + < AÊ, F > = 0, ∀ F ∈ H¹
// -i ω ϵ (β E,G) + (H,∇ × G) + < Ĥ, G × n > = (J,G) ∀ G ∈ H(curl,Ω)
// Ê = E₀ on ∂Ω
// ---------------------------------------------------------------------------------
@@ -121,10 +122,10 @@
//
// in 3D
// E,H ∈ (L^2(Ω))³
// Ê ∈ H_0^1/2(Ω)(curl, Γ_h), Ĥ ∈ H^-1/2(curl, Γₕ)
// Ê ∈ H_0^1/2(Ω)(curl, Γ), Ĥ ∈ H^-1/2(curl, Γₕ)
// i ω μ (α⁻¹ H,F) + (E,∇ × F) + < Ê, F × n > = 0, ∀ F ∈ H(curl,Ω)
// -i ω ϵ (β E,G) + (H,∇ × G) + < Ĥ, G × n > = (J,G) ∀ G ∈ H(curl,Ω)
// × n = E_0 on ∂Ω
// -i ω ϵ (β E,G) + (H,∇ × G) + < Ĥ, G × n > = (J,G) ∀ G ∈ H(curl,Ω)
// Ê × n = E on ∂Ω
// -------------------------------------------------------------------------------
// | | E | H | Ê | Ĥ | RHS |
// -------------------------------------------------------------------------------
@@ -138,6 +139,7 @@
#include "mfem.hpp"
#include "util/pcomplexweakform.hpp"
#include "util/pml.hpp"
#include "util/preconditioners.hpp"
#include "../common/mfem-common.hpp"
#include <fstream>
#include <iostream>
@@ -146,13 +148,13 @@ using namespace std;
using namespace mfem;
using namespace mfem::common;
void E_exact_r(const Vector &x, Vector & E_r);
void E_exact_i(const Vector &x, Vector & E_i);
void H_exact_r(const Vector &x, Vector & H_r);
void H_exact_i(const Vector &x, Vector & H_i);
void rhs_func_r(const Vector &x, Vector & J_r);
void rhs_func_i(const Vector &x, Vector & J_i);
@@ -221,6 +223,9 @@ int main(int argc, char *argv[])
int delta_order = 1;
real_t rnum=1.0;
real_t theta = 0.0;
bool pmg = false;
int pmg_levels = -1;
real_t relax_factor = 2.0/3;
bool static_cond = false;
int iprob = 0;
int sr = 0;
@@ -255,6 +260,12 @@ int main(int argc, char *argv[])
"Number of parallel refinements.");
args.AddOption(&pr, "-pref", "--parallel-ref",
"Number of parallel refinements.");
args.AddOption(&pmg, "-pmg", "--p-refinement-multigrid", "-no-pmg",
"--no-p-refinement-multigrid", "Enable P-Refinement Multigrid.");
args.AddOption(&pmg_levels, "-pmgl","--p-refinement-multigrid-levels",
"Number of levels for P-Refinement Multigrid.");
args.AddOption(&relax_factor, "-rf", "--relaxation-factor",
"Relaxation factor for the p-multigrid smoother.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
@@ -821,89 +832,68 @@ int main(int argc, char *argv[])
a->FormLinearSystem(ess_tdof_list,x,Ah, X,B);
ComplexOperator * Ahc = Ah.As<ComplexOperator>();
BlockOperator * BlockA_r = dynamic_cast<BlockOperator *>(&Ahc->real());
BlockOperator * BlockA_i = dynamic_cast<BlockOperator *>(&Ahc->imag());
int num_blocks = BlockA_r->NumRowBlocks();
Array<int> tdof_offsets(2*num_blocks+1);
tdof_offsets[0] = 0;
int skip = (static_cond) ? 0 : 2;
int k = (static_cond) ? 2 : 0;
for (int i=0; i<num_blocks; i++)
Array<ParFiniteElementSpace *> prec_fes;
if (static_cond)
{
tdof_offsets[i+1] = trial_fes[i+k]->GetTrueVSize();
tdof_offsets[num_blocks+i+1] = trial_fes[i+k]->GetTrueVSize();
}
tdof_offsets.PartialSum();
BlockOperator blockA(tdof_offsets);
for (int i = 0; i<num_blocks; i++)
{
for (int j = 0; j<num_blocks; j++)
{
blockA.SetBlock(i,j,&BlockA_r->GetBlock(i,j));
blockA.SetBlock(i,j+num_blocks,&BlockA_i->GetBlock(i,j), -1.0);
blockA.SetBlock(i+num_blocks,j+num_blocks,&BlockA_r->GetBlock(i,j));
blockA.SetBlock(i+num_blocks,j,&BlockA_i->GetBlock(i,j));
}
}
X = 0.;
BlockDiagonalPreconditioner M(tdof_offsets);
if (!static_cond)
{
HypreBoomerAMG * solver_E = new HypreBoomerAMG((HypreParMatrix &)
BlockA_r->GetBlock(0,0));
solver_E->SetPrintLevel(0);
solver_E->SetSystemsOptions(dim);
HypreBoomerAMG * solver_H = new HypreBoomerAMG((HypreParMatrix &)
BlockA_r->GetBlock(1,1));
solver_H->SetPrintLevel(0);
solver_H->SetSystemsOptions(dim);
M.SetDiagonalBlock(0,solver_E);
M.SetDiagonalBlock(1,solver_H);
M.SetDiagonalBlock(num_blocks,solver_E);
M.SetDiagonalBlock(num_blocks+1,solver_H);
}
HypreSolver * solver_hatH = nullptr;
HypreAMS * solver_hatE = new HypreAMS((HypreParMatrix &)BlockA_r->GetBlock(skip,
skip),
hatE_fes);
solver_hatE->SetPrintLevel(0);
if (dim == 2)
{
solver_hatH = new HypreBoomerAMG((HypreParMatrix &)BlockA_r->GetBlock(skip+1,
skip+1));
dynamic_cast<HypreBoomerAMG*>(solver_hatH)->SetPrintLevel(0);
a->GetTraceFESpaces(prec_fes);
}
else
{
solver_hatH = new HypreAMS((HypreParMatrix &)BlockA_r->GetBlock(skip+1,skip+1),
hatH_fes);
dynamic_cast<HypreAMS*>(solver_hatH)->SetPrintLevel(0);
prec_fes = trial_fes;
}
Solver * cprec = nullptr;
if (pmg)
{
#ifdef MFEM_USE_MUMPS
bool mumps_coarse_solver = true;
#else
bool mumps_coarse_solver = false;
#endif
std::vector<Array<int>> ess_bdr_marker(prec_fes.Size());
for (int b = 0; b<prec_fes.Size(); b++)
{
if (pmesh.bdr_attributes.Size())
{
ess_bdr_marker[b].SetSize(pmesh.bdr_attributes.Max());
int ess_block = (static_cond) ? 0 : 2;
if (b == ess_block) // hatE
{
ess_bdr_marker[b] = ess_bdr;
}
else
{
ess_bdr_marker[b] = 0;
}
}
}
cprec = new ComplexPRefinementMultigrid(prec_fes, ess_bdr_marker, *Ahc,
pmg_levels, relax_factor, mumps_coarse_solver);
}
else
{
BlockDiagonalPreconditioner * real_prec = new BlockDiagonalPreconditioner(
BlockA_r->RowOffsets());
real_prec->owns_blocks = 1;
for (int i = 0; i<BlockA_r->NumRowBlocks(); i++)
{
auto prec = MakeFESpaceDefaultSolver(prec_fes[i],0);
prec->SetOperator(BlockA_r->GetBlock(i,i));
real_prec->SetDiagonalBlock(i,prec);
}
cprec = new ComplexPreconditioner(real_prec, true);
}
M.SetDiagonalBlock(skip,solver_hatE);
M.SetDiagonalBlock(skip+1,solver_hatH);
M.SetDiagonalBlock(skip+num_blocks,solver_hatE);
M.SetDiagonalBlock(skip+num_blocks+1,solver_hatH);
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-6);
cg.SetMaxIter(10000);
cg.SetPrintLevel(0);
cg.SetPreconditioner(M);
cg.SetOperator(blockA);
cg.SetOperator(*Ahc);
cg.SetPreconditioner(*cprec);
cg.Mult(B, X);
for (int i = 0; i<num_blocks; i++)
{
delete &M.GetDiagonalBlock(i);
}
delete cprec;
int num_iter = cg.GetNumIterations();
+17 -3
View File
@@ -69,12 +69,14 @@ void BlockStaticCondensation::SetSpaces(Array<FiniteElementSpace*> & fes_)
nblocks = fes.Size();
rblocks = 0;
tr_fes.SetSize(nblocks);
tr_fec.SetSize(nblocks);
mesh = fes[0]->GetMesh();
IsTraceSpace.SetSize(nblocks);
const FiniteElementCollection * fec;
for (int i = 0; i < nblocks; i++)
{
tr_fec[i] = nullptr;
fec = fes[i]->FEColl();
IsTraceSpace[i] =
(dynamic_cast<const H1_Trace_FECollection*>(fec) ||
@@ -86,21 +88,24 @@ void BlockStaticCondensation::SetSpaces(Array<FiniteElementSpace*> & fes_)
pmesh = dynamic_cast<ParMesh *>(mesh);
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new ParFiniteElementSpace(pmesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new ParFiniteElementSpace(pmesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
}
else
{
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new FiniteElementSpace(mesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new FiniteElementSpace(mesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
}
#else
// skip if it's an L2 space (no trace space to construct)
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new FiniteElementSpace(mesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new FiniteElementSpace(mesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
#endif
if (tr_fes[i]) { rblocks++; }
@@ -976,6 +981,15 @@ BlockStaticCondensation::~BlockStaticCondensation()
delete lmat[i]; lmat[i] = nullptr;
delete lvec[i]; lvec[i] = nullptr;
}
for (int i = 0; i<tr_fes.Size(); i++)
{
if (tr_fec[i])
{
delete tr_fes[i];
delete tr_fec[i];
}
}
}
} // namespace mfem
+6
View File
@@ -40,6 +40,7 @@ class BlockStaticCondensation
// New set of "reduced" Finite Element Spaces
// (after static condensation)
Array<FiniteElementSpace *> tr_fes;
Array<FiniteElementCollection *> tr_fec;
Array<int> dof_offsets;
Array<int> tdof_offsets;
@@ -185,6 +186,11 @@ public:
full linear system, compute the solution of the full system 'sol'. */
void ComputeSolution(const Vector &sc_sol, Vector &sol) const;
void GetTraceFESpaces(Array<FiniteElementSpace *> & trace_fes) const
{
trace_fes = tr_fes;
}
};
}
+18 -3
View File
@@ -71,12 +71,14 @@ void ComplexBlockStaticCondensation::SetSpaces(Array<FiniteElementSpace*> &
nblocks = fes.Size();
rblocks = 0;
tr_fes.SetSize(nblocks);
tr_fec.SetSize(nblocks);
mesh = fes[0]->GetMesh();
IsTraceSpace.SetSize(nblocks);
const FiniteElementCollection * fec;
for (int i = 0; i < nblocks; i++)
{
tr_fec[i] = nullptr;
fec = fes[i]->FEColl();
IsTraceSpace[i] =
(dynamic_cast<const H1_Trace_FECollection*>(fec) ||
@@ -88,21 +90,24 @@ void ComplexBlockStaticCondensation::SetSpaces(Array<FiniteElementSpace*> &
pmesh = dynamic_cast<ParMesh *>(mesh);
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new ParFiniteElementSpace(pmesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new ParFiniteElementSpace(pmesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
}
else
{
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new FiniteElementSpace(mesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new FiniteElementSpace(mesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
}
#else
// skip if it's an L2 space (no trace space to construct)
tr_fes[i] = (fec->GetContType() == FiniteElementCollection::DISCONTINUOUS) ?
nullptr : (IsTraceSpace[i]) ? fes[i] :
new FiniteElementSpace(mesh, fec->GetTraceCollection(), fes[i]->GetVDim(),
new FiniteElementSpace(mesh, tr_fec[i] = fec->GetTraceCollection(),
fes[i]->GetVDim(),
fes[i]->GetOrdering());
#endif
if (tr_fes[i]) { rblocks++; }
@@ -1157,6 +1162,16 @@ ComplexBlockStaticCondensation::~ComplexBlockStaticCondensation()
delete lmat[i]; lmat[i] = nullptr;
delete lvec[i]; lvec[i] = nullptr;
}
for (int i = 0; i<tr_fes.Size(); i++)
{
if (tr_fec[i])
{
delete tr_fes[i];
delete tr_fec[i];
}
}
}
}
+6
View File
@@ -36,6 +36,7 @@ class ComplexBlockStaticCondensation
// New set of "reduced" Finite Element Spaces
// (after static condensation)
Array<FiniteElementSpace *> tr_fes;
Array<FiniteElementCollection *> tr_fec;
Array<int> dof_offsets;
Array<int> tdof_offsets;
@@ -214,6 +215,11 @@ public:
full linear system, compute the solution of the full system 'sol'. */
void ComputeSolution(const Vector &sc_sol, Vector &sol) const;
void GetTraceFESpaces(Array<FiniteElementSpace *> & trace_fes) const
{
trace_fes = tr_fes;
}
};
}
+17
View File
@@ -273,6 +273,23 @@ public:
Vector & ComputeResidual(const Vector & x);
void GetTraceFESpaces(Array<FiniteElementSpace *> & trace_fes) const
{
trace_fes.SetSize(0);
Array<FiniteElementSpace *> trace_fes_all;
if (static_cond)
{
static_cond->GetTraceFESpaces(trace_fes_all);
for (int i = 0; i < trace_fes_all.Size(); i++)
{
if (trace_fes_all[i])
{
trace_fes.Append(trace_fes_all[i]);
}
}
}
}
/// Destroys bilinear form.
virtual ~ComplexDPGWeakForm();
+11
View File
@@ -98,6 +98,17 @@ public:
virtual void Update();
void GetTraceFESpaces(Array<ParFiniteElementSpace *> & trace_fes) const
{
Array<FiniteElementSpace *> sr_trace_fes;
ComplexDPGWeakForm::GetTraceFESpaces(sr_trace_fes);
trace_fes.SetSize(sr_trace_fes.Size());
for (int i = 0; i < sr_trace_fes.Size(); i++)
{
trace_fes[i] = dynamic_cast<ParFiniteElementSpace *>(sr_trace_fes[i]);
}
}
/// Destroys bilinear form.
virtual ~ParComplexDPGWeakForm();
+5
View File
@@ -9,6 +9,9 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_DPG_PML
#define MFEM_DPG_PML
#include "pml.hpp"
namespace mfem
@@ -292,3 +295,5 @@ void abs_detJ_Jt_J_inv_2_function(const Vector &x, CartesianPML * pml,
}
} // namespace mfem
#endif // MFEM_DPG_PML
+570
View File
@@ -0,0 +1,570 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "preconditioners.hpp"
namespace mfem
{
#ifdef MFEM_USE_MPI
Solver * MakeFESpaceDefaultSolver(
const ParFiniteElementSpace * pfespace, int print_level)
{
FiniteElementCollection const &fec = *(pfespace->FEColl());
const int vdim = pfespace->GetVDim();
const int dim = pfespace->GetParMesh()->Dimension();
Solver * prec = nullptr;
if (dynamic_cast<const H1_FECollection*>(&fec) ||
dynamic_cast<const L2_FECollection*>(&fec))
{
prec = new HypreBoomerAMG();
dynamic_cast<HypreBoomerAMG*>(prec)->SetPrintLevel(print_level);
if (vdim > 1)
{
dynamic_cast<HypreBoomerAMG*>(prec)->SetSystemsOptions(vdim);
}
return prec;
}
else if (dynamic_cast<const RT_FECollection*>(&fec) && dim == 3)
{
prec = new HypreADS(const_cast<ParFiniteElementSpace*>(pfespace));
dynamic_cast<HypreADS*>(prec)->SetPrintLevel(print_level);
return prec;
}
else if (dynamic_cast<const ND_FECollection*>(&fec) ||
dynamic_cast<const RT_FECollection*>(&fec))
{
prec = new HypreAMS(const_cast<ParFiniteElementSpace*>(pfespace));
dynamic_cast<HypreAMS*>(prec)->SetPrintLevel(print_level);
return prec;
}
else
{
MFEM_ABORT("Unsupported FiniteElementCollection type");
}
return prec;
}
PRefinementHierarchy::PRefinementHierarchy(const Array<ParFiniteElementSpace*>
&pfes_,
const std::vector<Array<int>> & ess_bdr_marker_)
: pfes(pfes_), ess_bdr_marker(ess_bdr_marker_), nblocks(pfes.Size())
{
MFEM_VERIFY(nblocks > 0, "Empty pfes.");
pmesh = pfes[0]->GetParMesh();
MFEM_VERIFY(pmesh, "pfes[0] has null ParMesh.");
MFEM_VERIFY(ess_bdr_marker.size() == static_cast<size_t>(nblocks),
"ess_bdr_marker size must match nblocks.");
int bdr_size = (pmesh->bdr_attributes.Size() > 0) ? pmesh->bdr_attributes.Max()
: 0;
for (int i = 0; i<nblocks; i++)
{
MFEM_VERIFY(ess_bdr_marker[i].Size() == bdr_size,
"ess_bdr_marker[" << i << "] size must match max bdr_attribute in mesh.");
}
}
const ParFiniteElementSpace* PRefinementHierarchy::GetParFESpace(int lev,
int b) const
{
if (lev == maxlevels - 1) { return pfes[b]; }
return fes_owned[lev][b].get();
}
int PRefinementHierarchy::GetFESpaceMinimumOrder(const ParFiniteElementSpace
*pfespace)
const
{
return (dynamic_cast<const L2_FECollection*>(pfespace->FEColl()) ||
dynamic_cast<const RT_FECollection*>(pfespace->FEColl())) ? 0 : 1;
}
void PRefinementHierarchy::BuildSpaceHierarchy(int mgmaxlevels)
{
orders.SetSize(nblocks);
Array<int> levels(nblocks);
for (int i = 0; i < nblocks; i++)
{
orders[i] = pfes[i]->FEColl()->GetConstructorOrder();
levels[i] = orders[i] - GetFESpaceMinimumOrder(pfes[i]);
}
maxlevels = levels.Min() + 1;
if (mgmaxlevels > 0)
{
maxlevels = std::min(maxlevels, mgmaxlevels);
}
MFEM_VERIFY(maxlevels >= 1, "Invalid maxlevels computed.");
fec_owned.resize(maxlevels-1);
fes_owned.resize(maxlevels-1);
T_level.resize(maxlevels-1);
for (int lev = 0; lev < maxlevels-1; lev++)
{
fec_owned[lev].resize(nblocks);
fes_owned[lev].resize(nblocks);
T_level[lev].resize(nblocks);
}
// Build ParFES hierarchy for each block
for (int b = 0; b < nblocks; b++)
{
const FiniteElementCollection *fec_ref = pfes[b]->FEColl();
const int vdim = pfes[b]->GetVDim();
const Ordering::Type ordering = pfes[b]->GetOrdering();
for (int lev = 1; lev <= maxlevels - 1; lev++)
{
const int p = orders[b] - lev;
auto &fec_ptr = fec_owned[maxlevels - lev - 1][b];
auto &fes_ptr = fes_owned[maxlevels - lev - 1][b];
fec_ptr.reset(fec_ref->Clone(p));
fes_ptr = std::make_unique<ParFiniteElementSpace>(pmesh, fec_ptr.get(),
vdim, ordering);
}
}
// build true dof lists for all levels
ess_tdof_list.resize(maxlevels);
Array<int> tdof_offsets(nblocks+1);
for (int i = 0; i< maxlevels; i++)
{
tdof_offsets[0] = 0;
for (int b = 0; b < nblocks; b++)
{
tdof_offsets[b+1] = GetParFESpace(i,b)->GetTrueVSize();
}
tdof_offsets.PartialSum();
Array<int> tdof_list;
Array<int> block_tdof_list;
for (int b = 0; b < nblocks; b++)
{
block_tdof_list.SetSize(0);
GetParFESpace(i,b)->GetEssentialTrueDofs(ess_bdr_marker[b], block_tdof_list);
for (int j = 0; j < block_tdof_list.Size(); j++)
{
block_tdof_list[j] += tdof_offsets[b];
}
tdof_list.Append(block_tdof_list);
}
ess_tdof_list[i] = tdof_list;
}
}
BlockOperator *PRefinementHierarchy::BuildProlongation(int lev)
{
MFEM_VERIFY(lev >= 0 &&
lev < maxlevels - 1, "Invalid level in BuildProlongation().");
Array<int> coarse_offsets(nblocks + 1); coarse_offsets[0] = 0;
Array<int> fine_offsets(nblocks + 1); fine_offsets[0] = 0;
for (int b = 0; b < nblocks; b++)
{
coarse_offsets[b+1] = coarse_offsets[b] + GetParFESpace(lev,
b)->GetTrueVSize();
fine_offsets[b+1] = fine_offsets[b] + GetParFESpace(lev+1,
b)->GetTrueVSize();
}
BlockOperator *Pblk = new BlockOperator(fine_offsets, coarse_offsets);
Pblk->owns_blocks = 0;
for (int b = 0; b < nblocks; b++)
{
T_level[lev][b] = std::make_unique<PRefinementTransferOperator>(
*GetParFESpace(lev, b), *GetParFESpace(lev+1, b), true);
HypreParMatrix *P =
dynamic_cast<HypreParMatrix*>(T_level[lev][b]->GetTrueTransferOperator());
MFEM_VERIFY(P, "PRefinement transfer returned null.");
Pblk->SetBlock(b, b, P);
}
return Pblk;
}
PRefinementMultigrid::PRefinementMultigrid(
const Array<ParFiniteElementSpace*> &pfes_,
const std::vector<Array<int>> & ess_bdr_marker_,
const BlockOperator &Op_, int mgmaxlevels,
real_t smoother_relax_factor, bool mumps_coarse_solver,
int coarse_cg_max_iter, real_t coarse_cg_rel_tol)
: Multigrid(), hierarchy(pfes_, ess_bdr_marker_), Op(Op_)
{
#ifndef MFEM_USE_MUMPS
if (mumps_coarse_solver)
{
MFEM_WARNING("MFEM not built with MUMPS."
"Switching to default coarse solver (CG).");
}
mumps_coarse_solver = false;
#endif
hierarchy.BuildSpaceHierarchy(mgmaxlevels);
const int maxlevels = hierarchy.maxlevels;
const int nblocks = hierarchy.nblocks;
operators.SetSize(maxlevels);
ownedOperators.SetSize(maxlevels);
smoothers.SetSize(maxlevels);
ownedSmoothers.SetSize(maxlevels);
operators[maxlevels-1] = const_cast<BlockOperator*>(&Op);
ownedOperators[maxlevels-1] = false;
const int nP = std::max(0, maxlevels - 1);
prolongations.SetSize(nP);
ownedProlongations.SetSize(nP);
// Build prolongations and Galerkin operators
for (int lev = nP - 1; lev >= 0; lev--)
{
BlockOperator *Pblk = hierarchy.BuildProlongation(lev);
prolongations[lev] =
new RectangularConstrainedOperator(Pblk, hierarchy.ess_tdof_list[lev],
hierarchy.ess_tdof_list[lev+1], true);
ownedProlongations[lev] = true;
BlockOperator *OpLevel = new BlockOperator(Pblk->ColOffsets());
OpLevel->owns_blocks = 1;
BlockOperator *OpFine = dynamic_cast<BlockOperator*>(operators[lev+1]);
MFEM_VERIFY(OpFine, "Expected BlockOperator at fine level.");
for (int i = 0; i < nblocks; i++)
{
HypreParMatrix *Pi = dynamic_cast<HypreParMatrix*>(&Pblk->GetBlock(i, i));
MFEM_VERIFY(Pi, "Expected HypreParMatrix prolongation block.");
HypreParMatrix *Pit = Pi->Transpose();
for (int j = 0; j < nblocks; j++)
{
if (OpFine->IsZeroBlock(i, j)) { continue; }
const HypreParMatrix *A_fine =
dynamic_cast<const HypreParMatrix*>(&OpFine->GetBlock(i, j));
MFEM_VERIFY(A_fine, "Expected HypreParMatrix block.");
if (i == j)
{
OpLevel->SetBlock(i, i, RAP(A_fine, Pi));
}
else
{
HypreParMatrix *Pj = dynamic_cast<HypreParMatrix*>(&Pblk->GetBlock(j, j));
MFEM_VERIFY(Pj, "Expected HypreParMatrix prolongation block.");
HypreParMatrix *APj = ParMult(A_fine, Pj, true);
HypreParMatrix *PtAP = ParMult(Pit, APj, true);
delete APj;
OpLevel->SetBlock(i, j, PtAP);
}
}
delete Pit;
}
operators[lev] = OpLevel;
ownedOperators[lev] = true;
}
// Build smoothers
for (int lev = 0; lev < operators.Size(); lev++)
{
auto *cOp = dynamic_cast<BlockOperator*>(operators[lev]);
MFEM_VERIFY(cOp, "Expected BlockOperator in operators[].");
if (lev == 0 && operators.Size() > 1) // coarse
{
#ifdef MFEM_USE_MUMPS
if (mumps_coarse_solver)
{
HypreParMatrix *Acoarse = cOp->GetMonolithicHypreParMatrix();
auto *mumps_solver = new MUMPSSolver(MPI_COMM_WORLD);
mumps_solver->SetPrintLevel(0);
mumps_solver->SetOperator(*Acoarse);
delete Acoarse;
smoothers[lev] = mumps_solver;
ownedSmoothers[lev] = true;
}
else
#endif
{
auto *bd = new BlockDiagonalPreconditioner(cOp->RowOffsets());
bd->owns_blocks = 1;
for (int b = 0; b < nblocks; b++)
{
const HypreParMatrix *Ab =
dynamic_cast<const HypreParMatrix*>(&cOp->GetBlock(b, b));
MFEM_VERIFY(Ab, "Expected HypreParMatrix block.");
auto solver = MakeFESpaceDefaultSolver(hierarchy.GetParFESpace(lev, b), 0);
solver->SetOperator(*Ab);
bd->SetDiagonalBlock(b, solver);
}
coarse_prec.reset(bd);
auto *cg = new CGSolver(MPI_COMM_WORLD);
cg->SetPrintLevel(-1);
cg->SetRelTol(coarse_cg_rel_tol);
cg->SetMaxIter(coarse_cg_max_iter);
cg->SetOperator(*cOp);
cg->SetPreconditioner(*coarse_prec);
smoothers[lev] = cg;
ownedSmoothers[lev] = true;
}
}
else
{
auto *prec = new SymmetricBlockDiagonalPreconditioner(cOp->RowOffsets(),
smoother_relax_factor);
prec->owns_blocks = 1;
for (int b = 0; b < nblocks; b++)
{
const HypreParMatrix *Ab =
dynamic_cast<const HypreParMatrix*>(&cOp->GetBlock(b, b));
MFEM_VERIFY(Ab, "Expected HypreParMatrix block.");
auto solver = MakeFESpaceDefaultSolver(hierarchy.GetParFESpace(lev, b), 0);
solver->SetOperator(*Ab);
prec->SetDiagonalBlock(b, solver);
}
smoothers[lev] = prec;
ownedSmoothers[lev] = true;
}
}
}
ComplexPRefinementMultigrid::ComplexPRefinementMultigrid(
const Array<ParFiniteElementSpace*> &pfes_,
const std::vector<Array<int>> & ess_bdr_marker,
const ComplexOperator &Op_, int mgmaxlevels,
real_t smoother_relax_factor, bool mumps_coarse_solver,
int coarse_cg_max_iter, real_t coarse_cg_rel_tol)
: Multigrid(), Op(Op_)
{
#ifndef MFEM_USE_MUMPS
if (mumps_coarse_solver)
{
MFEM_WARNING("MFEM not built with MUMPS."
"Switching to default coarse solver (CG).");
}
mumps_coarse_solver = false;
#endif
const auto *Op_r = dynamic_cast<const BlockOperator*>(&Op.real());
const auto *Op_i = dynamic_cast<const BlockOperator*>(&Op.imag());
MFEM_VERIFY(Op_r, "Expected BlockOperator from ComplexOperator real part.");
MFEM_VERIFY(Op_i, "Expected BlockOperator from ComplexOperator imag part.");
const int nblocks = Op_r->NumRowBlocks();
MFEM_VERIFY(nblocks == Op_i->NumRowBlocks(), "Real/imag block counts differ.");
hierarchy = std::make_unique<PRefinementHierarchy>(pfes_, ess_bdr_marker);
hierarchy->BuildSpaceHierarchy(mgmaxlevels);
const int maxlevels = hierarchy->maxlevels;
operators.SetSize(maxlevels);
ownedOperators.SetSize(maxlevels);
smoothers.SetSize(maxlevels);
ownedSmoothers.SetSize(maxlevels);
operators[maxlevels-1] = const_cast<ComplexOperator*>(&Op);
ownedOperators[maxlevels-1] = false;
const int nP = std::max(0, maxlevels - 1);
prolongations.SetSize(nP);
ownedProlongations.SetSize(nP);
for (int lev = nP - 1; lev >= 0; lev--)
{
BlockOperator *Pblk = hierarchy->BuildProlongation(lev);
auto ConstrOp = new RectangularConstrainedOperator(Pblk,
hierarchy->ess_tdof_list[lev],
hierarchy->ess_tdof_list[lev+1], true);
// prolongation as complex (real=Pblk, imag=nullptr)
prolongations[lev] = new ComplexOperator(ConstrOp, nullptr, true, true);
ownedProlongations[lev] = true;
auto *OpLevel_r = new BlockOperator(Pblk->ColOffsets());
auto *OpLevel_i = new BlockOperator(Pblk->ColOffsets());
OpLevel_r->owns_blocks = 1;
OpLevel_i->owns_blocks = 1;
auto *cOp = dynamic_cast<ComplexOperator*>(operators[lev+1]);
MFEM_VERIFY(cOp, "Expected ComplexOperator at fine level.");
auto *cOp_r = dynamic_cast<BlockOperator*>(&cOp->real());
auto *cOp_i = dynamic_cast<BlockOperator*>(&cOp->imag());
MFEM_VERIFY(cOp_r, "Expected BlockOperator fine real part.");
MFEM_VERIFY(cOp_i, "Expected BlockOperator fine imag part.");
for (int i = 0; i < nblocks; i++)
{
HypreParMatrix *Pi = dynamic_cast<HypreParMatrix*>(&Pblk->GetBlock(i, i));
MFEM_VERIFY(Pi, "Expected HypreParMatrix prolongation block.");
HypreParMatrix *Pit = Pi->Transpose();
for (int j = 0; j < nblocks; j++)
{
if (!cOp_r->IsZeroBlock(i, j))
{
const HypreParMatrix *A_fine_r =
dynamic_cast<const HypreParMatrix*>(&cOp_r->GetBlock(i, j));
MFEM_VERIFY(A_fine_r, "Expected HypreParMatrix block (real).");
if (i == j)
{
OpLevel_r->SetBlock(i, i, RAP(A_fine_r, Pi));
}
else
{
HypreParMatrix *Pj = dynamic_cast<HypreParMatrix*>(&Pblk->GetBlock(j, j));
MFEM_VERIFY(Pj, "Expected HypreParMatrix prolongation block.");
HypreParMatrix *APj = ParMult(A_fine_r, Pj, true);
HypreParMatrix *PtAP = ParMult(Pit, APj, true);
delete APj;
OpLevel_r->SetBlock(i, j, PtAP);
}
}
if (!cOp_i->IsZeroBlock(i, j))
{
const HypreParMatrix *A_fine_i =
dynamic_cast<const HypreParMatrix*>(&cOp_i->GetBlock(i, j));
MFEM_VERIFY(A_fine_i, "Expected HypreParMatrix block (imag).");
if (i == j)
{
OpLevel_i->SetBlock(i, i, RAP(A_fine_i, Pi));
}
else
{
HypreParMatrix *Pj = dynamic_cast<HypreParMatrix*>(&Pblk->GetBlock(j, j));
MFEM_VERIFY(Pj, "Expected HypreParMatrix prolongation block.");
HypreParMatrix *APj = ParMult(A_fine_i, Pj, true);
HypreParMatrix *PtAP = ParMult(Pit, APj, true);
delete APj;
OpLevel_i->SetBlock(i, j, PtAP);
}
}
}
delete Pit;
}
auto *OpLevel_c = new ComplexOperator(OpLevel_r, OpLevel_i, true, true);
operators[lev] = OpLevel_c;
ownedOperators[lev] = true;
}
// smoothers
for (int lev = 0; lev < operators.Size(); lev++)
{
auto *cOp = dynamic_cast<ComplexOperator*>(operators[lev]);
MFEM_VERIFY(cOp, "Expected ComplexOperator in operators[].");
auto *cOp_r = dynamic_cast<BlockOperator*>(&cOp->real());
MFEM_VERIFY(cOp_r, "Expected BlockOperator real part in ComplexOperator.");
if (lev == 0 && operators.Size() > 1) // coarse
{
#ifdef MFEM_USE_MUMPS
if (mumps_coarse_solver)
{
ComplexHypreParMatrix *Ahc = cOp->AsComplexHypreParMatrix();
HypreParMatrix *A = Ahc->GetSystemMatrix();
delete Ahc;
auto *mumps_solver = new MUMPSSolver(MPI_COMM_WORLD);
mumps_solver->SetPrintLevel(0);
mumps_solver->SetOperator(*A);
delete A;
smoothers[lev] = mumps_solver;
ownedSmoothers[lev] = true;
}
else
#endif
{
auto *prec_r = new BlockDiagonalPreconditioner(cOp_r->RowOffsets());
prec_r->owns_blocks = 1;
for (int b = 0; b < nblocks; b++)
{
const HypreParMatrix *Ab =
dynamic_cast<const HypreParMatrix*>(&cOp_r->GetBlock(b, b));
MFEM_VERIFY(Ab, "Expected HypreParMatrix block.");
auto solver = MakeFESpaceDefaultSolver(hierarchy->GetParFESpace(lev, b), 0);
solver->SetOperator(*Ab);
prec_r->SetDiagonalBlock(b, solver);
}
coarse_prec.reset(new ComplexPreconditioner(prec_r, true));
auto *cg = new CGSolver(MPI_COMM_WORLD);
cg->SetPrintLevel(-1);
cg->SetRelTol(coarse_cg_rel_tol);
cg->SetMaxIter(coarse_cg_max_iter);
cg->SetOperator(*cOp);
cg->SetPreconditioner(*coarse_prec);
smoothers[lev] = cg;
ownedSmoothers[lev] = true;
}
}
else
{
auto *prec_r = new SymmetricBlockDiagonalPreconditioner(cOp_r->RowOffsets(),
smoother_relax_factor);
prec_r->owns_blocks = 1;
for (int b = 0; b < nblocks; b++)
{
const HypreParMatrix *Ab =
dynamic_cast<const HypreParMatrix*>(&cOp_r->GetBlock(b, b));
MFEM_VERIFY(Ab, "Expected HypreParMatrix block.");
auto solver = MakeFESpaceDefaultSolver(hierarchy->GetParFESpace(lev, b), 0);
solver->SetOperator(*Ab);
prec_r->SetDiagonalBlock(b, solver);
}
smoothers[lev] = new ComplexPreconditioner(prec_r, true);
ownedSmoothers[lev] = true;
}
}
}
#endif
} // namespace mfem
+208
View File
@@ -0,0 +1,208 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_DPG_PRECONDITIONERS
#define MFEM_DPG_PRECONDITIONERS
#include "mfem.hpp"
namespace mfem
{
// A BlockDiagonalPreconditioner which assumes that all the blocks are symmetric
// Convenient to use with Multigrid Class in which a MultTranspose is needed
class SymmetricBlockDiagonalPreconditioner : public BlockDiagonalPreconditioner
{
private:
real_t c;
public:
/// @brief Constructs a symmetric block-diagonal preconditioner
/// with the given block offsets and scaling factor.
/// @param offsets The block offsets of the block-diagonal preconditioner.
/// @param c_ The scaling factor to be applied to the result.
SymmetricBlockDiagonalPreconditioner(const Array<int> & offsets,
real_t c_ = 1.0)
: BlockDiagonalPreconditioner(offsets), c(c_) { }
void Mult(const Vector & x, Vector & y) const override
{
BlockDiagonalPreconditioner::Mult(x,y);
y*=c;
}
void MultTranspose (const Vector & x, Vector & y) const override
{
this->Mult(x,y);
}
};
#ifdef MFEM_USE_MPI
/** @brief Creates a default solver for a given parallel FE space.
The default solvers are the following:
- For H1 and L2 spaces: HypreBoomerAMG
- For 3D RT spaces: HypreADS
- For 2D RT and ND spaces: HypreAMS
@param pfespace The parallel FE space for which the solver is to be created.
@param print_level The printing level for the solver.
@return a pointer to the created solver.
*/
Solver * MakeFESpaceDefaultSolver(
const ParFiniteElementSpace * pfespace, int print_level);
/// Shared helper class (real/complex) p-refinement multigrid:
/// - builds FE hierarchy
/// - builds transfer operators
///
class PRefinementHierarchy
{
public:
Array<int> orders;
const Array<ParFiniteElementSpace*> &pfes;
std::vector<Array<int>> ess_bdr_marker;
std::vector<Array<int>> ess_tdof_list;
ParMesh *pmesh = nullptr;
int nblocks;
int maxlevels = 1;
// Owned levels: 0..maxlevels-2
std::vector<std::vector<std::unique_ptr<FiniteElementCollection>>> fec_owned;
std::vector<std::vector<std::unique_ptr<ParFiniteElementSpace>>> fes_owned;
// Transfer operators per level and block (owned here)
std::vector<std::vector<std::unique_ptr<PRefinementTransferOperator>>> T_level;
PRefinementHierarchy(const Array<ParFiniteElementSpace*> &pfes_,
const std::vector<Array<int>> & ess_bdr_marker_);
const ParFiniteElementSpace* GetParFESpace(int lev, int b) const;
int GetFESpaceMinimumOrder(const ParFiniteElementSpace *pfespace) const;
/** @brief Computes orders/maxlevels and constructs fec/fes hierarchy
and T_level storage. */
void BuildSpaceHierarchy(int mgmaxlevels = -1);
/** @brief Builds block-diagonal prolongation for level lev (coarse=lev, fine=lev+1).
Its diagonal blocks are HypreParMatrix*
returned by the transfer operators stored in T_level[lev][b]. */
BlockOperator *BuildProlongation(int lev);
};
/// @brief Creates a p-refinement multigrid preconditioner for a
/// given set of parallel finite element spaces and block operators.
class PRefinementMultigrid : public Multigrid
{
private:
PRefinementHierarchy hierarchy;
const BlockOperator &Op;
std::unique_ptr<Solver> coarse_prec;
public:
PRefinementMultigrid(const Array<ParFiniteElementSpace*> &pfes_,
const std::vector<Array<int>> & ess_bdr_marker_,
const BlockOperator &Op_, int mgmaxlevels = -1,
real_t smoother_relax_factor = 2.0/3,
bool mumps_coarse_solver = false,
int coarse_cg_max_iter = 10,
real_t coarse_cg_rel_tol = 1e-3);
~PRefinementMultigrid() override = default;
};
/// @brief Creates a p-refinement multigrid preconditioner for a
/// given set of parallel finite element spaces and complex operators.
class ComplexPRefinementMultigrid : public Multigrid
{
private:
// NOTE: nblocks for the hierarchy is derived from Op.real() at construction time,
// so we store hierarchy behind a pointer to avoid a "dummy nblocks" constructor.
std::unique_ptr<PRefinementHierarchy> hierarchy;
const ComplexOperator &Op;
std::unique_ptr<Solver> coarse_prec;
public:
ComplexPRefinementMultigrid(const Array<ParFiniteElementSpace*> &pfes_,
const std::vector<Array<int>> & ess_bdr_marker,
const ComplexOperator &Op_, int mgmaxlevels = -1,
real_t smoother_relax_factor = 2.0/3,
bool mumps_coarse_solver = false,
int coarse_cg_max_iter = 10,
real_t coarse_cg_rel_tol = 1e-3);
~ComplexPRefinementMultigrid() override = default;
};
#endif
// Applies a given real preconditioner to the real and imaginary parts of a complex vector
class ComplexPreconditioner : public Solver
{
private:
const Operator *op = nullptr;
const Solver * prec = nullptr;
bool own_prec = false;
public:
ComplexPreconditioner(const Solver * real_prec, bool own = false)
: Solver(2*real_prec->Height()), prec(real_prec), own_prec(own) { }
virtual void Mult(const Vector &x, Vector &y) const override
{
int n = x.Size()/2;
MFEM_VERIFY(x.Size() == 2*n, "Invalid x vector size");
MFEM_VERIFY(y.Size() == 2*n, "Invalid y vector size");
Vector x_r(const_cast<Vector&>(x), 0, n);
Vector x_i(const_cast<Vector&>(x), n, n);
Vector y_r(y, 0, n);
Vector y_i(y, n, n);
// Apply the preconditioner to the real and imaginary parts separately
prec->Mult(x_r, y_r);
prec->Mult(x_i, y_i);
}
virtual void MultTranspose(const Vector &x, Vector &y) const override
{
int n = x.Size()/2;
MFEM_VERIFY(x.Size() == 2*n, "Invalid x vector size");
MFEM_VERIFY(y.Size() == 2*n, "Invalid y vector size");
Vector x_r(const_cast<Vector&>(x), 0, n);
Vector x_i(const_cast<Vector&>(x), n, n);
Vector y_r(y, 0, n);
Vector y_i(y, n, n);
// Apply the preconditioner to the real and imaginary parts separately
prec->MultTranspose(x_r, y_r);
prec->MultTranspose(x_i, y_i);
}
void SetOperator(const Operator &op_) override
{
MFEM_VERIFY(dynamic_cast<const ComplexOperator*>(&op_),
"ComplexPreconditioner::SetOperator only accepts ComplexOperator");
this->op = &op_;
}
~ComplexPreconditioner()
{
if (own_prec) { delete prec; }
}
};
} // namespace mfem
#endif // MFEM_DPG_PRECONDITIONERS
+11 -1
View File
@@ -76,7 +76,6 @@ public:
SetSpaces(trial_sfes,fecol_);
}
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
@@ -98,6 +97,17 @@ public:
virtual void Update();
void GetTraceFESpaces(Array<ParFiniteElementSpace *> & trace_fes) const
{
Array<FiniteElementSpace *> sr_trace_fes;
DPGWeakForm::GetTraceFESpaces(sr_trace_fes);
trace_fes.SetSize(sr_trace_fes.Size());
for (int i = 0; i < sr_trace_fes.Size(); i++)
{
trace_fes[i] = dynamic_cast<ParFiniteElementSpace *>(sr_trace_fes[i]);
}
}
/// Destroys bilinear form.
virtual ~ParDPGWeakForm();
+17
View File
@@ -290,6 +290,23 @@ public:
/// Compute DPG residual based error estimator
Vector & ComputeResidual(const BlockVector & x);
void GetTraceFESpaces(Array<FiniteElementSpace *> & trace_fes) const
{
trace_fes.SetSize(0);
Array<FiniteElementSpace *> trace_fes_all;
if (static_cond)
{
static_cond->GetTraceFESpaces(trace_fes_all);
for (int i = 0; i < trace_fes_all.Size(); i++)
{
if (trace_fes_all[i])
{
trace_fes.Append(trace_fes_all[i]);
}
}
}
}
virtual ~DPGWeakForm();
};
+5 -74
View File
@@ -41,7 +41,6 @@ using namespace mfem;
using namespace std;
Mesh MakeBoundingBoxMesh(Mesh &mesh, GridFunction &nodal_bb_gf);
void GetDeterminantJacobianGF(ParMesh *mesh, ParGridFunction *detgf);
void VisualizeBB(Mesh &mesh, char *title, int pos_x, int pos_y);
void VisualizeField(ParMesh &pmesh, ParGridFunction &input,
char *title, int pos_x, int pos_y);
@@ -149,12 +148,7 @@ int main (int argc, char *argv[])
if (!jacobian) { return 0; }
// Setup gridfunction for the determinant of the Jacobian.
// Note: determinant order = rdim*mesh_order - 1 for quads/hexes
int det_order = rdim*mesh_poly_deg-1;
L2_FECollection fec_det(det_order, rdim, BasisType::GaussLobatto);
ParFiniteElementSpace fespace_det(&pmesh, &fec_det);
ParGridFunction detgf(&fespace_det);
GetDeterminantJacobianGF(&pmesh, &detgf);
auto detgf = pmesh.GetJacobianDeterminantGF();
// Setup piecewise constant gridfunction to save bounds on the determinant
// of the Jacobian
@@ -164,13 +158,13 @@ int main (int argc, char *argv[])
ParGridFunction bounds_detgf_upper(&fes_det_pc);
// Compute bounds
detgf.GetElementBounds(bounds_detgf_lower, bounds_detgf_upper, ref_factor);
detgf->GetElementBounds(bounds_detgf_lower, bounds_detgf_upper, ref_factor);
// GLVis Visualization
if (visualization)
{
char title1[] = "Determinant of Jacobian (det J)";
VisualizeField(pmesh, detgf, title1, 0, 465);
VisualizeField(pmesh, *detgf, title1, 0, 465);
char title2[] = "Element-wise lower bound on det J";
VisualizeField(pmesh, bounds_detgf_lower, title2, 400, 465);
char title3[] = "Element-wise upper bound on det J";
@@ -181,14 +175,14 @@ int main (int argc, char *argv[])
{
VisItDataCollection visit_dc("jacobian-determinant-bounds", &pmesh);
visit_dc.SetFormat(DataCollection::PARALLEL_FORMAT);
visit_dc.RegisterField("determinant", &detgf);
visit_dc.RegisterField("determinant", detgf.get());
visit_dc.RegisterField("det-lower-bound", &bounds_detgf_lower);
visit_dc.RegisterField("det-upper-bound", &bounds_detgf_upper);
visit_dc.Save();
}
// Print min and max bound of determinant gridfunction
detgf.GetBounds(lower, upper, ref_factor);
detgf->GetBounds(lower, upper, ref_factor);
if (Mpi::Root())
{
out << "Jacobian determinant minimum bound: " << lower(0) << endl;
@@ -294,69 +288,6 @@ Mesh MakeBoundingBoxMesh(Mesh &mesh, GridFunction &nodal_bb_gf)
return meshbb;
}
IntegrationRule PermuteIR(const IntegrationRule &irule,
const Array<int> ordering)
{
const int np = irule.GetNPoints();
MFEM_VERIFY(np == ordering.Size(), "Invalid permutation size");
IntegrationRule ir(np);
ir.SetOrder(irule.GetOrder());
for (int i = 0; i < np; i++)
{
IntegrationPoint &ip_new = ir.IntPoint(i);
const IntegrationPoint &ip_old = irule.IntPoint(ordering[i]);
ip_new.Set(ip_old.x, ip_old.y, ip_old.z, ip_old.weight);
}
return ir;
}
void GetDeterminantJacobianGF(ParMesh *mesh, ParGridFunction *detgf)
{
int dim = mesh->Dimension();
FiniteElementSpace *fespace = detgf->FESpace();
Array<int> dofs;
for (int e = 0; e < mesh->GetNE(); e++)
{
const FiniteElement *fe = fespace->GetFE(e);
const IntegrationRule ir = fe->GetNodes();
ElementTransformation *transf = mesh->GetElementTransformation(e);
DenseMatrix Jac(fe->GetDim());
const NodalFiniteElement *nfe = dynamic_cast<const NodalFiniteElement*>
(fe);
const Array<int> &irordering = nfe->GetLexicographicOrdering();
IntegrationRule ir2 = irordering.Size() ?
PermuteIR(ir, irordering) :
ir;
Vector detvals(ir2.GetNPoints());
Vector loc(dim);
for (int q = 0; q < ir2.GetNPoints(); q++)
{
IntegrationPoint ip = ir2.IntPoint(q);
transf->SetIntPoint(&ip);
transf->Transform(ip, loc);
Jac = transf->Jacobian();
detvals(q) = Jac.Weight();
}
fespace->GetElementDofs(e, dofs);
if (irordering.Size())
{
for (int i = 0; i < dofs.Size(); i++)
{
(*detgf)(dofs[i]) = detvals(irordering[i]);
}
}
else
{
detgf->SetSubVector(dofs, detvals);
}
}
}
void VisualizeBB(Mesh &mesh, char *title, int pos_x, int pos_y)
{
socketstream sock;
+53 -18
View File
@@ -74,12 +74,14 @@
//
// Adaptive limiting:
// mesh-optimizer -m stretched2D.mesh -rs 1 -o 2 -mid 2 -tid 1 -ni 50 -qo 5 -nor -vl 1 -alc 1.0
// mesh-optimizer -m stretched3D.mesh -rs 2 -o 2 -mid 302 -tid 1 -ni 50 -qo 5 -nor -vl 1 -alc 2.0 -pa
// mesh-optimizer -m stretched3D.mesh -rs 2 -o 2 -mid 302 -tid 1 -rtol 1e-7 -qo 5 -nor -vl 1 -alc 2.0 -pa
// Adaptive limiting through the L-BFGS solver:
// mesh-optimizer -m stretched2D.mesh -o 2 -mid 2 -tid 1 -ni 400 -qo 5 -nor -vl 1 -alc 0.5 -st 1 -rtol 1e-8
//
// Blade shape:
// mesh-optimizer -m blade.mesh -o 4 -mid 2 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8
// Blade shape + bounded Jacobian determinant:
// * mesh-optimizer -m blade.mesh -o 4 -mid 2 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8 -db
// Blade shape (AD):
// mesh-optimizer -m blade.mesh -o 4 -mid 11 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8
// (requires CUDA):
@@ -161,6 +163,7 @@ int main(int argc, char *argv[])
int mesh_node_order = 0;
int barrier_type = 0;
int worst_case_type = 0;
bool detj_bound = false;
// Parse command-line options.
OptionsParser args(argc, argv);
@@ -317,6 +320,10 @@ int main(int argc, char *argv[])
"0 - None,"
"1 - Beta,"
"2 - PMean.");
args.AddOption(&detj_bound, "-db", "--detj-bound",
"-no-db", "--no-detj-bound",
"Enable or disable strict enforcement of positive Jacobian "
"determinants to guarantee mesh validity for tensor-product " "elements.");
args.Parse();
if (!args.Good())
{
@@ -872,13 +879,18 @@ int main(int argc, char *argv[])
if (lim_const != 0.0) { tmop_integ->EnableLimiting(x0, dist, lim_coeff); }
// Adaptive limiting.
GridFunction adapt_lim_gf0(&ind_fes);
ConstantCoefficient adapt_lim_coeff(adapt_lim_const);
GridFunction adapt_lim_gf0_1(&ind_fes);
GridFunction adapt_lim_gf0_2(&ind_fes);
ConstantCoefficient adapt_lim_coeff_1(adapt_lim_const);
const real_t adapt_lim_const_2 = 0.5 * adapt_lim_const;
ConstantCoefficient adapt_lim_coeff_2(adapt_lim_const_2);
AdaptivityEvaluator *adapt_lim_eval = NULL;
if (adapt_lim_const > 0.0)
{
FunctionCoefficient adapt_lim_gf0_coeff(adapt_lim_fun);
adapt_lim_gf0.ProjectCoefficient(adapt_lim_gf0_coeff);
FunctionCoefficient adapt_lim_gf0_coeff_1(adapt_lim_fun);
FunctionCoefficient adapt_lim_gf0_coeff_2(adapt_lim_fun2);
adapt_lim_gf0_1.ProjectCoefficient(adapt_lim_gf0_coeff_1);
adapt_lim_gf0_2.ProjectCoefficient(adapt_lim_gf0_coeff_2);
if (adapt_eval == 0) { adapt_lim_eval = new AdvectorCG(al); }
else if (adapt_eval == 1)
@@ -891,13 +903,23 @@ int main(int argc, char *argv[])
}
else { MFEM_ABORT("Bad interpolation option."); }
tmop_integ->EnableAdaptiveLimiting(adapt_lim_gf0, adapt_lim_coeff,
*adapt_lim_eval, 1.0);
Array<const GridFunction *> z0(2);
Array<Coefficient *> coeff(2);
Array<real_t> delta_max(2);
z0[0] = &adapt_lim_gf0_1;
z0[1] = &adapt_lim_gf0_2;
coeff[0] = &adapt_lim_coeff_1;
coeff[1] = &adapt_lim_coeff_2;
delta_max[0] = 1.0;
delta_max[1] = 0.5;
tmop_integ->EnableAdaptiveLimiting(z0, coeff, *adapt_lim_eval, delta_max);
if (visualization)
{
socketstream vis1;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0,
"Zeta 0 - initial mesh", 300, 600, 300, 300);
socketstream vis1, vis2;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0_1,
"Zeta0(1) - initial mesh", 300, 600, 300, 300);
common::VisualizeField(vis2, "localhost", 19916, adapt_lim_gf0_2,
"Zeta0(2) - initial mesh", 300, 900, 300, 300);
}
}
@@ -1005,11 +1027,13 @@ int main(int argc, char *argv[])
if (lim_const > 0.0 || adapt_lim_const > 0.0)
{
lim_coeff.constant = 0.0;
adapt_lim_coeff.constant = 0.0;
adapt_lim_coeff_1.constant = 0.0;
adapt_lim_coeff_2.constant = 0.0;
init_metric_energy = a.GetGridFunctionEnergy(periodic ? dx : x) /
(hradaptivity ? mesh->GetNE() : 1);
lim_coeff.constant = lim_const;
adapt_lim_coeff.constant = adapt_lim_const;
adapt_lim_coeff_1.constant = adapt_lim_const;
adapt_lim_coeff_2.constant = adapt_lim_const_2;
}
// Visualize the starting mesh and metric values.
@@ -1149,6 +1173,12 @@ int main(int argc, char *argv[])
{
solver.SetAdaptiveLinRtol(solver_art_type, 0.5, 0.9);
}
if (detj_bound)
{
const int bound_refs = 4; // number of refinements to compute bounds
const int bound_recs = 4; // number of recursions for the bound search
solver.EnsurePositiveDeterminantBound(*mesh, bound_refs, bound_recs);
}
// Level of output.
IterativeSolver::PrintLevel newton_print;
if (verbosity_level > 0) { newton_print.Errors().Warnings().Iterations(); }
@@ -1169,7 +1199,8 @@ int main(int argc, char *argv[])
hr_solver.AddFESpaceForUpdate(&fes_h1);
if (adapt_lim_const > 0.)
{
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0);
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0_1);
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0_2);
hr_solver.AddFESpaceForUpdate(&ind_fes);
}
hr_solver.Mult();
@@ -1196,11 +1227,13 @@ int main(int argc, char *argv[])
if (lim_const > 0.0 || adapt_lim_const > 0.0)
{
lim_coeff.constant = 0.0;
adapt_lim_coeff.constant = 0.0;
adapt_lim_coeff_1.constant = 0.0;
adapt_lim_coeff_2.constant = 0.0;
fin_metric_energy = a.GetGridFunctionEnergy(periodic ? dx : x) /
(hradaptivity ? mesh->GetNE() : 1);
lim_coeff.constant = lim_const;
adapt_lim_coeff.constant = adapt_lim_const;
adapt_lim_coeff_1.constant = adapt_lim_const;
adapt_lim_coeff_2.constant = adapt_lim_const_2;
}
std::cout << std::scientific << std::setprecision(4);
cout << "Initial strain energy: " << init_energy
@@ -1221,9 +1254,11 @@ int main(int argc, char *argv[])
if (adapt_lim_const > 0.0 && visualization)
{
socketstream vis0;
common::VisualizeField(vis0, "localhost", 19916, adapt_lim_gf0,
"Zeta 0 - final mesh", 600, 600, 300, 300);
socketstream vis1, vis2;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0_1,
"Zeta0(1) - final mesh", 600, 600, 300, 300);
common::VisualizeField(vis2, "localhost", 19916, adapt_lim_gf0_2,
"Zeta0(2) - final mesh", 600, 900, 300, 300);
}
// Visualize the mesh displacement.
+23 -1
View File
@@ -437,7 +437,7 @@ real_t weight_fun(const Vector &x)
real_t adapt_lim_fun(const Vector &x)
{
// Bump between these rad values, with sf sharpness.
real_t r1 = 0.45, r2 = 0.55, sf=30.0, r;
real_t r1 = 0.25, r2 = 0.35, sf = 30.0, r;
if (x.Size() == 2)
{
const real_t xc = x(0) - 0.1, yc = x(1) - 0.2;
@@ -455,6 +455,28 @@ real_t adapt_lim_fun(const Vector &x)
return val;
}
// Second field for adaptive limiting examples: uses different xc, yc, zc.
real_t adapt_lim_fun2(const Vector &x)
{
// Bump between these rad values, with sf sharpness.
real_t r1 = 0.25, r2 = 0.35, sf = 30.0, r;
if (x.Size() == 2)
{
const real_t xc = x(0) - 0.9, yc = x(1) - 0.2;
r = sqrt(xc*xc + yc*yc);
}
else
{
const real_t xc = x(0) - 0.1, yc = x(1) - 0.2, zc = x(2) - 0.0;
r = sqrt(xc*xc + yc*yc + zc*zc);
}
real_t val = 0.5*(1+std::tanh(sf*(r-r1))) - 0.5*(1+std::tanh(sf*(r-r2)));
val = std::max((real_t) 0.,val);
val = std::min((real_t) 1.,val);
return val;
}
// Used for exact surface alignment
real_t surface_level_set(const Vector &x)
{
+53 -19
View File
@@ -76,12 +76,14 @@
//
// Adaptive limiting:
// mpirun -np 4 pmesh-optimizer -m stretched2D.mesh -rs 1 -o 2 -mid 2 -tid 1 -ni 50 -qo 5 -nor -vl 1 -alc 1.0
// mpirun -np 8 pmesh-optimizer -m stretched3D.mesh -rs 2 -o 2 -mid 302 -tid 1 -ni 50 -qo 5 -nor -vl 1 -alc 2.0 -pa
// mpirun -np 8 pmesh-optimizer -m stretched3D.mesh -rs 2 -o 2 -mid 302 -tid 1 -rtol 1e-7 -qo 5 -nor -vl 1 -alc 2.0 -pa
// Adaptive limiting through the L-BFGS solver:
// mpirun -np 4 pmesh-optimizer -m stretched2D.mesh -o 2 -mid 2 -tid 1 -ni 400 -qo 5 -nor -vl 1 -alc 1.0 -st 1 -rtol 1e-8
//
// Blade shape:
// mpirun -np 4 pmesh-optimizer -m blade.mesh -o 4 -mid 2 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8
// Blade shape + bounded Jacobian determinant:
// * mpirun -np 4 pmesh-optimizer -m blade.mesh -o 4 -mid 2 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8 -db
// Blade shape (AD):
// mpirun -np 4 pmesh-optimizer -m blade.mesh -o 4 -mid 11 -tid 1 -ni 30 -ls 3 -art 1 -bnd -qt 1 -qo 8
// (requires CUDA):
@@ -173,6 +175,7 @@ int main (int argc, char *argv[])
int mesh_node_order = 0;
int barrier_type = 0;
int worst_case_type = 0;
bool detj_bound = false;
// Parse command-line options.
OptionsParser args(argc, argv);
@@ -331,7 +334,10 @@ int main (int argc, char *argv[])
"0 - None,"
"1 - Beta,"
"2 - PMean.");
args.AddOption(&detj_bound, "-db", "--detj-bound",
"-no-db", "--no-detj-bound",
"Enable or disable strict enforcement of positive Jacobian "
"determinants to guarantee mesh validity for tensor-product " "elements.");
args.Parse();
if (!args.Good())
{
@@ -909,13 +915,18 @@ int main (int argc, char *argv[])
if (lim_const != 0.0) { tmop_integ->EnableLimiting(x0, dist, lim_coeff); }
// Adaptive limiting.
ParGridFunction adapt_lim_gf0(&ind_fes);
ConstantCoefficient adapt_lim_coeff(adapt_lim_const);
ParGridFunction adapt_lim_gf0_1(&ind_fes);
ParGridFunction adapt_lim_gf0_2(&ind_fes);
ConstantCoefficient adapt_lim_coeff_1(adapt_lim_const);
const real_t adapt_lim_const_2 = 0.5 * adapt_lim_const;
ConstantCoefficient adapt_lim_coeff_2(adapt_lim_const_2);
AdaptivityEvaluator *adapt_lim_eval = NULL;
if (adapt_lim_const > 0.0)
{
FunctionCoefficient adapt_lim_gf0_coeff(adapt_lim_fun);
adapt_lim_gf0.ProjectCoefficient(adapt_lim_gf0_coeff);
FunctionCoefficient adapt_lim_gf0_coeff_1(adapt_lim_fun);
FunctionCoefficient adapt_lim_gf0_coeff_2(adapt_lim_fun2);
adapt_lim_gf0_1.ProjectCoefficient(adapt_lim_gf0_coeff_1);
adapt_lim_gf0_2.ProjectCoefficient(adapt_lim_gf0_coeff_2);
if (adapt_eval == 0) { adapt_lim_eval = new AdvectorCG(al); }
else if (adapt_eval == 1)
@@ -928,13 +939,23 @@ int main (int argc, char *argv[])
}
else { MFEM_ABORT("Bad interpolation option."); }
tmop_integ->EnableAdaptiveLimiting(adapt_lim_gf0, adapt_lim_coeff,
*adapt_lim_eval, 1.0);
Array<const ParGridFunction *> z0(2);
Array<Coefficient *> coeff(2);
Array<real_t> delta_max(2);
z0[0] = &adapt_lim_gf0_1;
z0[1] = &adapt_lim_gf0_2;
coeff[0] = &adapt_lim_coeff_1;
coeff[1] = &adapt_lim_coeff_2;
delta_max[0] = 1.0;
delta_max[1] = 0.5;
tmop_integ->EnableAdaptiveLimiting(z0, coeff, *adapt_lim_eval, delta_max);
if (visualization)
{
socketstream vis1;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0,
"Zeta 0 - initial mesh", 300, 600, 300, 300);
socketstream vis1, vis2;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0_1,
"Zeta0(1) - initial mesh", 300, 600, 300, 300);
common::VisualizeField(vis2, "localhost", 19916, adapt_lim_gf0_2,
"Zeta0(2) - initial mesh", 300, 900, 300, 300);
}
}
@@ -1050,11 +1071,13 @@ int main (int argc, char *argv[])
if (lim_const > 0.0 || adapt_lim_const > 0.0)
{
lim_coeff.constant = 0.0;
adapt_lim_coeff.constant = 0.0;
adapt_lim_coeff_1.constant = 0.0;
adapt_lim_coeff_2.constant = 0.0;
init_metric_energy = a.GetParGridFunctionEnergy(periodic ? dx : x) /
(hradaptivity ? pmesh->GetGlobalNE() : 1);
lim_coeff.constant = lim_const;
adapt_lim_coeff.constant = adapt_lim_const;
adapt_lim_coeff_1.constant = adapt_lim_const;
adapt_lim_coeff_2.constant = adapt_lim_const_2;
}
// Visualize the starting mesh and metric values.
@@ -1196,6 +1219,12 @@ int main (int argc, char *argv[])
{
solver.SetAdaptiveLinRtol(solver_art_type, 0.5, 0.9);
}
if (detj_bound)
{
const int bound_refs = 4; // number of refinements to compute bounds
const int bound_recs = 4; // number of recursions for the bound search
solver.EnsurePositiveDeterminantBound(*pmesh, bound_refs, bound_recs);
}
// Level of output.
IterativeSolver::PrintLevel newton_print;
if (verbosity_level > 0) { newton_print.Errors().Warnings().Iterations(); }
@@ -1216,7 +1245,8 @@ int main (int argc, char *argv[])
hr_solver.AddFESpaceForUpdate(&pfes_h1);
if (adapt_lim_const > 0.)
{
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0);
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0_1);
hr_solver.AddGridFunctionForUpdate(&adapt_lim_gf0_2);
hr_solver.AddFESpaceForUpdate(&ind_fes);
}
hr_solver.Mult();
@@ -1245,11 +1275,13 @@ int main (int argc, char *argv[])
if (lim_const > 0.0 || adapt_lim_const > 0.0)
{
lim_coeff.constant = 0.0;
adapt_lim_coeff.constant = 0.0;
adapt_lim_coeff_1.constant = 0.0;
adapt_lim_coeff_2.constant = 0.0;
fin_metric_energy = a.GetParGridFunctionEnergy(periodic ? dx : x) /
(hradaptivity ? pmesh->GetGlobalNE() : 1);
lim_coeff.constant = lim_const;
adapt_lim_coeff.constant = adapt_lim_const;
adapt_lim_coeff_1.constant = adapt_lim_const;
adapt_lim_coeff_2.constant = adapt_lim_const_2;
}
if (myid == 0)
{
@@ -1273,9 +1305,11 @@ int main (int argc, char *argv[])
if (adapt_lim_const > 0.0 && visualization)
{
socketstream vis0;
common::VisualizeField(vis0, "localhost", 19916, adapt_lim_gf0,
"Zeta 0 - final mesh", 600, 600, 300, 300);
socketstream vis1, vis2;
common::VisualizeField(vis1, "localhost", 19916, adapt_lim_gf0_1,
"Zeta0(1) - final mesh", 600, 600, 300, 300);
common::VisualizeField(vis2, "localhost", 19916, adapt_lim_gf0_2,
"Zeta0(2) - final mesh", 600, 900, 300, 300);
}
// Visualize the mesh displacement.
+61 -33
View File
@@ -21,6 +21,7 @@
#ifdef MFEM_USE_BENCHMARK
#include <cassert>
#include <functional>
#include <string>
#include "fem/qinterp/det.hpp" // IWYU pragma: keep
@@ -29,7 +30,7 @@
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
// Custom benchmark arguments generator
static void CustomArguments(bmi::Benchmark *b) noexcept
static void CustomArguments(bm::Benchmark *b) noexcept
{
constexpr int MAX_NDOFS = 16 * 1024 * (mfem_use_gpu ? 1024 : 8);
@@ -86,17 +87,22 @@ static void AddKernelSpecializations()
}
// Bake-off base class
template <int BFI, int VDIM, bool GLL>
template <int BFI, int VDIM, bool GLL, bool SIMPLICES>
struct BakeOff
{
inline static constexpr int DIM = 3;
static constexpr int DIM = 3;
static constexpr bool visualization = false;
static constexpr bool Simplices = SIMPLICES;
const int p, c, q, n, nx, ny, nz;
Mesh mesh;
H1_FECollection fec;
FiniteElementSpace fes;
const Geometry::Type geom_type;
IntegrationRules irs;
const IntegrationRule *ir;
const IntegrationRule *ir, *ir_rhs;
ConstantCoefficient one;
Vector uvec;
VectorConstantCoefficient unit_vec;
@@ -107,17 +113,24 @@ struct BakeOff
BilinearFormIntegrator *bfi;
BakeOff(int p, int side):
p(p), c(side), q(2 * p + (GLL ? -1 : 3)),
p(p), c(side),
q(2 * p + (GLL ? (SIMPLICES && BFI != 7) ? 0 : -1 : 3)),
n((assert(c >= p), c / p)),
nx(n + (p * (n + 1) * p * n * p * n < c * c * c ? 1 : 0)),
ny(n + (p * (n + 1) * p * (n + 1) * p * n < c * c * c ? 1 : 0)),
nz(n),
mesh(Mesh::MakeCartesian3D(nx, ny, nz, Element::HEXAHEDRON)),
fec(p, DIM, BasisType::GaussLobatto),
mesh(Mesh::MakeCartesian3D(nx, ny, nz,
SIMPLICES
? Element::TETRAHEDRON
: Element::HEXAHEDRON)),
fec(p, DIM, SIMPLICES ? BasisType::Positive : BasisType::GaussLobatto),
fes(&mesh, &fec, VDIM, VDIM == 3 ? Ordering::byVDIM : Ordering::byNODES),
geom_type(mesh.GetTypicalElementGeometry()),
irs(0, GLL ? Quadrature1D::GaussLobatto : Quadrature1D::GaussLegendre),
ir(&irs.Get(geom_type, q)),
ir(SIMPLICES
? &StroudIntRules.Get(geom_type, q)
: &irs.Get(geom_type, q)),
ir_rhs(&IntRules.Get(geom_type, 2*p)),
one(1.0),
uvec(DIM),
unit_vec((uvec = 1.0, uvec /= uvec.Norml2(), uvec)),
@@ -135,7 +148,7 @@ struct BakeOff
{
bfi = new VectorMassIntegrator(one, ir);
}
else if constexpr (BFI == 3 || BFI == 5)
else if constexpr (BFI == 3 || BFI == 5 || BFI == 7)
{
bfi = new DiffusionIntegrator(one, ir);
}
@@ -158,8 +171,8 @@ struct BakeOff
};
// Bake-off Problems (BPs)
template <int BFI, int VDIM, bool GLL>
struct BP : public BakeOff<BFI, VDIM, GLL>
template <int BFI, int VDIM, bool GLL, bool SIMPLICES>
struct BP : public BakeOff<BFI, VDIM, GLL, SIMPLICES>
{
const int max_it = 32, print_lvl = -1;
@@ -170,9 +183,9 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
Vector B, X;
CGSolver cg;
using base = BakeOff<BFI, VDIM, GLL>;
using base = BakeOff<BFI, VDIM, GLL, SIMPLICES>;
using base::a;
using base::ir;
using base::ir_rhs;
using base::one;
using base::mesh;
using base::fes;
@@ -191,11 +204,11 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
if constexpr (VDIM == 1)
{
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.AddDomainIntegrator(new DomainLFIntegrator(one, ir_rhs));
}
else
{
b.AddDomainIntegrator(new VectorDomainLFIntegrator(unit_vec));
b.AddDomainIntegrator(new VectorDomainLFIntegrator(unit_vec, ir_rhs));
}
b.UseFastAssembly(true);
b.Assemble();
@@ -213,6 +226,12 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
cg.SetRelTol(1e-8);
cg.Mult(B, X);
MFEM_VERIFY(cg.GetConverged(), "CG solver did not converge!");
if constexpr (base::visualization)
{
a.RecoverFEMSolution(X, b, x);
socketstream glvis("localhost", 19916);
glvis << "solution\n" << mesh << x << std::flush;
}
}
cg.SetRelTol(0.0);
cg.SetMaxIter(max_it);
@@ -231,12 +250,12 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
};
// Bake-off Kernels (BKs)
template <int BFI, int VDIM, bool GLL>
struct BK : public BakeOff<BFI, VDIM, GLL>
template <int BFI, int VDIM, bool GLL, bool SIMPLICES>
struct BK : public BakeOff<BFI, VDIM, GLL, SIMPLICES>
{
Vector xe, ye;
using base = BakeOff<BFI, VDIM, GLL>;
using base = BakeOff<BFI, VDIM, GLL, SIMPLICES>;
using base::ir;
using base::one;
using base::bfi;
@@ -282,47 +301,56 @@ static void Benchmark(bm::State& state) noexcept
state.counters["Dofs"] = bm::Counter(run.dofs);
state.counters["MDof/s"] = bm::Counter(run.SumMdofs(), bm::Counter::kIsRate);
state.counters["Order"] = bm::Counter(state.range(0));
state.counters["Simplices"] = bm::Counter(run.Simplices);
}
#define REGISTER(PK, BFI, VDIM, GLL) \
BENCHMARK_TEMPLATE(Benchmark, PK<BFI, VDIM, GLL>) \
->Name(#PK #BFI)->Apply(CustomArguments)->Unit(bm::kMillisecond)
#define MAKE_NAME_false(PK, BFI) #PK #BFI
#define MAKE_NAME_true(PK, BFI) #PK #BFI "tet"
#define MAKE_NAME(PK, BFI, SIMPLICES) MAKE_NAME_ ## SIMPLICES (PK, BFI)
#define REGISTER(PK, BFI, VDIM, GLL, SIMPLICES) \
BENCHMARK_TEMPLATE(Benchmark, PK<BFI, VDIM, GLL, SIMPLICES>) \
->Name(MAKE_NAME(PK, BFI, SIMPLICES))->Apply(CustomArguments)->Unit(bm::kMillisecond)
// BP1: scalar PCG with mass matrix, q=p+2
REGISTER(BP, 1, 1, false);
REGISTER(BP, 1, 1, false, false); // hex
REGISTER(BP, 1, 1, false, true); // tet
// BP2: vector PCG with mass matrix, q=p+2
REGISTER(BP, 2, 3, false);
REGISTER(BP, 2, 3, false, false);
// BP3: scalar PCG with stiffness matrix, q=p+2
REGISTER(BP, 3, 1, false);
REGISTER(BP, 3, 1, false, false); // hex
REGISTER(BP, 3, 1, false, true); // tet
// BP4: vector PCG with stiffness matrix, q=p+2
REGISTER(BP, 4, 3, false);
REGISTER(BP, 4, 3, false, false);
// BP5: scalar PCG with stiffness matrix, q=p+1
REGISTER(BP, 5, 1, true);
REGISTER(BP, 5, 1, true, false); // hex
REGISTER(BP, 5, 1, true, true); // tet
REGISTER(BP, 7, 1, true, true); // tet
// BP6: vector PCG with stiffness matrix, q=p+1
REGISTER(BP, 6, 3, true);
REGISTER(BP, 6, 3, true, false);
// BK1: scalar E-vector-to-E-vector evaluation of mass matrix, q=p+2
REGISTER(BK, 1, 1, false);
REGISTER(BK, 1, 1, false, false);
// BK2: vector E-vector-to-E-vector evaluation of mass matrix, q=p+2
REGISTER(BK, 2, 3, false);
REGISTER(BK, 2, 3, false, false);
// BK3: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
REGISTER(BK, 3, 1, false);
REGISTER(BK, 3, 1, false, false);
// BK4: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+2
REGISTER(BK, 4, 3, false);
REGISTER(BK, 4, 3, false, false);
// BK5: scalar E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
REGISTER(BK, 5, 1, true);
REGISTER(BK, 5, 1, true, false);
// BK6: vector E-vector-to-E-vector evaluation of stiffness matrix, q=p+1
REGISTER(BK, 6, 3, true);
REGISTER(BK, 6, 3, true, false);
/**
* @brief CEED Bake-off Problems main entry point
+1
View File
@@ -147,6 +147,7 @@ set(UNIT_TESTS_SRCS
fem/test_pa_grad.cpp
fem/test_pa_idinterp.cpp
fem/test_pa_kernels.cpp
fem/test_pa_simplices.cpp
fem/test_particleset.cpp
fem/test_pgridfunc_save_serial.cpp
fem/test_poly1d.cpp
+117
View File
@@ -169,6 +169,44 @@ TEST_CASE("Integration rule weights",
REQUIRE(Geometry::Volume[geom] == MFEM_Approx(weight_sum));
}
// Test the Gauss-Jacobi rules over a range of alpha and beta. The n-point rule is
// exact for integrands of the form
// x^beta * (1-x)^alpha * p(x),
// where p(x) is a degree 2*n-1 polynomial. We test monomials up to degree 2*n-1 here,
// meaning that the exact integral is given by Beta(beta+2*n, alpha+1).
TEST_CASE("Gauss-Jacobi integration rules", "[GaussJacobiRules]")
{
const auto alpha = GENERATE(-0.25, 0.0, 0.25, 0.5, 0.75, 1.0, 1.25, 1.5, 1.75,
2.0, 2.25, 2.5, 2.75, 3.0, 3.25, 3.5, 3.75, 4.0);
const auto beta = GENERATE(-0.25, 0.0, 0.25, 0.5, 0.75, 1.0, 1.25, 1.5, 1.75,
2.0, 2.25, 2.5, 2.75, 3.0, 3.25, 3.5, 3.75, 4.0);
for (int np = 1; np <= 50; np++)
{
const int p = 2*np - 1;
IntegrationRule ir_a_b;
QuadratureFunctions1D::GaussJacobi(np, alpha, beta, &ir_a_b);
// Gauss-Jacobi rule (alpha,beta) is exact up to polynomials of degree 2*np-1
for (int n = 0; n <= p; n++)
{
double integral = 0.0;
for (int i = 0; i < ir_a_b.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir_a_b.IntPoint(i);
integral += ip.weight * pow(ip.x, n);
}
double exact = std::tgamma(beta+n+1) * std::tgamma(alpha+1) / std::tgamma(
alpha+n+2+beta);
// use tgamma instead of beta for compliance with C++11 standard
double relerr = 1. - integral/exact;
INFO("p=" << n << ", alpha=" << alpha << ", beta=" << beta);
REQUIRE(fabs(relerr) < 1e-11);
}
}
}
double poly2d(const IntegrationPoint &ip, int m, int n)
{
return pow(ip.x, m)*pow(ip.y, n);
@@ -272,6 +310,85 @@ TEST_CASE("Simplex integration rules", "[SimplexRules]")
}
}
TEST_CASE("Stroud conical quadrature rules in a simplex",
"[SimplexRules][StroudRules]")
{
const int maxn = 32;
int binom[maxn+1][maxn+1];
for (int n = 0; n <= maxn; n++)
{
binom[n][0] = binom[n][n] = 1;
for (int k = 1; k < n; k++)
{
binom[n][k] = binom[n-1][k] + binom[n-1][k-1];
}
}
SECTION("low triangle integration error on reference element for f=x^m y^n, where m+n <= p")
{
for (int order = 0; order <= 25; order++)
{
const IntegrationRule &ir = StroudIntRules.Get(Geometry::TRIANGLE, order);
// using the monomial basis: x^m y^n, 0 <= m+n <= order
for (int p = 0; p <= order; p++)
{
for (int m = p; m >= 0; m--)
{
int n = p - m;
double integral = 0.0;
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
integral += ip.weight*poly2d(ip, m, n);
}
double exact = 1.0/binom[p][m]/(p + 1)/(p + 2);
double relerr = 1. - integral/exact;
// If a test fails any INFO statements preceding the REQUIRE are displayed
INFO("p=" << p << ", m=" << m << ", n=" << n);
REQUIRE(fabs(relerr) < 1e-11);
}
}
}
}
SECTION("low tet integration error on reference element for f=x^l y^m z^n, where l+m+n <= p")
{
for (int order = 0; order <= 21; order++)
{
const IntegrationRule &ir = StroudIntRules.Get(Geometry::TETRAHEDRON, order);
for (int p = 0; p <= order; p++)
{
for (int l = p; l >= 0; l--)
{
for (int m = p - l; m >= 0; m--)
{
int n = p - l - m;
double integral = 0.0;
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
integral += ip.weight*poly3d(ip, l, m, n);
}
double exact = 1.0/binom[p][l+m]/binom[l+m][l]/(p+1)/(p+2)/(p+3);
double relerr = 1. - integral/exact;
// If a test fails any INFO statements preceding the REQUIRE are displayed
INFO("p=" << p << ", l=" << l << ", m=" << m << ", n=" << n);
REQUIRE(fabs(relerr) < 1e-11);
}
}
}
}
}
}
// Monomial exactness is tested by [SimplexRules] above, which now uses
// positive-weight rules by default. The tests below verify properties
+134
View File
@@ -0,0 +1,134 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifdef _WIN32
#define _USE_MATH_DEFINES
#include <cmath>
#endif
#include "unit_tests.hpp"
#include "mfem.hpp"
using namespace mfem;
namespace pa_kernels
{
void test_pa_simplices(const char *filename, int p)
{
CAPTURE(filename, p);
Mesh mesh(filename);
if (mesh.GetTypicalElementGeometry() == Geometry::SQUARE ||
mesh.GetTypicalElementGeometry() == Geometry::CUBE)
{
mesh = Mesh::MakeSimplicial(mesh);
}
const int dim = mesh.Dimension();
MFEM_VERIFY(!mesh.IsMixedMesh(), "Mesh is mixed");
H1_FECollection fec(p, dim, BasisType::Positive);
FiniteElementSpace fes(&mesh, &fec);
GridFunction x(&fes), y_fa(&fes), y_pa(&fes);
x.Randomize(0x100001b3);
y_fa.Randomize(0x9e3779b9);
y_pa = y_fa;
const auto &fe = *fes.GetTypicalFE();
const auto &Tr = *mesh.GetTypicalElementTransformation();
const auto order = 2 * fe.GetOrder() + Tr.OrderW();
const auto *ir = &StroudIntRules.Get(fe.GetGeomType(), order);
const auto *ir1 = &StroudIntRules.Get(fe.GetGeomType(), 2*fe.GetOrder()-1);
ConstantCoefficient const_coeff(M_2_SQRTPI);
FunctionCoefficient funct_coeff([](const Vector &x)
{ return M_1_PI + x[0] * x[0]; });
BilinearForm fa(&fes), pa(&fes);
fa.AddDomainIntegrator(new MassIntegrator(ir));
fa.AddDomainIntegrator(new MassIntegrator(ir));
fa.AddDomainIntegrator(new MassIntegrator(const_coeff, ir));
fa.AddDomainIntegrator(new MassIntegrator(funct_coeff, ir));
fa.AddDomainIntegrator(new DiffusionIntegrator(ir1));
fa.AddDomainIntegrator(new DiffusionIntegrator(ir));
fa.AddDomainIntegrator(new DiffusionIntegrator(const_coeff, ir));
fa.AddDomainIntegrator(new DiffusionIntegrator(funct_coeff, ir));
fa.Assemble();
fa.Finalize();
pa.AddDomainIntegrator(new MassIntegrator());
pa.AddDomainIntegrator(new MassIntegrator(ir));
pa.AddDomainIntegrator(new MassIntegrator(const_coeff, ir));
pa.AddDomainIntegrator(new MassIntegrator(funct_coeff, ir));
pa.AddDomainIntegrator(new DiffusionIntegrator());
pa.AddDomainIntegrator(new DiffusionIntegrator(ir));
pa.AddDomainIntegrator(new DiffusionIntegrator(const_coeff, ir));
pa.AddDomainIntegrator(new DiffusionIntegrator(funct_coeff, ir));
pa.SetAssemblyLevel(AssemblyLevel::PARTIAL);
pa.Assemble();
fa.Mult(x, y_fa);
pa.Mult(x, y_pa);
y_fa -= y_pa;
REQUIRE(y_fa.Norml2() == MFEM_Approx(0.0));
}
TEST_CASE("PA Simplices", "[PartialAssembly][Simplices][GPU]")
{
const auto all_tests = launch_all_non_regression_tests;
const auto p = !all_tests ? GENERATE(1, 2) : GENERATE(1, 2, 3, 4);
const auto GenMesh = [&](const auto &meshs, const auto &extra)
{
return !all_tests
? GENERATE_REF(from_range(meshs))
: GENERATE_REF(from_range(meshs), from_range(extra));
};
SECTION("2D")
{
auto meshs = { "../../data/beam-tri.mesh",
"../../data/inline-tri.mesh",
"../../data/ref-triangle.mesh",
"../../data/rt-2d-p4-tri.mesh",
"../../data/square-disc-p2.mesh",
"../../data/square-disc-p3.mesh",
"../../data/periodic-annulus-sector.msh"
};
auto extra = { "../../data/star-q2.mesh",
"../../data/star-q3.mesh",
"../../data/inline-quad.mesh",
"../../data/klein-donut.mesh",
"../../data/fichera-quad.mesh",
"../../data/periodic-square.mesh"
};
test_pa_simplices(GenMesh(meshs, extra), p);
}
SECTION("3D")
{
auto meshs = { "../../data/beam-tet.mesh",
"../../data/inline-tet.mesh",
"../../data/ref-tetrahedron.mesh"
};
auto extra = { "../../data/escher.mesh",
"../../data/escher-p2.mesh",
"../../data/inline-hex.mesh",
"../../data/fichera-q2.mesh",
"../../data/periodic-cube.mesh"
};
test_pa_simplices(GenMesh(meshs, extra), p);
}
}
} // namespace pa_kernels
+436 -19
View File
@@ -9,8 +9,10 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "unit_tests.hpp"
#include "mfem.hpp"
#include "unit_tests.hpp"
#include <memory>
using namespace mfem;
@@ -75,14 +77,15 @@ void vectorcoeff(const Vector& x, Vector& y)
}
}
enum class VecSpace { H1, VectorH1, ND, RT };
enum class VecSpace { H1, VectorH1nodes, VectorH1vdim, ND, RT };
std::string VecSpaceName(VecSpace vectorspace)
{
switch (vectorspace)
{
case VecSpace::H1: return "H1";
case VecSpace::VectorH1: return "Vector H1";
case VecSpace::VectorH1nodes: return "Vector H1 by nodes";
case VecSpace::VectorH1vdim: return "Vector H1 by vdim";
case VecSpace::ND: return "Nedelec";
case VecSpace::RT: return "Raviart-Thomas";
}
@@ -91,8 +94,8 @@ std::string VecSpaceName(VecSpace vectorspace)
TEST_CASE("Transfer", "[Transfer]")
{
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1, VecSpace::ND,
VecSpace::RT);
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim, VecSpace::ND, VecSpace::RT);
auto geometric = GENERATE(true, false);
auto simplex = GENERATE(true, false);
dimension = GENERATE(2, 3);
@@ -124,7 +127,8 @@ TEST_CASE("Transfer", "[Transfer]")
switch (vectorspace)
{
case VecSpace::H1:
case VecSpace::VectorH1:
case VecSpace::VectorH1nodes:
case VecSpace::VectorH1vdim:
c_fec = new H1_FECollection(order, dimension);
f_fec = geometric ? c_fec : new H1_FECollection(fineOrder, dimension);
break;
@@ -144,12 +148,14 @@ TEST_CASE("Transfer", "[Transfer]")
fineMesh.UniformRefinement();
}
const int vdim = (vectorspace == VecSpace::VectorH1) ? dimension : 1;
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
FiniteElementSpace *c_fespace =
new FiniteElementSpace(&mesh, c_fec, vdim);
new FiniteElementSpace(&mesh, c_fec, vdim, ordering);
FiniteElementSpace *f_fespace =
new FiniteElementSpace(&fineMesh, f_fec, vdim);
new FiniteElementSpace(&fineMesh, f_fec, vdim, ordering);
Operator* referenceOperator = nullptr;
@@ -217,7 +223,8 @@ TEST_CASE("Transfer", "[Transfer]")
TEST_CASE("Variable Order Transfer", "[Transfer][VariableOrder]")
{
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1, VecSpace::ND,
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim, VecSpace::ND,
VecSpace::RT);
dimension = GENERATE(2, 3);
@@ -244,7 +251,8 @@ TEST_CASE("Variable Order Transfer", "[Transfer][VariableOrder]")
switch (vectorspace)
{
case VecSpace::H1:
case VecSpace::VectorH1:
case VecSpace::VectorH1nodes:
case VecSpace::VectorH1vdim:
c_fec = new H1_FECollection(order, dimension);
f_fec = new H1_FECollection(order, dimension);
break;
@@ -261,12 +269,15 @@ TEST_CASE("Variable Order Transfer", "[Transfer][VariableOrder]")
mesh.EnsureNCMesh();
mesh.RandomRefinement(0.5);
const int vdim = (vectorspace == VecSpace::VectorH1) ? dimension : 1;
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
FiniteElementSpace *c_fespace =
new FiniteElementSpace(&mesh, c_fec, vdim);
new FiniteElementSpace(&mesh, c_fec, vdim, ordering);
FiniteElementSpace *f_fespace =
new FiniteElementSpace(&mesh, f_fec, vdim);
new FiniteElementSpace(&mesh, f_fec, vdim, ordering);
RandomPRefinement(*f_fespace);
@@ -322,7 +333,8 @@ TEST_CASE("Variable Order Transfer", "[Transfer][VariableOrder]")
TEST_CASE("Variable Order True Transfer", "[Transfer][VariableOrder]")
{
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1);
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim);
dimension = GENERATE(2, 3);
int ne = 2;
@@ -348,12 +360,15 @@ TEST_CASE("Variable Order True Transfer", "[Transfer][VariableOrder]")
f_fec = new H1_FECollection(order, dimension);
mesh.EnsureNCMesh();
mesh.RandomRefinement(0.5);
const int vdim = (vectorspace == VecSpace::VectorH1) ? dimension : 1;
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
FiniteElementSpace *c_fespace =
new FiniteElementSpace(&mesh, c_fec, vdim);
new FiniteElementSpace(&mesh, c_fec, vdim, ordering);
FiniteElementSpace *f_fespace =
new FiniteElementSpace(&mesh, f_fec, vdim);
new FiniteElementSpace(&mesh, f_fec, vdim, ordering);
RandomPRefinement(*f_fespace);
@@ -425,6 +440,99 @@ TEST_CASE("Variable Order True Transfer", "[Transfer][VariableOrder]")
delete c_fec;
}
TEST_CASE("H1 L2 transfer with consistent mass", "[Transfer]")
{
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim);
dimension = GENERATE(2, 3);
const int order = 2;
const int ne = 2;
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
CAPTURE(VecSpaceName(vectorspace), dimension, order);
Mesh mesh;
if (dimension == 2)
{
mesh = Mesh::MakeCartesian2D(ne, ne, Element::QUADRILATERAL,
1, 1.0, 1.0);
}
else
{
mesh = Mesh::MakeCartesian3D(ne, ne, ne, Element::HEXAHEDRON,
1.0, 1.0, 1.0);
}
Mesh fineMesh(mesh);
fineMesh.UniformRefinement();
H1_FECollection fec(order, dimension);
FiniteElementSpace c_fespace(&mesh, &fec, vdim, ordering);
FiniteElementSpace f_fespace(&fineMesh, &fec, vdim, ordering);
L2ProjectionGridTransfer transfer(c_fespace, f_fespace);
transfer.UseConsistentMass();
const Operator &R = transfer.ForwardOperator();
GridFunction X(&c_fespace);
GridFunction Y(&f_fespace);
GridFunction Y_ref(&f_fespace);
coeff_order = 1;
LinearForm rhs(&f_fespace);
BilinearForm mass(&f_fespace);
FunctionCoefficient funcCoeff(&coeff);
VectorFunctionCoefficient vecCoeff(dimension, &vectorcoeff);
if (vectorspace == VecSpace::H1)
{
X.ProjectCoefficient(funcCoeff);
rhs.AddDomainIntegrator(new DomainLFIntegrator(funcCoeff));
mass.AddDomainIntegrator(new MassIntegrator);
}
else
{
X.ProjectCoefficient(vecCoeff);
rhs.AddDomainIntegrator(new VectorDomainLFIntegrator(vecCoeff));
mass.AddDomainIntegrator(new VectorMassIntegrator);
}
rhs.Assemble();
mass.Assemble();
SparseMatrix M;
Array<int> empty;
mass.FormSystemMatrix(empty, M);
GSSmoother M_prec(M);
Y_ref = 0.0;
PCG(M, M_prec, rhs, Y_ref, 0, 500, 1e-24, 0.0);
Y = 0.0;
R.Mult(X, Y);
Y -= Y_ref;
REQUIRE(Y.Norml2() < 1e-11 * Y_ref.Norml2());
Vector x(c_fespace.GetVSize());
Vector y(f_fespace.GetVSize());
Vector Ry(f_fespace.GetVSize());
Vector Rtx(c_fespace.GetVSize());
x.Randomize(1);
y.Randomize(2);
R.Mult(x, Ry);
R.MultTranspose(y, Rtx);
const real_t ip1 = InnerProduct(Ry, y);
const real_t ip2 = InnerProduct(x, Rtx);
REQUIRE(std::abs(ip1 - ip2) <
1e-10 * std::max(std::abs(ip1), std::abs(ip2)));
REQUIRE_FALSE(transfer.SupportsBackwardsOperator());
}
TEST_CASE("Restriction Transpose Operator")
{
int order = GENERATE(1, 2);
@@ -460,6 +568,204 @@ TEST_CASE("Restriction Transpose Operator")
REQUIRE(y3.Normlinf() == MFEM_Approx(0.0));
}
real_t sin_func(const Vector &x)
{
return sin(M_PI * x.Sum());
}
void sin_vfunc(const Vector &x, Vector &y)
{
y.SetSize(x.Size());
for (int i = 0; i < y.Size(); i++)
{
y(i) = sin(M_PI * x[i]);
}
}
namespace
{
struct TraceCollections
{
std::unique_ptr<FiniteElementCollection> c_fec;
std::unique_ptr<FiniteElementCollection> f_fec;
std::unique_ptr<FiniteElementCollection> c_trace_fec;
std::unique_ptr<FiniteElementCollection> f_trace_fec;
};
TraceCollections MakeTraceCollections(const VecSpace vectorspace,
const int order, const int dim)
{
TraceCollections fec;
switch (vectorspace)
{
case VecSpace::H1:
case VecSpace::VectorH1nodes:
case VecSpace::VectorH1vdim:
fec.c_fec = std::make_unique<H1_FECollection>(order, dim);
fec.f_fec = std::make_unique<H1_FECollection>(order+1, dim);
fec.c_trace_fec = std::make_unique<H1_Trace_FECollection>(order, dim);
fec.f_trace_fec = std::make_unique<H1_Trace_FECollection>(order+1, dim);
break;
case VecSpace::ND:
fec.c_fec = std::make_unique<ND_FECollection>(order, dim);
fec.f_fec = std::make_unique<ND_FECollection>(order+1, dim);
fec.c_trace_fec = std::make_unique<ND_Trace_FECollection>(order, dim);
fec.f_trace_fec = std::make_unique<ND_Trace_FECollection>(order+1, dim);
break;
case VecSpace::RT:
fec.c_fec = std::make_unique<RT_FECollection>(order-1, dim);
fec.f_fec = std::make_unique<RT_FECollection>(order, dim);
fec.c_trace_fec = std::make_unique<RT_Trace_FECollection>(order-1, dim);
fec.f_trace_fec = std::make_unique<RT_Trace_FECollection>(order, dim);
break;
}
return fec;
}
template <typename MeshT, typename FESpaceT, typename GridFunctionT>
void CheckTracePRefinementTrueTransfer(MeshT &mesh,
FESpaceT &c_fes, FESpaceT &f_fes,
FESpaceT &c_trace_fes, FESpaceT &f_trace_fes,
const VecSpace vectorspace,
const int dim,
const bool assembleP)
{
GridFunctionT x_c(&c_fes); x_c = 0.0;
GridFunctionT x_f(&f_fes); x_f = 0.0;
GridFunctionT x_trace_c(&c_trace_fes); x_trace_c = 0.0;
GridFunctionT x_trace_f(&f_trace_fes); x_trace_f = 0.0;
if (vectorspace == VecSpace::H1)
{
FunctionCoefficient cf(sin_func);
x_c.ProjectCoefficient(cf);
x_trace_c.ProjectTraceCoefficient(cf);
}
else
{
VectorFunctionCoefficient vec_cf(dim, &sin_vfunc);
x_c.ProjectCoefficient(vec_cf);
switch (vectorspace)
{
case VecSpace::VectorH1nodes:
case VecSpace::VectorH1vdim:
x_trace_c.ProjectTraceCoefficient(vec_cf);
break;
case VecSpace::ND:
x_trace_c.ProjectTraceCoefficientTangent(vec_cf);
break;
case VecSpace::RT:
x_trace_c.ProjectTraceCoefficientNormal(vec_cf);
break;
default:
break;
}
}
// Generate transfer operators for field and trace spaces
PRefinementTransferOperator P(c_fes, f_fes, assembleP);
PRefinementTransferOperator P_trace(c_trace_fes, f_trace_fes, assembleP);
Vector x_c_true(c_fes.GetTrueVSize());
Vector x_f_true(f_fes.GetTrueVSize());
Vector x_trace_c_true(c_trace_fes.GetTrueVSize());
Vector x_trace_f_true(f_trace_fes.GetTrueVSize());
x_c.GetTrueDofs(x_c_true);
x_trace_c.GetTrueDofs(x_trace_c_true);
P.GetTrueTransferOperator()->Mult(x_c_true, x_f_true);
P_trace.GetTrueTransferOperator()->Mult(x_trace_c_true, x_trace_f_true);
x_f.SetFromTrueDofs(x_f_true);
x_trace_f.SetFromTrueDofs(x_trace_f_true);
// zero out interior dofs before and after p-ref to compare with trace
Array<int> vdofs;
for (int i = 0; i < mesh.GetNE(); i++)
{
c_fes.GetElementInteriorVDofs(i, vdofs);
x_c.SetSubVector(vdofs, 0.0);
f_fes.GetElementInteriorVDofs(i, vdofs);
x_f.SetSubVector(vdofs, 0.0);
}
// Embed the trace dofs to a field GridFunction for comparison
Array<int> face_vdofs, trace_vdofs;
Vector values;
GridFunctionT x_embedded_trace_c(&c_fes); x_embedded_trace_c = 0.0;
GridFunctionT x_embedded_trace_f(&f_fes); x_embedded_trace_f = 0.0;
for (int i = 0; i < mesh.GetNumFaces(); i++)
{
c_trace_fes.GetFaceVDofs(i, trace_vdofs);
x_trace_c.GetSubVector(trace_vdofs, values);
c_fes.GetFaceVDofs(i, face_vdofs);
x_embedded_trace_c.SetSubVector(face_vdofs, values);
f_trace_fes.GetFaceVDofs(i, trace_vdofs);
x_trace_f.GetSubVector(trace_vdofs, values);
f_fes.GetFaceVDofs(i, face_vdofs);
x_embedded_trace_f.SetSubVector(face_vdofs, values);
}
x_embedded_trace_c -= x_c;
REQUIRE(x_embedded_trace_c.Norml2() == MFEM_Approx(0.0));
x_embedded_trace_f -= x_f;
REQUIRE(x_embedded_trace_f.Norml2() == MFEM_Approx(0.0));
}
} // namespace
TEST_CASE("Trace PRefinement Serial TrueTransfer", "[Transfer]")
{
auto simplex = GENERATE(true, false);
dimension = GENERATE(2, 3);
constexpr int ne = 4;
auto order = GENERATE(1,2,3);
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim,
VecSpace::ND,VecSpace::RT);
auto assembleP = GENERATE(false, true);
auto amr = GENERATE(false, true);
// Log test case information
const int total_ne = static_cast<int>(std::pow(ne, dimension));
CAPTURE(VecSpaceName(vectorspace),dimension, simplex, total_ne, order,
assembleP);
Mesh mesh;
if (dimension == 2)
{
Element::Type type = simplex ? Element::TRIANGLE : Element::QUADRILATERAL;
mesh = Mesh::MakeCartesian2D(ne, ne, type, 1, 1.0, 1.0);
}
else
{
Element::Type type = simplex ? Element::TETRAHEDRON : Element::HEXAHEDRON;
mesh = Mesh::MakeCartesian3D(ne, ne, ne, type, 1.0, 1.0, 1.0);
}
if (amr) { mesh.RandomRefinement(0.5); }
auto fec = MakeTraceCollections(vectorspace, order, dimension);
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
FiniteElementSpace c_fes(&mesh, fec.c_fec.get(), vdim, ordering);
FiniteElementSpace f_fes(&mesh, fec.f_fec.get(), vdim, ordering);
FiniteElementSpace c_trace_fes(&mesh, fec.c_trace_fec.get(), vdim, ordering);
FiniteElementSpace f_trace_fes(&mesh, fec.f_trace_fec.get(), vdim, ordering);
CheckTracePRefinementTrueTransfer<Mesh, FiniteElementSpace, GridFunction>(
mesh, c_fes, f_fes, c_trace_fes, f_trace_fes, vectorspace, dimension,
assembleP);
}
#ifdef MFEM_USE_MPI
TEST_CASE("Parallel Transfer", "[Transfer][Parallel]")
@@ -584,4 +890,115 @@ TEST_CASE("Parallel Transfer", "[Transfer][Parallel]")
delete pmesh;
}
TEST_CASE("Parallel H1 L2 transfer with consistent mass",
"[Transfer][Parallel]")
{
dimension = GENERATE(2, 3);
const int order = 2;
const int ne = 2;
const int vdim = 1;
CAPTURE(dimension, order);
Mesh mesh;
if (dimension == 2)
{
mesh = Mesh::MakeCartesian2D(ne, ne, Element::QUADRILATERAL,
1, 1.0, 1.0);
}
else
{
mesh = Mesh::MakeCartesian3D(ne, ne, ne, Element::HEXAHEDRON,
1.0, 1.0, 1.0);
}
ParMesh pmesh(MPI_COMM_WORLD, mesh);
ParMesh pfineMesh(MPI_COMM_WORLD, mesh);
pfineMesh.UniformRefinement();
H1_FECollection fec(order, dimension);
ParFiniteElementSpace c_fespace(&pmesh, &fec, vdim);
ParFiniteElementSpace f_fespace(&pfineMesh, &fec, vdim);
L2ProjectionGridTransfer transfer(c_fespace, f_fespace);
transfer.UseConsistentMass();
const Operator &R = transfer.TrueForwardOperator();
Vector x(c_fespace.GetTrueVSize());
Vector y(f_fespace.GetTrueVSize());
Vector Rx(f_fespace.GetTrueVSize());
Vector Rty(c_fespace.GetTrueVSize());
x.Randomize(1);
y.Randomize(2);
R.Mult(x, Rx);
R.MultTranspose(y, Rty);
const real_t ip1 = InnerProduct(MPI_COMM_WORLD, Rx, y);
const real_t ip2 = InnerProduct(MPI_COMM_WORLD, x, Rty);
REQUIRE(std::abs(ip1 - ip2) <
1e-10 * std::max(std::abs(ip1), std::abs(ip2)));
REQUIRE_FALSE(transfer.SupportsBackwardsOperator());
}
TEST_CASE("Trace PRefinement Parallel TrueTransfer", "[Transfer][Parallel]")
{
auto simplex = GENERATE(true, false);
dimension = GENERATE(2, 3);
constexpr int ne = 4;
auto order = GENERATE(1,2,3);
auto vectorspace = GENERATE(VecSpace::H1, VecSpace::VectorH1nodes,
VecSpace::VectorH1vdim,
VecSpace::ND,VecSpace::RT);
auto assembleP = GENERATE(true, false);
auto amr = GENERATE(true, false);
// Log test case information
const int total_ne = static_cast<int>(std::pow(ne, dimension));
CAPTURE(VecSpaceName(vectorspace),dimension, simplex, total_ne, order,
assembleP);
Mesh mesh;
if (dimension == 2)
{
Element::Type type = simplex ? Element::TRIANGLE : Element::QUADRILATERAL;
mesh = Mesh::MakeCartesian2D(ne, ne, type, 1, 1.0, 1.0);
}
else
{
Element::Type type = simplex ? Element::TETRAHEDRON : Element::HEXAHEDRON;
mesh = Mesh::MakeCartesian3D(ne, ne, ne, type, 1.0, 1.0, 1.0);
}
if (amr) { mesh.EnsureNCMesh(true); }
ParMesh pmesh(MPI_COMM_WORLD, mesh);
if (amr) { pmesh.RandomRefinement(0.5); }
auto fec = MakeTraceCollections(vectorspace, order, dimension);
const int vdim = (vectorspace == VecSpace::VectorH1nodes
|| vectorspace == VecSpace::VectorH1vdim) ? dimension : 1;
Ordering::Type ordering = (vectorspace == VecSpace::VectorH1vdim)
? Ordering::byVDIM : Ordering::byNODES;
ParFiniteElementSpace c_fes(&pmesh, fec.c_fec.get(), vdim, ordering);
ParFiniteElementSpace f_fes(&pmesh, fec.f_fec.get(), vdim, ordering);
ParFiniteElementSpace c_trace_fes(&pmesh, fec.c_trace_fec.get(), vdim,
ordering);
ParFiniteElementSpace f_trace_fes(&pmesh, fec.f_trace_fec.get(), vdim,
ordering);
CheckTracePRefinementTrueTransfer<ParMesh, ParFiniteElementSpace, ParGridFunction>
(
pmesh, c_fes, f_fes, c_trace_fes, f_trace_fes, vectorspace, dimension,
assembleP);
}
#endif
+51
View File
@@ -146,6 +146,57 @@ TEST_CASE("Gecko integration in MFEM", "[Mesh]")
}
}
TEST_CASE("Hilbert reordering boundary element consistency", "[Mesh]")
{
// After ReorderElements the boundary[] array must be sorted by the index of
// the adjacent interior element (faces_info[be_to_face[i]].Elem1No).
// This ensures spatial locality between volume and boundary elements.
auto test = [](Mesh & mesh)
{
// Record the number of boundary elements before reordering
const int nbe = mesh.GetNBE();
REQUIRE(nbe > 0);
Array<int> perm;
mesh.GetHilbertElementOrdering(perm);
mesh.ReorderElements(perm);
REQUIRE(mesh.GetNBE() == nbe); // count must not change
// Adjacent element indices must be non-decreasing across the boundary
// element list.
for (int i = 0; i < mesh.GetNBE() - 1; ++i)
{
int fi, fj, o;
mesh.GetBdrElementFace(i, &fi, &o);
mesh.GetBdrElementFace(i + 1, &fj, &o);
int eli, elj, dummy;
mesh.GetFaceElements(fi, &eli, &dummy);
mesh.GetFaceElements(fj, &elj, &dummy);
REQUIRE(eli <= elj);
}
};
SECTION("3D hex mesh boundary elements sorted after Hilbert reordering")
{
Mesh mesh = Mesh::MakeCartesian3D(3, 4, 5, Element::HEXAHEDRON);
test(mesh);
}
SECTION("2D quad mesh boundary elements sorted after Hilbert reordering")
{
Mesh mesh = Mesh::MakeCartesian2D(4, 5, Element::QUADRILATERAL);
test(mesh);
}
SECTION("3D tet mesh boundary elements sorted after Hilbert reordering")
{
Mesh mesh = Mesh::MakeCartesian3D(3, 4, 5, Element::TETRAHEDRON);
test(mesh);
}
}
TEST_CASE("MakeSimplicial", "[Mesh]")
{
auto mesh_fname = GENERATE("../../data/star.mesh",