Compare commits

..
Author SHA1 Message Date
psocratis ef7a33b64d complex-jacobi minor cleanup 2020-07-30 17:01:53 -07:00
psocratis abaa73e4dc Starting ComplexOperatorJacobiSmoother 2020-07-29 14:41:31 -07:00
stefanhenneking c346d4601d Updating changelog. 2020-07-29 12:33:18 -05:00
stefanhenneking d1b2b6eabf ex25p working with cuda. 2020-07-29 12:11:16 -05:00
stefanhenneking fd45550d7d minor 2020-07-29 12:09:58 -05:00
stefanhenneking 602f9522be Adding PA and device option to ex25p (not yet cuda tested) 2020-07-29 11:21:17 -05:00
stefanhenneking f02d161457 minor 2020-07-29 10:53:43 -05:00
stefanhenneking 8228f99711 Ex25 tested with GPU. 2020-07-29 10:41:47 -05:00
stefanhenneking 8d87e4a93a Merge branch 'curl-curl-coef' of github.com:mfem/mfem into ex25-gpu 2020-07-28 17:16:02 -05:00
Dylan Copeland 79a1aaaa98 Fixing coefficient dimensions in the 2D case. 2020-07-28 17:14:39 -05:00
stefanhenneking 793cf0c173 Minor update to ex25. 2020-07-28 17:13:45 -05:00
stefanhenneking 68e930cc3b Merge branch 'master' of github.com:mfem/mfem into ex25-gpu 2020-07-28 16:08:34 -05:00
stefanhenneking a98ef3ae5b Merge branch 'master' of github.com:mfem/mfem into curl-curl-coef 2020-07-28 15:48:47 -05:00
Dylan Copeland f17d263064 Fixing coefficient dimensions in the 2D case. 2020-07-28 12:51:39 -07:00
stefanhenneking 05b0a7897c Merge branch 'curl-curl-coef' of github.com:mfem/mfem into ex25-gpu 2020-07-28 10:44:45 -05:00
Dylan Copeland 411d3fc96d Implemented vector and matrix coefficients for 3D curl-curl PA integrator, with unit tests. Added the option to specify integration rule, which is necessary for the unit tests. 2020-07-27 16:21:40 -07:00
Tzanio Kolev bba973db73 Merge pull request #1468 from mfem/blockop_cuda
BlockOperator on device
2020-07-27 12:11:58 -07:00
Tzanio Kolev c3394d330e Merge pull request #1587 from flomnes/master
Simplify memory management in example 1
2020-07-27 12:06:30 -07:00
psocratis a817874f12 make style 2020-07-24 16:15:15 -07:00
psocratis 48c9b0f92f Merge branch 'master' into curl-curl-coef 2020-07-24 16:06:45 -07:00
psocratis 5977679b6b fixed typo 2020-07-24 16:06:31 -07:00
psocratis 25ded86cd3 Replaced PMLMatrixCoefficient with PMLDiagMatrixCoefficient (VectorCoefficient) in ex25p.cpp 2020-07-24 15:55:24 -07:00
psocratis 8fa68c42ff Modified ex25 to use VectorCoefficient instead of a MatrixCoefficient 2020-07-24 15:22:40 -07:00
psocratis c5538ff8dc Added Diagonal Matrix Coefficient (VectorCoefficient) in CurlCurlintegrator 2020-07-24 15:09:13 -07:00
Veselin Dobrev 57557ec53b Merge pull request #1141 from mfem/feature/artv3/quad-data
Add QuadratueFunctionCoefficient support in PA assembly [feature/artv3/quad-data]
2020-07-24 14:29:20 -07:00
stefanhenneking 3645f47cc1 minor 2020-07-24 10:40:37 -05:00
stefanhenneking 3da3f275bf Ex25 adding PA and device option (not yet working). 2020-07-24 10:39:42 -05:00
stefanhenneking 58e23e3b2d Merge branch 'matcoefpa' of github.com:mfem/mfem into ex25-gpu 2020-07-23 16:40:46 -05:00
Dylan Copeland e00be4f28e Adding 2D versions of H(curl)-H(div) mixed mass PA operators with support for all coefficient types. 2020-07-23 11:13:01 -07:00
stefanhenneking c922f6926e Minor change in comments. 2020-07-23 12:20:30 -05:00
stefanhenneking 768a689aa5 Merging support for block operator on device into feature branch. 2020-07-23 12:16:53 -05:00
stefanhenneking a16150a436 Minor change to changelog. 2020-07-23 12:13:48 -05:00
Dylan CopelandandTzanio Kolev 2fa920a88b Update CHANGELOG
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2020-07-22 17:34:14 -07:00
Dylan Copeland eb6a7afb9c Adding PA for 3D mass operator with H(div) trial and H(curl) test functions, supporting all coefficient types, and with unit tests. 2020-07-22 13:54:30 -07:00
stefanhenneking 585f9149d1 Merge branch 'master' of github.com:mfem/mfem into blockop_cuda 2020-07-22 14:19:34 -05:00
stefanhenneking b6a3a119a1 Minor change in comments. 2020-07-22 14:18:59 -05:00
Dylan Copeland 95985e9c83 Adding PA for H(curl)-H(div) mass operator with scalar, diagonal vector, or matrix (symmetric or asymmetric) coefficients. Unit tests cover the new features. Fixed a bug in PAHcurlHdivApply3D which had no effect so far. 2020-07-21 22:16:28 -07:00
Tzanio Kolev 5457d033d4 Merge pull request #1380 from mfem/CVODESSolver-dev
CVODES Adjoint-Sensitivity Support
2020-07-21 16:05:35 -07:00
Tzanio 1deb071ada Update documentation 2020-07-21 16:03:53 -07:00
Veselin Dobrev 64ac989d07 Merge pull request #1582 from mfem/multigrid-nonsymmetric-smoother
[multigrid-nonsymmetric-smoother] Allow nonsymmetric smoothers in Multigrid object.
2020-07-21 15:25:30 -07:00
Tzanio bfff83d9de Fixed README 2020-07-21 15:00:30 -07:00
Tzanio 0efa0dcb21 Merge branch 'master' into CVODESSolver-dev
Conflicts:
	CHANGELOG
2020-07-21 14:19:48 -07:00
stefanhenneking d1b5234a09 Merging master into feature branch. 2020-07-21 16:18:38 -05:00
Tzanio Kolev 944cd2f09f Merge pull request #1567 from mfem/tmop-derivatives-dev
Complete action for the TMOP Integrator
2020-07-21 14:18:02 -07:00
Tzanio 01bf5db292 remove whitespace 2020-07-21 14:13:21 -07:00
Ketan Mittal 0b76f8d984 update changelog 2020-07-21 14:10:20 -07:00
Tzanio b52671541e minor styling 2020-07-21 13:57:48 -07:00
Tzanio e09103966e Merge branch 'master' into tmop-derivatives-dev 2020-07-21 12:54:43 -07:00
Tzanio Kolev 697cb9bb95 Merge pull request #1563 from mfem/complex-operator-pa
Partial assembly for complex operators
2020-07-21 10:52:24 -07:00
Dylan Copeland 8412926d1f Documentation 2020-07-21 10:27:53 -07:00
Dylan CopelandandStefan Henneking e6a0818041 Minor change.
Co-authored-by: Stefan Henneking <stefan.henneking@gmail.com>
2020-07-21 10:08:16 -07:00
Dylan CopelandandStefan Henneking 1bb517c695 Minor change.
Co-authored-by: Stefan Henneking <stefan.henneking@gmail.com>
2020-07-21 10:02:49 -07:00
Arturo Vargas c1071bf82e compare integration rule addr 2020-07-20 19:06:54 -07:00
Veselin Dobrev fe0211557b Minor: better formatting for multi-line string constants. 2020-07-20 18:34:06 -07:00
Tzanio Kolev 4d3db426c6 Merge pull request #1635 from mfem/fix/tpls_urls
Using MFEM hosted dependencies in CI context
2020-07-20 16:37:04 -07:00
Adrien M. Bernede 24fd1e1fc0 Revert "Suggesting hypre 2.18.2 instead of brand new 2.19"
This reverts commit 5a8cebeea7.
2020-07-20 16:25:59 -07:00
Adrien M. Bernede 5a8cebeea7 Suggesting hypre 2.18.2 instead of brand new 2.19 2020-07-20 15:25:29 -07:00
Veselin Dobrev 4cd4e21bc6 In .appveyor.yml, fix the hypre path given to mfem. 2020-07-20 13:34:52 -07:00
Veselin Dobrev 6489ecb59e In travis and appveyor, update to hypre v2.19.0. 2020-07-20 13:28:20 -07:00
Adrien M. Bernede 4e8aaf7f11 Fix hypre tar name in travis 2020-07-20 11:19:07 -07:00
Adrien M. Bernede ccce5c8217 Fix file name 2020-07-20 10:53:39 -07:00
Adrien M. Bernede ad9adab6dc Fix extension 2020-07-20 10:44:55 -07:00
Adrien M. Bernede b2cdfbf8bc Reverting changes for Hypre 2020-07-20 10:30:23 -07:00
Adrien BernedeandTzanio Kolev fb9c3fa30a Update .travis.yml
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2020-07-20 10:23:31 -07:00
Adrien BernedeandTzanio Kolev 8b1ecbc3af Update .travis.yml
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2020-07-20 10:22:53 -07:00
Adrien BernedeandTzanio Kolev 66cbd5450e Update .appveyor.yml
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
2020-07-20 10:22:38 -07:00
Tzanio Kolev a965c079fb Merge pull request #1604 from mfem/Complex-umfpacksolver
ComplexUMFPackSolver
2020-07-19 21:51:51 -07:00
Tzanio Kolev 31d2ff15e2 Merge pull request #1588 from mfem/hip-defs
[HIP/AMD] #define fix & Diffusion kernel optimization
2020-07-19 21:49:40 -07:00
Jonathan Wong 72d7f7ccb7 Added to miniapps to code documentation 2020-07-17 21:39:42 -07:00
Jonathan Wong 44e8364877 modified CHANGELOG to accommodate column limit 2020-07-17 21:09:57 -07:00
Jonathan Wong b4b6c7e106 updated CHANGELOG 2020-07-17 21:05:58 -07:00
Adrien M. Bernede d64ce893ad Using MFEM hosted dependencies in CI context 2020-07-17 18:09:25 -07:00
stefanhenneking 5c1f9b64af Removing faulty initialization of the solution vector. 2020-07-16 13:24:55 -05:00
Dylan Copeland 77b6729309 Updating description of input parameters omitted from a previous PR. 2020-07-16 10:11:18 -07:00
Veselin Dobrev 0effa7abff Merge pull request #1430 from mfem/ode-state-io-dev
Add access mechanism for state vectors in ODE solvers [ode-state-io-dev]
2020-07-15 18:18:58 -07:00
Veselin Dobrev 127fa2645c Merge pull request #1569 from mfem/coefficient-use-cases-doc-dev
Add comment about coefficient use cases [coefficient-use-cases-doc-dev]
2020-07-15 18:16:24 -07:00
Veselin Dobrev f4959fc875 Merge pull request #1503 from najlkin/pr13
Fixed GeometryRefiner::RefineInterior() to not cause invalid deallocations
2020-07-15 18:14:17 -07:00
Veselin Dobrev dd3c075b1b Merge pull request #1550 from mfem/block-nlf-bc-fix
Performance fix with essential dofs in BlockNonlinearForm
2020-07-15 18:11:19 -07:00
Tomov 27ef7812f8 Minor. 2020-07-15 17:00:57 -07:00
Tomov e29ee5f5fc Changed the exact derivative test (to cover more geometric parameters). 2020-07-15 16:59:14 -07:00
stefanhenneking b570911a15 Updating changelog. 2020-07-15 17:31:56 -05:00
stefanhenneking a9f3f42289 Updating changelog. 2020-07-15 17:24:04 -05:00
stefanhenneking 3da43efb86 Merge branch 'master' of github.com:mfem/mfem into complex-operator-gpu 2020-07-15 16:32:57 -05:00
Veselin Dobrev f3771a22a5 Merge branch 'master' into ode-state-io-dev 2020-07-15 14:29:50 -07:00
Veselin Dobrev 31e4a9d8ee Merge branch 'master' into multigrid-nonsymmetric-smoother 2020-07-15 14:27:20 -07:00
Tzanio Kolev 75ef30918c Merge branch 'master' into complex-operator-pa 2020-07-15 14:25:25 -07:00
Tzanio Kolev 6871b1c6dc Merge branch 'master' into blockop_cuda 2020-07-15 14:23:17 -07:00
Tzanio 92ef9d0629 minor 2020-07-15 14:21:39 -07:00
Tzanio Kolev 1e15e6b57f Merge pull request #1615 from mfem/array-doc-pod-dev
Document that Array<T> operates correctly only with POD types
2020-07-15 14:06:07 -07:00
Tzanio Kolev d1beabdd0d Merge branch 'master' into Complex-umfpacksolver 2020-07-15 14:05:02 -07:00
stefanhenneking 064a859fd1 Minor fix in member variable initialization. 2020-07-15 15:18:36 -05:00
stefanhenneking 56211dfeb9 Merging complex-operator-pa branch. 2020-07-15 15:15:59 -05:00
Dylan Copeland ea4d8c365c Restoring another unit test. 2020-07-15 11:29:00 -07:00
Stefan HennekingandWill Pazner fc1d4fffaf Apply suggestions from code review
A few simplifications.

Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2020-07-15 12:17:54 -05:00
stefanhenneking d79f834750 Use dynamic_cast to check operator type. 2020-07-15 12:02:47 -05:00
stefanhenneking b765ebad81 Minor correction. 2020-07-15 11:54:38 -05:00
Stefan HennekingandWill Pazner 61be39191a Apply suggestions from code review
minor

Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2020-07-15 11:50:13 -05:00
camierjs cfe7834b95 minor 2020-07-15 08:51:17 -07:00
Stefan Henneking ac69933f77 Merge branch 'master' into complex-operator-gpu 2020-07-15 10:50:21 -05:00
camierjs 1a8d88440e Remove dead code, add comments and move AtomicAdd function for visibility 2020-07-15 08:22:43 -07:00
camierjs 99c4becfae Merge branch 'master' into hip-defs 2020-07-15 07:24:28 -07:00
Dylan Copeland 96dd27f68f Restoring changes. 2020-07-14 22:30:10 -07:00
Dylan Copeland ab51c0ad38 Merge branch 'master' of https://github.com/mfem/mfem into matcoefpa 2020-07-14 21:33:08 -07:00
Tzanio 40e633b36f minor styling 2020-07-14 18:04:38 -07:00
Tzanio Kolev 94a5e625be Merge branch 'master' into CVODESSolver-dev 2020-07-14 17:22:46 -07:00
stefanhenneking 7009af9ecc Simplifying MakeRef functions. 2020-07-14 19:21:54 -05:00
Tzanio Kolev 840ff99288 Merge pull request #1506 from mfem/yohann/fa
Full Assembly built on top of Element Assembly
2020-07-14 17:21:46 -07:00
Tzanio fbe242a2e2 Updated CHANGELOG. Minor styling 2020-07-14 17:18:59 -07:00
Tzanio Kolev ae880b4ee8 Merge branch 'master' into yohann/fa 2020-07-14 16:42:03 -07:00
stefanhenneking 33b413042a Destroying aliased block vectors to avoid issues with dangling references in memory manager. 2020-07-14 18:32:26 -05:00
Tzanio Kolev 9cff5875c7 Merge pull request #1463 from mfem/mixedcurl
PA for mixed curl integrators
2020-07-14 16:22:54 -07:00
stefanhenneking a2da036bdb Destroying alias vectors to avoid issues with dangling references in memory manager. 2020-07-14 18:18:04 -05:00
Tzanio 80c787a79f minor 2020-07-14 16:15:33 -07:00
Tzanio 9e76838fe4 minor 2020-07-14 16:08:33 -07:00
Tzanio Kolev 75f5a89d2c Merge branch 'master' into array-doc-pod-dev 2020-07-14 15:56:09 -07:00
Tzanio Kolev c4d8bd4744 Merge branch 'master' into yohann/fa 2020-07-14 15:27:10 -07:00
Tzanio Kolev 0fb7f04d6c Merge branch 'master' into mixedcurl 2020-07-14 15:26:40 -07:00
Tzanio Kolev 01476b98cb Merge branch 'master' into CVODESSolver-dev 2020-07-14 15:26:30 -07:00
Tzanio d2ae506e8e minor 2020-07-14 12:20:11 -07:00
Ketan Mittal 65dfcd5e0a minor 2020-07-12 13:11:24 -07:00
Ketan Mittal c3d869cd6c flag to enable exact action 2020-07-10 17:00:59 -07:00
Ketan Mittal 6b256c7cbb Merge branch 'master' of https://github.com/mfem/mfem into tmop-derivatives-dev 2020-07-10 12:03:05 -07:00
Ketan Mittal b73225de21 reviewer comments 2020-07-10 12:02:39 -07:00
Stefan Henneking c3dd82b5ba Merge branch 'master' into complex-operator-pa 2020-07-10 11:36:55 -05:00
stefanhenneking 6cb82fa126 Merge branch 'master' of github.com:mfem/mfem into complex-operator-gpu 2020-07-10 11:35:55 -05:00
Bob Anderson 291875509d document that Array operates correctly only with POD types 2020-07-09 15:32:00 -07:00
Veselin Dobrev d81b3fa05a Merge pull request #1573 from mfem/vector-add-dev
Add a simple missing member function [vector-add-dev]
2020-07-09 15:17:58 -07:00
Arturo Vargas 07f7b0a943 PR comments 2020-07-09 13:28:43 -07:00
Veselin Dobrev b4b72a95ee Merge branch 'master' into vector-add-dev 2020-07-09 12:31:06 -07:00
Veselin Dobrev 57981bc329 Merge pull request #1576 from mfem/gpufix/artv3/SaveAsOne
Missing HostReadWrite: pgridfunc.cpp::SaveAsOne
2020-07-09 12:30:06 -07:00
Arturo Vargas 43ceae8f46 Merge branch 'master' into feature/artv3/quad-data 2020-07-09 11:19:45 -07:00
psocratis f329c3b760 fixed ifdef for superlu 2020-07-08 20:16:24 -07:00
Socratis ceb49b322c removed space in comment 2020-07-08 18:04:51 -07:00
psocratis 6e145a4ccf Added print_level for UMFPackSolver in ex25. Moved contruction of monolithic HypreParMatrix to the case where SuperLU is used 2020-07-08 16:57:13 -07:00
psocratis 79e9a3d320 fixed typos in complex_operator 2020-07-08 13:54:45 -07:00
stefanhenneking 8e90fcde40 Merging complex-operator-pa features into this complex-operator-gpu. 2020-07-08 14:48:54 -05:00
stefanhenneking ef41d0f3c1 Merge branch 'complex-operator-gpu' of github.com:mfem/mfem into complex-operator-gpu 2020-07-08 14:38:12 -05:00
stefanhenneking d32f760854 Merge branch 'master' of github.com:mfem/mfem into complex-operator-gpu
Merging master into feature branch.
2020-07-08 14:37:12 -05:00
stefanhenneking 80681fa56a A few small changes to simplify ex22p. Note: Hypre prec. always uses DIAG_ONE policy (no need to set it separately as in ex22). 2020-07-08 14:09:28 -05:00
stefanhenneking 8e8868e4a0 Simplifying ex22 and correcting preconditioner diag policy. 2020-07-08 12:35:39 -05:00
Tzanio Kolev b671a7a679 Merge branch 'master' into Complex-umfpacksolver 2020-07-08 10:26:35 -07:00
stefanhenneking 6d7b38c02f Fixed a typo. 2020-07-08 11:57:35 -05:00
stefanhenneking 3e6889145e Adding assembly level member function to par-/sesquilinear form. 2020-07-08 10:13:46 -05:00
stefanhenneking 60378b79af Fixed a typo. 2020-07-08 09:43:37 -05:00
Veselin Dobrev 5a30e94472 Merge branch 'master' into pr13 2020-07-07 22:07:35 -07:00
Veselin Dobrev cdfe1db094 Merge branch 'master' into ode-state-io-dev 2020-07-07 21:26:29 -07:00
Veselin Dobrev 99a98596ae Merge branch 'master' into gpufix/artv3/SaveAsOne 2020-07-07 21:25:28 -07:00
Tzanio Kolev 7947109731 Merge branch 'master' into master 2020-07-07 21:19:05 -07:00
Veselin Dobrev ab13556e8c Merge branch 'master' into block-nlf-bc-fix 2020-07-07 21:16:24 -07:00
Tzanio Kolev df7aec2506 Merge pull request #1513 from mfem/quadspace-visit-dev
Add quadraturefunctions to visit datacollection [quadspace-visit-dev]
2020-07-07 21:13:53 -07:00
Veselin Dobrev f07cb7379d Merge branch 'master' into block-nlf-bc-fix 2020-07-07 21:13:31 -07:00
psocratis cbb167f231 minor comment fix 2020-07-07 19:46:17 -07:00
psocratis 49e6225b7d minor fixes 2020-07-07 19:41:36 -07:00
Jan Nikl 5f0630a550 Replaced the refined IntegrationRule existence check by an assert. 2020-07-07 21:52:44 +02:00
Florian Omnes 525f6d2a44 Apply code-style 2020-07-07 21:48:03 +02:00
Veselin Dobrev 6e113683af Additions and changes:
* Added makefile in miniapps/adjoint.
* Fixed interface inconsistency with the ToNVector() methods.
* Fix various warnings.
* In the CMake build system distinguish the SUNDIALS components
  CVODE and CVODES.
* Fix a file name in .gitignore.
2020-07-07 00:09:27 -07:00
Jonathan Wong 99e69b93e5 Merge branch 'CVODESSolver-dev' of github.com:mfem/mfem into CVODESSolver-dev 2020-07-06 21:39:20 -07:00
Jonathan Wong 53e952f1fd Address makefile issues 2020-07-06 21:38:36 -07:00
Florian Omnes 3dd0f5c328 Tqke v-dobrev's remarks into account in ex1p.cpp 2020-07-06 23:30:20 +02:00
Florian Omnes 827fbfb14a Take v-dobrev's remark into account in ex1.cpp 2020-07-06 23:25:51 +02:00
camierjs 7c3912b2e3 Merge branch 'master' into hip-defs 2020-07-06 10:39:25 -07:00
Tzanio 39a6c88595 Various styling edits 2020-07-05 18:41:58 -07:00
Tzanio Kolev 9145d4b1de Merge branch 'master' into Complex-umfpacksolver 2020-07-05 17:33:27 -07:00
psocratis 9fb9c4937a Merge branch 'Complex-umfpacksolver' of https://github.com/mfem/mfem into Complex-umfpacksolver 2020-07-04 21:00:45 -07:00
psocratis 9d3e3dd017 make style 2020-07-04 21:00:12 -07:00
Tzanio Kolev e8458444c6 Merge branch 'master' into CVODESSolver-dev 2020-07-04 18:57:24 -07:00
Tzanio Kolev 2d065de342 Merge branch 'master' into mixedcurl 2020-07-04 14:25:04 -07:00
Tzanio Kolev 7c3a368562 Merge branch 'master' into Complex-umfpacksolver 2020-07-04 13:51:10 -07:00
psocratis e2ff03e4ba fixed numbering of comments 2020-07-03 16:31:06 -07:00
psocratis a510328015 Remove Constuction of monolithic SparseMatrix from a ComplexSparseMatrix. Not needed anymore 2020-07-03 15:56:24 -07:00
psocratis 97440f9500 Added ComplexUMFPackSolver to ex25.cpp. Removed unused variable u_gf from ex25[p].cpp 2020-07-03 15:33:57 -07:00
psocratis 781fddddc6 Added ComplexUMFPackSolver to ComplexOperator 2020-07-03 15:10:23 -07:00
psocratis 7fcc651020 Added GetConvention to ComplexOperator 2020-07-03 14:54:04 -07:00
Tzanio 4c719ad706 minor: 2020-07-02 21:40:45 -07:00
Yohann Dudouit 2de28abb19 Revert changes in ex9 and ex9p. 2020-07-02 15:51:05 -07:00
Veselin Dobrev 79dd7c14b2 Merge branch 'master' into ode-state-io-dev 2020-07-02 14:47:31 -07:00
Dylan Copeland c1320238ae Minor changes to MFEM_ABORT_KERNEL. 2020-07-02 10:51:12 -07:00
Veselin Dobrev 4b10b7c44b Merge branch 'master' into block-nlf-bc-fix 2020-07-01 23:01:59 -07:00
Stefan Henneking a298f02b4c Enable block diagonal preconditioner for device computation. 2020-07-01 13:51:50 -07:00
Stefan Henneking ec8b00ea1e Merge branch 'blockop_cuda' of github.com:mfem/mfem into complex-operator-gpu
Merging support for BlockOperator on device from feature branch.
2020-07-01 13:24:56 -07:00
Stefan Henneking 91eacf5af7 Removing typo. 2020-07-01 09:37:46 -07:00
Stefan Henneking 6543ffb790 Update method should support vectors allocated on device. 2020-07-01 09:14:04 -07:00
Stefan Henneking e4290e6d33 Make memory allocation precise to avoid futures issues. 2020-07-01 09:13:08 -07:00
Stefan Henneking de34bf094c Setting block vector memory type to support device computation. 2020-07-01 09:12:14 -07:00
Stefan Henneking a52a59b524 Removing obsolete comments. 2020-07-01 07:24:42 -07:00
Ido Akkerman 30b2b43814 Correct typo 2020-07-01 09:39:35 +02:00
Dylan Copeland 1f4024879c Adding -d cuda support to ex5p. 2020-06-30 17:51:38 -07:00
Stefan Henneking d949f69a4b Modifying BlockOperator and BlockDiagonalPreconditioner MultTranspose for device support. 2020-06-30 15:53:41 -07:00
stefanhenneking 6cdad9b4ee Minor style change. 2020-06-30 17:25:40 -05:00
Stefan Henneking d727b1a14b Minor change. 2020-06-30 15:23:41 -07:00
Stefan Henneking a6c6fb18cf Using alias to compute BlockOperator and BlockDiagonalPreconditioner Mult on device. 2020-06-30 15:20:28 -07:00
Stefan Henneking 069e57b3a4 Removing HostRead (should not be necessary here). 2020-06-30 15:17:36 -07:00
Dylan Copeland e77e7f592b Adding support for matrix coefficients in H(curl) mass diagonal assembly, with unit tests. 2020-06-30 15:10:55 -07:00
Florian Omnes df59effa09 Revert config 2020-06-30 23:04:52 +02:00
Florian Omnes 09c9c94916 Simplify memory management in ex1p.cpp 2020-06-30 23:00:53 +02:00
stefanhenneking 30ad68af20 Merge branch 'master' of github.com:mfem/mfem into blockop_cuda
Merging master into feature branch.
2020-06-30 14:59:44 -05:00
Dylan Copeland 237905c956 Addressing Veselin's comments about MFEM_ABORT_KERNEL. 2020-06-30 11:15:14 -07:00
Veselin Dobrev b4d870c3be Merge branch 'master' into gpufix/artv3/SaveAsOne 2020-06-29 21:47:12 -07:00
Arturo Vargas 213a290511 use host read only 2020-06-29 16:28:16 -07:00
Andrew T. Barker 2fa65d4846 Add unit test for symmetry of OperatorChebyshevSmoother 2020-06-29 15:08:37 -07:00
camierjs 661a7f6f38 Update CHANGELOG 2020-06-29 15:08:10 -07:00
camierjs 14e663d663 Update general/CMakeLists.txt 2020-06-29 15:06:59 -07:00
stefanhenneking f9ed143f40 Removing typos. 2020-06-29 16:00:21 -05:00
Florian Omnes b8270effa2 Simplify memory management in example 1 2020-06-29 22:24:16 +02:00
Stefan Henneking e5570e9e4c Sync memory after recovering FEM solution on device. 2020-06-29 12:24:42 -07:00
camierjs 99bc161b86 Cleanup & Meld toward master 2020-06-29 10:52:16 -07:00
camierjs 85b642bd77 Merge branch 'master' into hip-defs 2020-06-29 10:32:59 -07:00
Stefan Henneking af900cf8d7 Merge branch 'master' of github.com:mfem/mfem into complex-operator-gpu
Merging master into feature branch.
2020-06-29 10:25:06 -07:00
Stefan Henneking a57a3eb070 Enabling device support for ComplexParLinearForm. 2020-06-29 10:23:07 -07:00
Stefan Henneking aea668a9f9 Adding MakeRef function to ParLinearForm. 2020-06-29 10:22:18 -07:00
Dylan Copeland b732ae829f Replacing comments. 2020-06-29 10:05:50 -07:00
Stefan Henneking a3ebecd8ac Minor change in function doc. 2020-06-29 09:49:17 -07:00
Stefan Henneking ac4aa43430 Enable device support for ParSesquilinearForm. 2020-06-26 15:16:41 -07:00
Stefan Henneking 47d3d7ead1 Enable device support for ParComplexGridFunction. 2020-06-26 14:35:07 -07:00
Stefan Henneking 03473d90fa Ensure vector is registered on device before using alias. 2020-06-26 14:33:26 -07:00
Andrew T. Barker 03533d095c Allow nonsymmetric smoothers in Multigrid object. 2020-06-26 13:22:25 -07:00
stefanhenneking e4529f82f7 Adding device option to ex22p. 2020-06-26 13:58:52 -05:00
stefanhenneking a883eb7287 Minor style change. 2020-06-26 11:53:34 -05:00
Stefan Henneking aaf321caab Fixing a few typos in documentation. 2020-06-26 09:49:53 -07:00
Stefan Henneking b9b7c7b046 Enabling device support for complex linear form. 2020-06-26 09:39:34 -07:00
Ido Akkerman bd695bc74c Make style 2020-06-26 11:02:08 +02:00
Ido Akkerman e165101b27 Revert gitignore and make clean back to old version as unit test output has disappeared 2020-06-26 10:40:43 +02:00
Ido Akkerman 096e5ffb93 Adding quadfunctions to visit unit test 2020-06-26 10:36:42 +02:00
Ido Akkerman fb8e595da3 Reverting back unit test and cmake file 2020-06-26 10:24:21 +02:00
Ido Akkerman 868d8aa057 Add comment regarding VisIt's inability to visualize quadfun as of yet 2020-06-26 09:59:50 +02:00
Ido Akkerman dd09413e47 Convert comment to correct doxygen format 2020-06-26 09:53:49 +02:00
Yohann Dudouit c6a5a75d35 Fix strange Valgrind uninitialized value. 2020-06-25 19:59:41 -07:00
Stefan Henneking 3a9bfe3c81 Enabling device support for ComplexGridFunction::Update(). 2020-06-25 15:39:12 -07:00
Stefan Henneking dce5bf5801 Enabling device support for example ex22. 2020-06-25 15:06:24 -07:00
Stefan Henneking 1fd05bf80d Enabling support for device computation for complex operator transpose mult. 2020-06-25 14:46:49 -07:00
Stefan Henneking 302886dda3 Enable device support for sesquilinear form and complex grid function. 2020-06-25 13:34:04 -07:00
Stefan Henneking c34f87aab7 Modifying complex operator mult for device support. 2020-06-25 13:11:58 -07:00
Arturo Vargas 66c6b9aa0d forgot to add file 2020-06-24 12:03:45 -07:00
Ketan Mittal c27db29466 minor 2020-06-24 11:39:59 -07:00
stefanhenneking b027c1c6cc Minor style changes. 2020-06-24 12:20:55 -05:00
stefanhenneking 29136050db Minor change in initializing member variable. 2020-06-24 12:07:09 -05:00
stefanhenneking 89aade4b2d Fixing minor bug. 2020-06-24 11:56:53 -05:00
stefanhenneking 0770a21d2a Fixing minor bug. 2020-06-24 11:10:34 -05:00
stefanhenneking 92cb4a02a7 Merge branch 'master' of github.com:mfem/mfem into complex-operator-pa
Merging changes from master into feature branch.
2020-06-24 10:57:44 -05:00
stefanhenneking 335d810155 Fixing typo. 2020-06-24 10:40:14 -05:00
stefanhenneking 18bb5a5ac0 Remove DIAG_KEEP option for now. 2020-06-24 10:39:07 -05:00
Ido Akkerman f29e1b82f7 Add vector unit test 2020-06-24 17:20:38 +02:00
Ido Akkerman 0b5ee4ea04 Add missing add vector member function 2020-06-24 17:20:20 +02:00
Ido Akkerman 6201d7c5bb Adding to clean 2020-06-24 10:40:15 +02:00
Ido Akkerman a5e0c3f856 Ignore output dir of unit test 2020-06-24 10:17:30 +02:00
Ketan Mittal e170d20edc minor 2020-06-23 17:35:23 -07:00
Ketan Mittal beedb1e931 Merge branch 'master' of https://github.com/mfem/mfem into tmop-derivatives-dev 2020-06-23 17:35:01 -07:00
Dylan Copeland 83ec745644 Revert "Implemented matrix coefficients in H(curl) mass integrator, for symmetric and asymmetric cases. Added symmetry property to MatrixCoefficient. Unit tests cover the new features."
This reverts commit 7929766814.

Conflicts:
	fem/coefficient.hpp
2020-06-23 14:22:07 -07:00
Tzanio Kolev 12590207fa Merge branch 'master' into CVODESSolver-dev 2020-06-23 13:25:01 -07:00
Jonathan Wong cc64d1fd9c changed cmake name from advection_diffusion to adjoint_advection_diffusion to highlight that the adjoint calculation is being performed 2020-06-23 13:21:57 -07:00
Bob Anderson ab4c17ab7d Add note about coefficient use cases 2020-06-23 11:37:00 -07:00
Ido Akkerman 18007107d8 make style 2020-06-23 10:34:21 +02:00
Jonathan Wong 85349d3a95 Properly intialize quad integration vectors.
Fixed memory errors in ToNVector and initialize in CVODESSolver::RHSB
2020-06-22 23:34:43 -07:00
Ketan Mittal 2e610493c6 Merge branch 'tmop-solvers-dev' of https://github.com/mfem/mfem into tmop-derivatives-dev 2020-06-22 16:35:46 -07:00
Ketan Mittal 79aa383c63 resolve conflicts 2020-06-22 15:38:53 -07:00
stefanhenneking 64bbbc3d8c Minor style changes. 2020-06-21 23:32:47 -05:00
Dylan Copeland 7e88d111f9 Merge branch 'master' of https://github.com/mfem/mfem into mixedcurl
Conflicts:
	fem/coefficient.hpp
2020-06-19 23:15:18 -07:00
Dylan Copeland 7929766814 Implemented matrix coefficients in H(curl) mass integrator, for symmetric and asymmetric cases. Added symmetry property to MatrixCoefficient. Unit tests cover the new features. 2020-06-19 22:58:08 -07:00
stefanhenneking bc795fc99a Adding PA option to example ex22. 2020-06-19 16:18:49 -05:00
stefanhenneking 5d80f8e195 Using OperatorHandle for SesquilinearForm to enable PA. 2020-06-19 15:58:54 -05:00
Veselin Dobrev a00bcade5f Merge branch 'master' into pr13 2020-06-19 12:47:06 -07:00
stefanhenneking 8491ec4183 Adding PA option to ex22p. 2020-06-19 13:45:21 -05:00
stefanhenneking dbba71bc7c Adding diagonal policy to ParSesquilinearForm for PA. 2020-06-19 11:59:56 -05:00
stefanhenneking aea81c2920 Adding diagonal policy to constrained operator. 2020-06-19 11:54:48 -05:00
stefanhenneking 4e7821c809 Remove operator ownership before calling destructor. 2020-06-19 09:58:08 -05:00
Jonathan Wong c85d34c34a corrected typo and added documentation that @jandrej suggested 2020-06-17 17:05:30 -07:00
Yohann Dudouit fd5fe341b1 Refine face_mat condition. 2020-06-17 11:51:13 -07:00
Julian Andrej b95407f6ea finalize jacobian blocks in BlockNonlinearForm before essential dof eliminiation 2020-06-15 15:09:04 -07:00
Jonathan Wong 02f6ca0ef2 Freed the nvector to prevent a memory leak 2020-06-15 15:05:20 -07:00
Yohann Dudouit fc57c1be85 Fix a bug. 2020-06-11 12:35:02 -07:00
Ido Akkerman 21e6e7f669 Adding quadfun read write unit_test 2020-06-11 18:21:30 +02:00
Ido Akkerman c9c2cd825e Merge branch 'master' into quadspace-visit-dev 2020-06-11 17:21:29 +02:00
Yohann Dudouit 6c52bec12d Try less verbose constant lambda capture. 2020-06-10 14:34:36 -07:00
Yohann Dudouit 01c8643d1d make style. 2020-06-10 14:10:59 -07:00
Yohann Dudouit 3b35126fbc Remove some of the #ifdef MFEM_USE_MPI (thanks Will Pazner) 2020-06-10 13:29:16 -07:00
Yohann Dudouit 3ec12f0866 Add device sample runs to ex9 and ex9p. 2020-06-10 12:45:10 -07:00
Yohann Dudouit 1c8a8dbcb1 Minor 2020-06-09 20:28:21 -07:00
Yohann Dudouit 55b26db935 make style. 2020-06-09 20:26:18 -07:00
Yohann Dudouit 5ebc924b20 Add parallel support for DG face terms. 2020-06-09 19:56:33 -07:00
Yohann Dudouit e975ad2950 Revert iteration count in ex1 to be different. 2020-06-09 14:37:19 -07:00
Jonathan Wong 58997c32bd initialized N_Vector pointer to struct type to NULL 2020-06-09 14:07:26 -07:00
Jonathan Wong c0395c6371 tidy up parallel output for advection_diffusion problem 2020-06-09 13:21:25 -07:00
Dylan Copeland 68fd30d2f9 Adding MFEM_ABORT_KERNEL, currently used only in SparseMatrix::DiagScale. 2020-06-08 22:42:48 -07:00
Dylan Copeland e4a6e1c724 Added MFEM_FORALL to SparseMatrix::DiagScale. Made some minor style improvements. 2020-06-08 21:40:11 -07:00
Arturo Vargas 802e20160c clean up pass 2020-06-08 19:53:03 -07:00
Arturo Vargas 0055894917 added support for user coefficient scaling for conv,trace,diff,mass 2020-06-08 19:29:14 -07:00
Yohann Dudouit f1ac4b1eff Rename mfemAtomicAdd ro AtomicAdd. 2020-06-08 18:40:24 -07:00
Yohann Dudouit c2de4ee220 Move synchronization inside Print(). 2020-06-08 18:37:11 -07:00
Yohann Dudouit c53fe3ffda Remove dead code. 2020-06-08 18:21:18 -07:00
Yohann Dudouit 105a991779 Remove MFEM_ABORT... 2020-06-08 18:17:52 -07:00
Yohann Dudouit e831e8007b Use OperatorJacobiSmoother in ex9.
- implement GetDiag() for SparseMatrix on device.
2020-06-08 18:10:13 -07:00
Arturo Vargas 01d53b32fa added user quadrature support to mass/diffusion integrators 2020-06-08 17:42:51 -07:00
Yohann Dudouit dee42c559a AppVeyor... 2020-06-08 17:19:56 -07:00
Yohann Dudouit 33c28f7591 Try to make AppVeyor happy... 2020-06-08 17:00:20 -07:00
Yohann Dudouit 03c0009414 Rename simplify into factorize_face_terms. 2020-06-08 15:52:56 -07:00
Yohann Dudouit 6f4c373d58 update AssemblyLevel documentation. 2020-06-08 15:33:14 -07:00
Yohann Dudouit 5a27cb9e54 update documentation. 2020-06-08 15:30:39 -07:00
Yohann Dudouit 48a018e7a3 Remove commented optimization ideas. 2020-06-08 15:25:05 -07:00
Yohann Dudouit 39fc24685f Attempt to make AppVeyor happy. 2020-06-08 15:08:53 -07:00
Yohann Dudouit db626777be make style 2020-06-08 13:31:42 -07:00
Yohann Dudouit b3f4e5bb0b Remove magic number. 2020-06-08 12:59:00 -07:00
Yohann Dudouit 04f1d562b7 Revert ex1 and ex1p. 2020-06-08 12:38:59 -07:00
Yohann c610cabe53 Merge branch 'master' into yohann/fa 2020-06-08 12:22:54 -07:00
Yohann Dudouit 14ac6afcd8 Rename GetNnzInd to GetAndIncrementNnzIndex. 2020-06-08 12:19:24 -07:00
Yohann Dudouit eea03f48bb Rename FactorizeBlocks to AddFaceMatricesToElementMatrices. 2020-06-08 12:04:39 -07:00
Yohann Dudouit bd10bef5d1 Rename FillSpMat to FillSparseMatrix. 2020-06-08 11:33:02 -07:00
Yohann Dudouit cb3202f9f9 Rename GetNnzInd to GetNnzIndex and add doc. 2020-06-08 11:11:42 -07:00
Yohann Dudouit 3fb736981e Rename FillJandData to FillJAndData. 2020-06-08 11:06:09 -07:00
Yohann Dudouit 802a4afce0 make style 2020-06-08 11:00:55 -07:00
Arturo Vargas 10bed2997c use quadraturefunction coefficient to supply user qpt scaling 2020-06-05 10:24:35 -07:00
Ido Akkerman ec8796ebde make style 2020-06-05 17:43:36 +02:00
Ido Akkerman e985684812 Improved get refinement routines 2020-06-05 17:12:26 +02:00
Arturo Vargas 55d27cbd91 Merge branch 'master' into feature/artv3/quad-data 2020-06-03 16:29:18 -07:00
Ido Akkerman 51faf60eab Corrections for double int conversion 2020-06-03 22:09:26 +02:00
Ido Akkerman 23cec0568b Corrections for double int conversion 2020-06-03 22:09:09 +02:00
Ido Akkerman d83112c174 Add individual LOD param to visit fields 2020-06-03 16:36:25 +02:00
Ido Akkerman cab24d6f3d Add GetRefinementLevel routine 2020-06-03 16:24:52 +02:00
Ido Akkerman 572470939a Adding ClosedGL pointset for ViSit vis of quadrature data 2020-06-03 16:22:11 +02:00
Yohann Dudouit ff57240475 Update unit tests. 2020-06-02 15:23:11 -07:00
Yohann Dudouit 650941acc9 Update CHANGELOG 2020-06-02 15:10:44 -07:00
Jonathan Wong 4540775fdd deleted extra blank lines 2020-06-02 12:17:29 -07:00
Jonathan Wong e18858ab09 adjusted comment in SUNIMplicitSolveB 2020-06-01 23:01:16 -07:00
Jonathan Wong 5878be5cb8 delete eliminated matrix that results from calls to EliminateRowsCols() 2020-06-01 22:58:10 -07:00
Jonathan Wong 17d4cba0c9 fixed memory leak issues 2020-06-01 22:51:44 -07:00
Jonathan Wong e30182268d fixed linewidths, removed numbering, removed unused objects 2020-06-01 22:17:40 -07:00
Jonathan Wong e6163eb49c fixed merge with master and changes in hyper ToNvector() 2020-06-01 16:14:00 -07:00
Jonathan Wong edee386ec8 fixed numbering issue in advection_diffusion.cpp 2020-05-28 20:51:48 -07:00
Dylan Copeland f5aa751bba Fixed SparseMatrix::DiagScale to work on GPU. 2020-05-28 16:46:41 -07:00
Dylan Copeland 81b1848021 Fixing some compiler warnings. 2020-05-28 12:57:24 -07:00
Tzanio Kolev caedc3be67 Merge branch 'master' into mixedcurl 2020-05-28 10:44:51 -07:00
Ketan Mittal 19f484ec72 Merge branch 'tmop-solvers-dev' of https://github.com/mfem/mfem into tmop-derivatives-dev 2020-05-27 11:26:16 -07:00
Ketan Mittal e29ae919ad minor 2020-05-27 11:24:06 -07:00
Ido Akkerman d87f4e347c add qspace to visit 2020-05-27 17:53:10 +02:00
Yohann Dudouit a502f38360 Simplify FillI() for face DG. 2020-05-27 01:15:26 -07:00
Yohann Dudouit fa5ccb3e9f Remove MultTranspose in ParL2FaceRestriction. 2020-05-27 00:54:04 -07:00
Yohann Dudouit fc24fad31c Revert the constructor of ParL2FaceRestriction. 2020-05-26 19:11:11 -07:00
Yohann Dudouit 96a5ef5392 Reorganize L2FaceRestriction constructors (thanks Will Pazner). 2020-05-26 17:17:19 -07:00
Yohann Dudouit 5fe2e5e573 Authorize SetAssemblyLevel(LEGACYFULL). 2020-05-26 17:05:15 -07:00
Yohann Dudouit bac87d5bdf Move constexpr inside MFEM_FORALL. 2020-05-26 16:07:20 -07:00
Yohann Dudouit 1441fca816 Another attempt for AppVeyor. 2020-05-26 16:01:56 -07:00
Yohann Dudouit 272d529aee Another attempt to make AppVeyor happy. 2020-05-26 15:40:39 -07:00
Yohann Dudouit 5e3d92693f Use contexpr to try to make AppVeyor happy. 2020-05-26 15:13:21 -07:00
Yohann Dudouit b0abbc3a24 Set LEGACYFULL to 0. 2020-05-26 14:20:11 -07:00
Yohann Dudouit 22afa40eb0 Update Documentation. 2020-05-26 14:12:51 -07:00
Yohann Dudouit 6e8001e206 Add a LEGACYFULL assembly mode. 2020-05-26 14:08:05 -07:00
Yohann Dudouit 27b6172257 Modify test_matrix_square to use the legacy full assembly. 2020-05-26 13:35:26 -07:00
Yohann Dudouit 00a431e5b3 Remove white space. 2020-05-26 13:19:59 -07:00
Yohann Dudouit cc9dd4e34e Modify parallel exemples for testing purpose. 2020-05-26 13:03:49 -07:00
Yohann Dudouit ca8e3e73a9 Rewrite a bit FactorizeBlocks. 2020-05-26 12:56:56 -07:00
Yohann Dudouit c4a1a23756 Use MFEM_FOR_ALL 2020-05-26 12:51:51 -07:00
Yohann Dudouit 6571977c97 Modify constructor of L2FaceRestriction. 2020-05-26 12:44:54 -07:00
Yohann Dudouit e0b19c5520 make style. 2020-05-26 11:44:04 -07:00
Yohann Dudouit 1ea27f2805 Make ParL2FaceRestriction inherit from L2FaceRestriction. 2020-05-26 11:19:29 -07:00
Yohann Dudouit 957aa9aeef Remove ex9pa. 2020-05-22 21:14:16 -07:00
Yohann Dudouit ae5da8e9ac Fix MultTranspose in Element Assembly. 2020-05-22 21:04:37 -07:00
Yohann Dudouit c45ed09112 Typos. 2020-05-22 21:00:18 -07:00
Yohann Dudouit 42522ddd43 Remove more dead code. 2020-05-22 20:57:12 -07:00
Yohann Dudouit 1ecf80a2f7 Remove dead code. 2020-05-22 20:49:46 -07:00
Yohann Dudouit 93c3684eb1 Use atomicAdd instead of ++... 2020-05-22 20:40:30 -07:00
Yohann Dudouit 84cc5c7f4b Minor 2020-05-22 20:22:06 -07:00
Yohann Dudouit 8a0724498c Change a bit the logic in FillI. 2020-05-22 20:12:41 -07:00
Yohann Dudouit f9976955cf Fix a bug in CG full assembly. 2020-05-22 19:16:52 -07:00
Yohann Dudouit 8ecb802662 More code cleaning. 2020-05-22 19:01:00 -07:00
Yohann Dudouit 568dff92bb Update ex9pa. 2020-05-22 18:15:35 -07:00
Yohann Dudouit bbc186136d Clean the code. 2020-05-22 18:05:52 -07:00
Yohann Dudouit c2253a9532 Replace size_t with int. 2020-05-22 17:46:12 -07:00
Yohann Dudouit 3309b8d49b Rewrite Full Assembly for DG. 2020-05-22 17:36:58 -07:00
Jan Nikl 9fb292898e Fixed GeometryRefiner::RefineInterior to not append existing integration rules. 2020-05-22 14:58:38 +02:00
Ketan Mittal 15c1ff069c minor 2020-05-21 14:29:06 -07:00
Ketan Mittal 061b70f461 adding missing terms in gradient 2020-05-21 14:21:25 -07:00
Yohann Dudouit a250b07a34 Fix a bug due to unsynchronized host/device array. 2020-05-20 19:48:46 -07:00
Yohann Dudouit b3a06ecbc0 Add host synchronization in the Print of SpMat. 2020-05-20 18:07:44 -07:00
Yohann Dudouit 12a8465047 Uncomment the actual SpMat print. 2020-05-20 17:32:48 -07:00
Yohann Dudouit 15f6269ad4 Add some debugging prints. 2020-05-20 17:30:42 -07:00
Yohann Dudouit e6ed2fafa0 Help the smart class to be less... 2020-05-20 16:59:46 -07:00
Yohann Dudouit 9c37a19c7f Same weird bug somewhere else. 2020-05-20 16:56:42 -07:00
Yohann Dudouit 27a3f4bfce Forgot one line in previous commit... 2020-05-20 16:50:04 -07:00
Yohann Dudouit 5def286b2a Fix strange bug. 2020-05-20 16:49:07 -07:00
Yohann Dudouit b5aa280711 Use MFEM_FORALL... 2020-05-20 16:28:55 -07:00
Yohann Dudouit ec8bcb8c16 More of the same. 2020-05-20 16:15:13 -07:00
Yohann Dudouit 8d5249c9ba Add MFEM_HOST_DEVICE qualifiers. 2020-05-20 16:13:44 -07:00
Yohann Dudouit 6cc9989653 Remove 'private' qualifier due to nvcc. 2020-05-20 16:11:57 -07:00
Yohann Dudouit fe59dc5f29 Remove commented old code. 2020-05-20 16:08:43 -07:00
Yohann Dudouit fd7b8c84e0 Comment unused code. 2020-05-20 16:06:03 -07:00
Yohann Dudouit 5bea192dd7 Minor for GPU. 2020-05-20 16:04:36 -07:00
Yohann Dudouit 596306b12d Add MFEM_FORALL. 2020-05-20 15:45:04 -07:00
Yohann Dudouit c5dea1ee17 Add temporarly modified ex1, ex9, and add ex9pa for testing purpose. 2020-05-20 15:26:27 -07:00
Yohann Dudouit 9c09fea06b Remove dead code. 2020-05-20 14:39:50 -07:00
Yohann Dudouit bc28f6f06e Replace += with mfemAtomicAdd. 2020-05-20 14:39:37 -07:00
Yohann Dudouit 76b044ae99 Fix bugs in the full assembly for DG. 2020-05-20 11:40:25 -07:00
Yohann Dudouit 27b4be77a9 Use Array<int> instead of Vector. 2020-05-20 11:39:22 -07:00
Dylan Copeland 039a04b3e2 New BlockOperator implementation based on using MakeRef in BlockVector. 2020-05-19 16:52:13 -07:00
Jonathan Wong ceba505e4e merged with master 2020-05-19 15:12:10 -07:00
Jonathan Wong ed556b5c63 tried to fix all line formatting to 80 cols. Added some more documentation and used QuadratureSensitivity in the advection_diffusion example 2020-05-19 15:08:24 -07:00
Yohann Dudouit 6f34ccec75 Change algorithms to assemble CG sparse matrices. 2020-05-19 15:00:59 -07:00
Dylan Copeland feb79f3d56 Merge branch 'master' of github.com:mfem/mfem into blockop_cuda 2020-05-19 14:34:54 -07:00
Yohann Dudouit 35704d508d Fix a bug in Element Assembly.
Integrators were not adding values.
2020-05-19 13:26:25 -07:00
Jonathan Wong 721ea4323b fixed documenation from SUNImplicitSetupB and moved ex9p to advection_diffusion.cpp 2020-05-19 12:30:03 -07:00
Jonathan Wong 8e36f5ebdc made sure linalg/operator.hpp is under 80 col 2020-05-19 12:25:25 -07:00
Ido Akkerman df650aab6b Switching from Array to std::vector for non-POD 2020-05-19 14:15:51 +02:00
Yohann Dudouit 4415622c99 Change the algorithm to initialize I for CG. 2020-05-18 14:04:07 -07:00
Yohann Dudouit 479a70c65f Fix some bugs 2020-05-18 14:03:32 -07:00
Ketan Mittal 83c48a33a3 Merge branch 'tmop-solvers-dev' of https://github.com/mfem/mfem 2020-05-18 07:30:50 -07:00
Ido Akkerman 21d77c738d Adding NULL vector as return value to make compilers happy 2020-05-14 11:34:41 +02:00
Ido Akkerman 10ec2a1818 Small typo 2020-05-12 15:35:35 +02:00
Ido Akkerman 49469131c2 Merge branch 'master' into ode-state-io-dev 2020-05-12 15:19:01 +02:00
Ido Akkerman 2dbb377f91 Adding alternative GetVector mechanism 2020-05-12 14:20:56 +02:00
Ido Akkerman 539176a2e1 Add override keyword 2020-05-12 13:11:22 +02:00
Tzanio Kolev e987383708 Merge branch 'master' into blockop_cuda 2020-05-11 17:15:35 -07:00
Jonathan Wong 0db220c79d started cleaning up the ex9p_adjoint example 2020-05-08 16:14:21 -07:00
Jonathan Wong 73d1d2a1f2 fixed quoting issue over multiple lines in operator.hpp 2020-05-08 16:04:34 -07:00
Jonathan Wong 185fc97786 Merge branch 'master' of github.com:mfem/mfem into CVODESSolver-dev 2020-05-08 15:48:09 -07:00
Jonathan Wong 146205eba6 under 80 char width 2020-05-08 15:26:45 -07:00
Yohann Dudouit d76811ac9a Skeleton for SpMat assembly. 2020-05-08 15:20:11 -07:00
Jonathan Wong 343f9e02a6 changed formating to 80 char width 2020-05-08 15:19:36 -07:00
Jonathan Wong 56d6efd7c6 updated .gitignore for miniapps/adjoint 2020-05-08 15:16:36 -07:00
Jonathan Wong 211738b616 formated CHANGELOG to 80 char width 2020-05-08 15:13:36 -07:00
Jonathan Wong 9e16d2c109 Moved files over to a miniapps directory and renamed ex26 to cvsRobers_ASAi_dns for the time being 2020-05-08 15:11:20 -07:00
Jonathan Wong 62f0f65d05 @gardner48 suggested removing the parhyp calls from hypre as they are not used anymore in the MFEM interfaces 2020-05-08 14:58:04 -07:00
Dylan Copeland 1ec02f0b23 Changed BlockOperator and BlockDiagonalPreconditioner to copy data to and from separate vectors for each block multiplication, in order to work on device. Updated MINRES for device runs. Now ex5 works on device. 2020-05-08 13:20:37 -07:00
Jonathan Wong c25845e724 Changed RobertsSUNDIALS to RobertsTDAOperator to highlight that it is a TimeDependentAdjointOperator which implements adjoint related rate equations as well as the time dependent operator rate equation 2020-05-07 21:31:06 -07:00
Jonathan Wong b0a7fd45e6 Made suggested documentation changes to test/ex26.cpp 2020-05-07 21:25:02 -07:00
Jonathan Wong 213b6e51d3 Added documentation for building with cvodes to default files. Changed name from objective sensitivity to quadrature sensitivity for consistency. Added documentation to sundials files. 2020-05-07 21:24:30 -07:00
Jonathan Wong e1f47f0079 corrected sundials makefile to use new test naming scheme 2020-05-07 15:55:41 -07:00
Jonathan Wong 91ed44e45a changed names of test.cpp and test_advDiff.cpp to ex26.cpp and ex27p.cpp. Replaced cvode with cvodes for SUNDIALS throughout. 2020-05-07 15:49:24 -07:00
Jonathan Wong 9a0f991da6 added NumSteps calls which might be useful for controlling aspects of stepsizes 2020-05-07 13:26:35 -07:00
Dylan Copeland 7cc35b68c2 Fixed a bug in a unit test. Defined MAX_D1D and MAX_Q1D in some functions that were using definitions from forall.hpp. 2020-05-07 09:38:27 -07:00
Dylan Copeland 525fb59dd2 Adding PA for mixed curl integrator with RT test functions, with sample runs in ex24. 2020-05-06 10:32:43 -07:00
Dylan Copeland ab74459790 Adding curl interpolation example to ex24/ex24p. 2020-05-05 11:18:55 -07:00
Dylan Copeland ec745ebb25 Adding PA implementations for mixed H(curl)-vector L2 forms, along with unit tests. Also extending PA for some vector mass and curl integrators to handle diagonal matrix coefficients. 2020-05-04 21:26:01 -07:00
Arturo Vargas b686dbb897 Update bilininteg_mass.cpp 2020-04-30 18:10:29 -07:00
Yohann Dudouit bfc4484715 Initial commit for Full Assembly 2020-04-28 15:54:11 -07:00
Vargas dfbb139273 make style 2020-04-27 22:25:00 -07:00
Arturo Vargas e18ae9cbfc user quadrature data 2020-04-27 22:16:37 -07:00
Arturo Vargas 8782faff18 quadrature coefficient 2020-04-27 21:11:16 -07:00
Arturo Vargas bcde578a5d Merge branch 'master' into feature/artv3/quad-data 2020-04-27 21:07:11 -07:00
Arturo Vargas 12609ea7dd dismiss changes 2020-04-27 21:06:50 -07:00
camierjs abc671eb04 Merge branch 'master' into hip-defs 2020-04-21 13:41:24 -07:00
Ido Akkerman 288cef9ccf Fine tunning the number of states available 2020-04-21 10:59:48 +02:00
camierjs c6f122e392 Qualifiers for device inline functions 2020-04-20 12:08:24 -07:00
camierjs 73de0369a1 Half & half B & G to save enough smem for order 8 2020-04-20 11:29:24 -07:00
camierjs 2d7c1c6756 SmemPADiffusion2Apply3D order 8 2020-04-20 11:06:04 -07:00
camierjs 64b51e3e2b SmemPADiffusion2Apply3D orders tries 2020-04-20 10:57:54 -07:00
camierjs b5e1a0732c Diffusion with shared mem & unroll 2020-04-20 10:39:19 -07:00
Ido Akkerman 27879297e4 Make style 2020-04-20 16:40:40 +02:00
Ido Akkerman b3babbef60 Adding state vector access mechanism for 2nd order to unit tests 2020-04-20 16:38:47 +02:00
Ido Akkerman 7909fab83f Adding state vector access mechanism for 1st order to unit tests 2020-04-20 16:28:19 +02:00
Ido Akkerman 9d0015949a Adding state vector access mechanism for first order ODE solvers 2020-04-20 16:27:28 +02:00
camierjs bfabd546fd Introduce MFEM_LAMBDA for the diffs CUDA vs HIP 2020-04-19 17:59:30 -07:00
camierjs c2796cd1c7 Lambda scope to getaround tile_static used in a non-amp error 2020-04-19 14:34:50 -07:00
camierjs 72ffdecfa6 Create backends.hpp to centralize defines 2020-04-19 13:01:37 -07:00
camierjs d765407dbe Revert logic try 2020-04-18 18:06:58 -07:00
camierjs f6c8898506 Not Cuda or Hip logic 2020-04-18 17:39:07 -07:00
camierjs 8f0817c621 Cuda header sync 2020-04-18 14:55:39 -07:00
camierjs 2ffb058e61 Use __HIP_DEVICE_COMPILE__ 2020-04-18 14:52:35 -07:00
Jonathan Wong 023e63dfb1 Reran astyle with astyle 2.05.1 2020-03-30 12:51:23 -07:00
Jonathan Wong e63ac279c1 Merge branch 'master' of github.com:mfem/mfem into CVODESSolver-dev
Fixed merge conflict in linalg/operator.hpp.
Cleaned up linalg/sundials.cpp a little bit.
Reverted defaults.mk change.
2020-03-30 12:49:40 -07:00
Jonathan Wong 8b7f15cdca revert astyle changes 2020-03-30 12:45:43 -07:00
Jonathan Wong a263658bdb Started working on CONTRIBUTING.md list 2020-03-27 12:09:53 -07:00
Jonathan Wong 3f4a6ce0d2 ran astyle 2020-03-25 12:58:46 -07:00
Jonathan Wong 4ee2b18d97 Changed from SetN_Vector to .ToNVector. Fixed implementation as suggested by gardner
Added an extra argument to Vector::ToNVector() to better handle parallel vectors
2020-03-25 12:58:33 -07:00
Jonathan Wong 5f3f056703 Initial commits of changes suggested by the SUNDIALS team 2020-01-14 13:53:05 -08:00
Jonathan Wong c933973249 Initial CVODESSolver implementation. Simple serial and parallel problem solved. Added Rootfunction finding and error control support 2020-01-06 15:40:05 -08:00
Vargas 7ee2335810 first pass at block diagonal mass matrix 2019-11-23 10:52:27 -08:00
Vargas 3c064fb4af Merge branch 'master' into feature/artv3/quad-data 2019-11-22 13:29:07 -08:00
artv3 6629cb4adb proof of concept for scaling 2019-11-01 17:48:56 -07:00
228 changed files with 10825 additions and 33080 deletions
+7 -7
View File
@@ -16,7 +16,7 @@ install:
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
# Install METIS
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
- cd metis-5.1.0
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
@@ -26,17 +26,17 @@ install:
- cd ..
# Install hypre
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/V2-10-0b.tar.gz'
- 7z x V2-10-0b.tar.gz -so | 7z x -si -ttar > nul
- cd hypre-2-10-0b
- cmake -H. -Bbuild -DHYPRE_USING_FEI=OFF -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/v2.19.0.tar.gz'
- 7z x v2.19.0.tar.gz -so | 7z x -si -ttar > nul
- cd hypre-2.19.0/src
- cmake -H. -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
- cmake --build build
- cmake --build build --target install
- cd ..
- cd ../..
# MFEM
before_build:
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2-10-0b\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2-10-0b\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_DIR=%cd%\hypre-2.19.0\src\hypre -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE
build_script:
+4 -14
View File
@@ -50,6 +50,7 @@ examples/ex1[04-9]
examples/ex1[0-9]p
examples/ex2[0-9]
examples/ex2[0-9]p
examples/ex25-gpu
examples/refined.mesh
examples/displaced.mesh
@@ -163,20 +164,6 @@ miniapps/electromagnetics/Tesla-AMR*
miniapps/electromagnetics/Maxwell-Parallel*
miniapps/electromagnetics/Joule_*
miniapps/hypsys/build
miniapps/hypsys/errors.txt
miniapps/hypsys/grid*
miniapps/hypsys/hypsys
miniapps/hypsys/initial*
miniapps/hypsys/output
miniapps/hypsys/phypsys
miniapps/hypsys/pressure*
miniapps/hypsys/results
miniapps/hypsys/scripts/gridfunc-scatter
miniapps/hypsys/ultimate*
miniapps/hypsys/velocity*
miniapps/hypsys/various
miniapps/meshing/mobius-strip
miniapps/meshing/klein-bottle
miniapps/meshing/toroid
@@ -258,6 +245,9 @@ miniapps/navier/navier_3dfoc
miniapps/navier/tgv_out*.txt
miniapps/navier/*_output
miniapps/adjoint/cvsRoberts_ASAi_dns
miniapps/adjoint/adjoint_advection_diffusion
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
+21 -20
View File
@@ -16,6 +16,12 @@ stages:
- tests
- optional
env:
global:
- HYPRE_ARCHIVE=v2.19.0.tar.gz
HYPRE_URL=https://github.com/hypre-space/hypre/archive/$HYPRE_ARCHIVE
HYPRE_TOP_DIR=hypre-2.19.0
jobs:
include:
@@ -138,8 +144,7 @@ jobs:
NPROCS=2
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -169,8 +174,7 @@ jobs:
NPROCS=2
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -193,7 +197,7 @@ jobs:
- cd ${TRAVIS_BUILD_DIR}/build
- cmake ..
-DMFEM_USE_MPI=ON
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../hypre-2.10.0b/src/hypre
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../$HYPRE_TOP_DIR/src/hypre
-DMFEM_MPI_NP=$NPROCS
- make -j3 mfem examples
- cd ${TRAVIS_BUILD_DIR}/build/tests/unit
@@ -201,8 +205,7 @@ jobs:
- ctest --output-on-failure
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -247,8 +250,7 @@ jobs:
TMPDIR=/tmp
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
@@ -268,8 +270,7 @@ jobs:
TMPDIR=/tmp
cache:
directories:
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
@@ -335,18 +336,18 @@ install:
# hypre
- if [ $MPI == "YES" ]; then
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
rm -rf hypre-2.10.0b;
tar xvzf hypre-2.10.0b.tar.gz;
cd hypre-2.10.0b/src;
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
if [ ! -e $HYPRE_TOP_DIR/src/hypre/lib/libHYPRE.a ]; then
wget $HYPRE_URL;
rm -rf $HYPRE_TOP_DIR;
tar xvzf $HYPRE_ARCHIVE;
cd $HYPRE_TOP_DIR/src;
./configure --disable-fortran CC=mpicc CXX=mpic++;
make -j3;
cd ../..;
else
echo "Reusing cached hypre-2.10.0b/";
echo "Reusing cached $HYPRE_TOP_DIR/";
fi;
ln -s hypre-2.10.0b hypre;
ln -s $HYPRE_TOP_DIR hypre;
else
echo "Serial build, not using hypre";
fi
@@ -354,7 +355,7 @@ install:
# METIS
- if [ $MPI == "YES" ]; then
if [ ! -e metis-4.0/libmetis.a ]; then
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
wget https://mfem.github.io/tpls/metis-4.0.3.tar.gz;
tar xvzf metis-4.0.3.tar.gz;
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
rm -rf metis-4.0;
+35 -5
View File
@@ -33,6 +33,9 @@ Meshing improvements
for example the periodic-annulus-sector and periodic-torus-sector files in
the data directory.
- Added complete action of the TMOP Integrator to account for the spatial
derivatives of discrete and analytic targets.
Performance improvements
------------------------
- Added support for explicit vectorization in the high-performance templated
@@ -48,6 +51,19 @@ Improved GPU capabilities
-------------------------
- Added support for Chebyshev accelerated polynomial smoother on GPU.
- Optimized AMD/HIP kernel support.
- Added a Full Assembly mode compatible with Device kernel execution. This
assembly level builds on top of the current Element Assembly kernels to
compute a global sparse matrix. All integrators supported by element assembly
are also supported by full assembly. See the '-fa' option in Example 9.
- Added support for BlockOperator on GPU. See the updated Example 5.
- Added partial assembly and GPU support for ComplexOperator,
[Par]ComplexGridFunction, [Par]ComplexLinearForm, and [Par]SesquilinearForm.
See the updated Example 22.
Discretization improvements
---------------------------
- Added support for matrix-free interpolation and restriction operators between
@@ -71,7 +87,7 @@ Discretization improvements
and, in the continuous field case, arbitrary mesh edges and faces.
- Added new coefficient and vector coefficient classes for QuadratureFunctions.
Additionaly, new LinearForm integrators were also added which make use of
Additionally, new LinearForm integrators were also added which make use of
these new QuadratureFunction coefficient classes.
- Added support face integrals on the boundaries of NURBS meshes.
@@ -88,6 +104,10 @@ Linear and nonlinear solvers
and solution during the solving process of an IterativeSolver after every
iteration.
- Added support for the CVODES package in SUNDIALS which provides ODE
solvers with sensitivity analysis capabilities. See the CVODESSolver
class and the new adjoint miniapps below.
- Block arrays of parallel matrices can now be merged into a single parallel
matrix with the function HypreParMatrixFromBlocks. This could be useful for
solving block systems with parallel direct solvers such as STRUMPACK.
@@ -115,6 +135,14 @@ New and updated examples and miniapps
equations of incompressible fluid dynamics. See the miniapps/navier directory
for more details.
- Added a new miniapps/adjoint directory with two miniapps demonstrating how to
solve adjoint problems in MFEM using the CVODES package in SUNDIALS. Both of
these miniapps require the MFEM_USE_SUNDIALS configuration option.
* The cvsRoberts_ASAi_dns miniapp solves a backward adjoint problem for a
system of ODEs, evaluating both forward and adjoint quadratures in serial.
* The adjoint_advection_diffusion miniapp solves a backward adjoint problem
for an advection diffusion PDE, evaluating adjoint quadratures in parallel.
- Ported Example 11p to SLEPc, to demonstrate solving the Laplace eigenvalue
equation with the shift-and-invert spectral transformation method.
@@ -125,11 +153,10 @@ New and updated examples and miniapps
- Added a new meshing miniapp, Minimal Surface, which solves Plateau's problem:
the Dirichlet problem for the minimal surface equation.
- Added partial assembly support to examples 4/4p and 5/5p, with diagonal
preconditioning.
- Added full assembly support in Example 9/9p.
- Added a new test problem in example 24/24p, demonstrating a mixed bilinear
form for H(div) and L_2, with partial assembly support.
- Added a new test problem in Example 24/24p, demonstrating a mixed bilinear
form for H1, H(curl), H(div) and L_2, with partial assembly support.
- Added weak Dirichlet boundary conditions (Nitsche) to the NURBS miniapp.
@@ -137,6 +164,9 @@ New and updated examples and miniapps
mesh based on element attributes. Any newly exposed boundary elements are
assigned attribute numbers related to the trimmed element attributes.
- Added partial assembly and device support to Example 4/4p, Example 5/5p,
Example 22/22p, and Example 25/25p, with diagonal preconditioning.
Improved testing
----------------
- Added a GitLab pipeline that automates PR testing on supercomputing systems
+2 -2
View File
@@ -211,10 +211,10 @@ endif()
# SUNDIALS
if (MFEM_USE_SUNDIALS)
if (NOT MFEM_USE_MPI)
find_package(SUNDIALS REQUIRED NVector_Serial CVODE ARKODE KINSOL)
find_package(SUNDIALS REQUIRED NVector_Serial CVODES ARKODE KINSOL)
else()
find_package(SUNDIALS REQUIRED
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
NVector_Serial NVector_Parallel NVector_ParHyp CVODES ARKODE KINSOL)
endif()
endif()
+1
View File
@@ -109,6 +109,7 @@ The MFEM source code has the following structure:
├── linalg
├── mesh
├── miniapps
│ ├── adjoint
│ ├── common
│ ├── electromagnetics
│ ├── gslib
+1
View File
@@ -25,5 +25,6 @@ mfem_find_package(SUNDIALS SUNDIALS SUNDIALS_DIR
ADD_COMPONENT NVector_ParHyp
"include" nvector/nvector_parhyp.h "lib" sundials_nvecparhyp
ADD_COMPONENT CVODE "include" cvode/cvode.h "lib" sundials_cvode
ADD_COMPONENT CVODES "include" cvodes/cvodes.h "lib" sundials_cvodes
ADD_COMPONENT ARKODE "include" arkode/arkode.h "lib" sundials_arkode
ADD_COMPONENT KINSOL "include" kinsol/kinsol.h "lib" sundials_kinsol)
+2
View File
@@ -88,6 +88,8 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
# and modify cmake variables for hypre for sundials
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-5.0.0/instdir" CACHE PATH
"Path to the SUNDIALS library.")
# The following may be necessary, if SUNDIALS was built with KLU:
+3 -1
View File
@@ -189,10 +189,12 @@ OPENMP_LIB =
POSIX_CLOCKS_LIB = -lrt
# SUNDIALS library configuration
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
# and modify cmake variables for hypre for sundials
SUNDIALS_DIR = @MFEM_DIR@/../sundials-5.0.0/instdir
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib64 -L$(SUNDIALS_DIR)/lib64\
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
-lsundials_arkode -lsundials_cvodes -lsundials_nvecserial -lsundials_kinsol
ifeq ($(MFEM_USE_MPI),YES)
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
+1
View File
@@ -770,6 +770,7 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/examples/pumi \
@MFEM_SOURCE_DIR@/examples/hiop \
@MFEM_SOURCE_DIR@/examples/sundials \
@MFEM_SOURCE_DIR@/miniapps/adjoint \
@MFEM_SOURCE_DIR@/miniapps/common \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/gslib \
+8 -4
View File
@@ -88,8 +88,8 @@ namespace mfem {
* - <a class="el" href="ex24p_8cpp_source.html">Example 24p</a>: parallel mixed finite element spaces and interpolators
* - <a class="el" href="ex25_8cpp_source.html">Example 25</a>: simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
* - <a class="el" href="ex25p_8cpp_source.html">Example 25p</a>: parallel simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
*
* <H4>SUNDIALS Examples</H4>
* - Variants of Examples
@@ -101,6 +101,9 @@ namespace mfem {
* and
* <a class="el" href="sundials_2ex16p_8cpp_source.html">16p</a>
* demonstrating the use of MFEM's \link sundials.hpp SUNDIALS classes\endlink
* - CVODES adjoint examples:
* <a class="el" href="cvsRoberts__ASAi__dns_8cpp_source.html">serial ODE system</a>,
* <a class="el" href="adjoint__advection__diffusion_8cpp_source.html">parallel advection-diffusion</a>
*
* <H4>PETSc Examples</H4>
* - Variants of Examples
@@ -140,7 +143,9 @@ namespace mfem {
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
* - <a class="el" href="toroid_8cpp_source.html">Toroid</a>: generate simple toroidal meshes
* - <a class="el" href="twist_8cpp_source.html">Twist</a>: generate simple periodic meshes
@@ -157,7 +162,6 @@ namespace mfem {
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
*
+34 -31
View File
@@ -103,8 +103,8 @@ int main(int argc, char *argv[])
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
Mesh mesh(mesh_file, 1, 1);
int dim = mesh.Dimension();
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
@@ -112,10 +112,10 @@ int main(int argc, char *argv[])
// elements.
{
int ref_levels =
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
(int)floor(log(50000./mesh.GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
mesh.UniformRefinement();
}
}
@@ -123,66 +123,70 @@ int main(int argc, char *argv[])
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (mesh->GetNodes())
else if (mesh.GetNodes())
{
fec = mesh->GetNodes()->OwnFEC();
fec = mesh.GetNodes()->OwnFEC();
delete_fec = false;
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
FiniteElementSpace fespace(&mesh, fec);
cout << "Number of finite element unknowns: "
<< fespace->GetTrueVSize() << endl;
<< fespace.GetTrueVSize() << endl;
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined by marking all
// the boundary attributes from the mesh as essential (Dirichlet) and
// converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (mesh->bdr_attributes.Size())
if (mesh.bdr_attributes.Size())
{
Array<int> ess_bdr(mesh->bdr_attributes.Max());
Array<int> ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 1;
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
// the basis functions in the finite element fespace.
LinearForm *b = new LinearForm(fespace);
LinearForm b(&fespace);
ConstantCoefficient one(1.0);
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
// 8. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
GridFunction x(fespace);
GridFunction x(&fespace);
x = 0.0;
// 9. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
BilinearForm *a = new BilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
BilinearForm a(&fespace);
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a.AddDomainIntegrator(new DiffusionIntegrator(one));
// 10. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: eliminating boundary
// conditions, applying conforming constraints for non-conforming AMR,
// static condensation, etc.
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
if (static_cond) { a.EnableStaticCondensation(); }
a.Assemble();
OperatorPtr A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
cout << "Size of linear system: " << A->Height() << endl;
@@ -203,9 +207,9 @@ int main(int argc, char *argv[])
}
else // Jacobi preconditioning in partial assembly mode
{
if (UsesTensorBasis(*fespace))
if (UsesTensorBasis(fespace))
{
OperatorJacobiSmoother M(*a, ess_tdof_list);
OperatorJacobiSmoother M(a, ess_tdof_list);
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
}
else
@@ -215,13 +219,13 @@ int main(int argc, char *argv[])
}
// 12. Recover the solution as a finite element grid function.
a->RecoverFEMSolution(X, *b, x);
a.RecoverFEMSolution(X, b, x);
// 13. Save the refined mesh and the solution. This output can be viewed later
// using GLVis: "glvis -m refined.mesh -g sol.gf".
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
mesh.Print(mesh_ofs);
ofstream sol_ofs("sol.gf");
sol_ofs.precision(8);
x.Save(sol_ofs);
@@ -233,15 +237,14 @@ int main(int argc, char *argv[])
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << x << flush;
sol_sock << "solution\n" << mesh << x << flush;
}
// 15. Free the used memory.
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete mesh;
if (delete_fec)
{
delete fec;
}
return 0;
}
+37 -35
View File
@@ -112,8 +112,8 @@ int main(int argc, char *argv[])
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
Mesh mesh(mesh_file, 1, 1);
int dim = mesh.Dimension();
// 5. Refine the serial mesh on all processors to increase the resolution. In
// this example we do 'ref_levels' of uniform refinement. We choose
@@ -121,23 +121,23 @@ int main(int argc, char *argv[])
// more than 10,000 elements.
{
int ref_levels =
(int)floor(log(10000./mesh->GetNE())/log(2.)/dim);
(int)floor(log(10000./mesh.GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
mesh.UniformRefinement();
}
}
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
ParMesh pmesh(MPI_COMM_WORLD, mesh);
mesh.Clear();
{
int par_ref_levels = 2;
for (int l = 0; l < par_ref_levels; l++)
{
pmesh->UniformRefinement();
pmesh.UniformRefinement();
}
}
@@ -145,13 +145,16 @@ int main(int argc, char *argv[])
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (pmesh->GetNodes())
else if (pmesh.GetNodes())
{
fec = pmesh->GetNodes()->OwnFEC();
fec = pmesh.GetNodes()->OwnFEC();
delete_fec = false;
if (myid == 0)
{
cout << "Using isoparametric FEs: " << fec->Name() << endl;
@@ -160,9 +163,10 @@ int main(int argc, char *argv[])
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
HYPRE_Int size = fespace->GlobalTrueVSize();
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_Int size = fespace.GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl;
@@ -173,44 +177,44 @@ int main(int argc, char *argv[])
// by marking all the boundary attributes from the mesh as essential
// (Dirichlet) and converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (pmesh->bdr_attributes.Size())
if (pmesh.bdr_attributes.Size())
{
Array<int> ess_bdr(pmesh->bdr_attributes.Max());
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
ess_bdr = 1;
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 9. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system, which in this case is
// (1,phi_i) where phi_i are the basis functions in fespace.
ParLinearForm *b = new ParLinearForm(fespace);
ParLinearForm b(&fespace);
ConstantCoefficient one(1.0);
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
// 10. Define the solution vector x as a parallel finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
ParGridFunction x(fespace);
ParGridFunction x(&fespace);
x = 0.0;
// 11. Set up the parallel bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
ParBilinearForm *a = new ParBilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
ParBilinearForm a(&fespace);
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a.AddDomainIntegrator(new DiffusionIntegrator(one));
// 12. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, static condensation, etc.
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
if (static_cond) { a.EnableStaticCondensation(); }
a.Assemble();
OperatorPtr A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
@@ -218,9 +222,9 @@ int main(int argc, char *argv[])
Solver *prec = NULL;
if (pa)
{
if (UsesTensorBasis(*fespace))
if (UsesTensorBasis(fespace))
{
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
else
@@ -238,7 +242,7 @@ int main(int argc, char *argv[])
// 14. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(X, *b, x);
a.RecoverFEMSolution(X, b, x);
// 15. Save the refined mesh and the solution in parallel. This output can
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
@@ -249,7 +253,7 @@ int main(int argc, char *argv[])
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh->Print(mesh_ofs);
pmesh.Print(mesh_ofs);
ofstream sol_ofs(sol_name.str().c_str());
sol_ofs.precision(8);
@@ -264,16 +268,14 @@ int main(int argc, char *argv[])
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << *pmesh << x << flush;
sol_sock << "solution\n" << pmesh << x << flush;
}
// 17. Free the used memory.
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete pmesh;
if (delete_fec)
{
delete fec;
}
MPI_Finalize();
return 0;
+63 -45
View File
@@ -13,6 +13,11 @@
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
//
// Device sample runs:
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1 -pa -d cuda
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2 -pa -d cuda
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0 -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
@@ -76,6 +81,8 @@ int main(int argc, char *argv[])
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -106,6 +113,10 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (!args.Good())
{
@@ -135,13 +146,18 @@ int main(int argc, char *argv[])
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 2. Read the mesh from the given mesh file. We can handle triangular,
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes
// with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Refine the mesh to increase resolution. In this example we do
// 4. Refine the mesh to increase resolution. In this example we do
// 'ref_levels' of uniform refinement where the user specifies
// the number of levels with the '-r' option.
for (int l = 0; l < ref_levels; l++)
@@ -149,7 +165,7 @@ int main(int argc, char *argv[])
mesh->UniformRefinement();
}
// 4. Define a finite element space on the mesh. Here we use continuous
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange, Nedelec, or Raviart-Thomas finite elements of the specified
// order.
if (dim == 1 && prob != 0 )
@@ -171,7 +187,7 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << fespace->GetTrueVSize()
<< endl;
// 5. Determine the list of true (i.e. conforming) essential boundary dofs.
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined based on the type
// of mesh and the problem type.
Array<int> ess_tdof_list;
@@ -183,12 +199,12 @@ int main(int argc, char *argv[])
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 6. Set up the linear form b(.) which corresponds to the right-hand side of
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
ComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 7. Define the solution vector u as a complex finite element grid function
// 8. Define the solution vector u as a complex finite element grid function
// corresponding to fespace. Initialize u with initial guess of 1+0i or
// the exact solution if it is known.
ComplexGridFunction u(fespace);
@@ -210,7 +226,6 @@ int main(int argc, char *argv[])
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
@@ -263,7 +278,7 @@ int main(int argc, char *argv[])
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 8. Set up the sesquilinear form a(.,.) on the finite element space
// 9. Set up the sesquilinear form a(.,.) on the finite element space
// corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
@@ -282,6 +297,7 @@ int main(int argc, char *argv[])
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -305,7 +321,7 @@ int main(int argc, char *argv[])
default: break; // This should be unreachable
}
// 8a. Set up the bilinear form for the preconditioner corresponding to the
// 9a. Set up the bilinear form for the preconditioner corresponding to the
// appropriate operator
//
// 0) A scalar H1 field
@@ -318,6 +334,8 @@ int main(int argc, char *argv[])
// -Grad(a Div) - omega^2 b + omega c
//
BilinearForm *pcOp = new BilinearForm(fespace);
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -338,9 +356,9 @@ int main(int argc, char *argv[])
default: break; // This should be unreachable
}
// 9. Assemble the form and the corresponding linear system, applying any
// necessary transformations such as: assembly, eliminating boundary
// conditions, conforming constraints for non-conforming AMR, etc.
// 10. Assemble the form and the corresponding linear system, applying any
// necessary transformations such as: assembly, eliminating boundary
// conditions, conforming constraints for non-conforming AMR, etc.
a->Assemble();
pcOp->Assemble();
@@ -348,28 +366,17 @@ int main(int argc, char *argv[])
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
cout << "Size of linear system: " << A->Width() << endl << endl;
{
ComplexSparseMatrix * Asp =
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Asp->real().Width() << endl << endl;
}
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
// 11. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the appropriate sparse smoother.
{
Array<int> blockOffsets;
blockOffsets.SetSize(3);
blockOffsets[0] = 0;
blockOffsets[1] = PCOp.Ptr()->Height();
blockOffsets[2] = PCOp.Ptr()->Height();
blockOffsets[1] = A->Height() / 2;
blockOffsets[2] = A->Height() / 2;
blockOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockOffsets);
@@ -377,22 +384,31 @@ int main(int argc, char *argv[])
Operator * pc_r = NULL;
Operator * pc_i = NULL;
double s = 1.0;
switch (prob)
if (pa)
{
case 0:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
s = -1.0;
break;
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
default: break; // This should be unreachable
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
}
else
{
OperatorHandle PCOp;
pcOp->SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
switch (prob)
{
case 0:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
break;
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
default:
break; // This should be unreachable
}
}
double s = (prob != 1) ? 1.0 : -1.0;
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
s:-s);
@@ -410,9 +426,11 @@ int main(int argc, char *argv[])
gmres.Mult(B, U);
}
// 11. Recover the solution as a finite element grid function and compute the
// 12. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a->RecoverFEMSolution(U, b, u);
u.real().SyncMemory(u);
u.imag().SyncMemory(u);
if (exact_sol)
{
@@ -442,7 +460,7 @@ int main(int argc, char *argv[])
cout << endl;
}
// 12. Save the refined mesh and the solution. This output can be viewed
// 13. Save the refined mesh and the solution. This output can be viewed
// later using GLVis: "glvis -m mesh -g sol".
{
ofstream mesh_ofs("refined.mesh");
@@ -457,7 +475,7 @@ int main(int argc, char *argv[])
u.imag().Save(sol_i_ofs);
}
// 13. Send the solution by socket to a GLVis server.
// 14. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -516,7 +534,7 @@ int main(int argc, char *argv[])
}
}
// 14. Free the used memory.
// 15. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
+66 -47
View File
@@ -13,6 +13,11 @@
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
//
// Device sample runs:
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 1 -p 1 -pa -d cuda
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 1 -p 2 -pa -d cuda
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0 -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
@@ -41,7 +46,6 @@
// We recommend viewing examples 1, 3 and 4 before viewing this
// example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
@@ -84,6 +88,8 @@ int main(int argc, char *argv[])
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -116,6 +122,10 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (!args.Good())
{
@@ -152,19 +162,24 @@ int main(int argc, char *argv[])
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 3. Read the (serial) mesh from the given mesh file on all processors. We
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the serial mesh on all processors to increase the resolution.
// 5. Refine the serial mesh on all processors to increase the resolution.
for (int l = 0; l < ser_ref_levels; l++)
{
mesh->UniformRefinement();
}
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
@@ -174,7 +189,7 @@ int main(int argc, char *argv[])
pmesh->UniformRefinement();
}
// 6. Define a parallel finite element space on the parallel mesh. Here we
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange, Nedelec, or Raviart-Thomas finite elements of
// the specified order.
if (dim == 1 && prob != 0 )
@@ -202,7 +217,7 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << size << endl;
}
// 7. Determine the list of true (i.e. parallel conforming) essential
// 8. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the type of mesh and the problem type.
Array<int> ess_tdof_list;
@@ -214,14 +229,14 @@ int main(int argc, char *argv[])
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 8. Set up the parallel linear form b(.) which corresponds to the
// 9. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system.
ParComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 9. Define the solution vector u as a parallel complex finite element grid
// function corresponding to fespace. Initialize u with initial guess of
// 1+0i or the exact solution if it is known.
// 10. Define the solution vector u as a parallel complex finite element grid
// function corresponding to fespace. Initialize u with initial guess of
// 1+0i or the exact solution if it is known.
ParComplexGridFunction u(fespace);
ParComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ParComplexGridFunction(fespace); }
@@ -241,7 +256,6 @@ int main(int argc, char *argv[])
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
@@ -296,7 +310,7 @@ int main(int argc, char *argv[])
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 10. Set up the parallel sesquilinear form a(.,.) on the finite element
// 11. Set up the parallel sesquilinear form a(.,.) on the finite element
// space corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
@@ -315,6 +329,7 @@ int main(int argc, char *argv[])
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -338,7 +353,7 @@ int main(int argc, char *argv[])
default: break; // This should be unreachable
}
// 10a. Set up the parallel bilinear form for the preconditioner
// 11a. Set up the parallel bilinear form for the preconditioner
// corresponding to the appropriate operator
//
// 0) A scalar H1 field
@@ -351,6 +366,7 @@ int main(int argc, char *argv[])
// -Grad(a Div) - omega^2 b + omega c
//
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -371,7 +387,7 @@ int main(int argc, char *argv[])
default: break; // This should be unreachable
}
// 11. Assemble the parallel bilinear form and the corresponding linear
// 12. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, etc.
@@ -382,30 +398,22 @@ int main(int argc, char *argv[])
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
if (myid == 0)
{
ComplexHypreParMatrix * Ahyp =
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
<< 2 * fespace->GlobalTrueVSize() << endl << endl;
}
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
// 13. Define and apply a parallel FGMRES solver for AU=B with a block
// diagonal preconditioner based on the appropriate multigrid
// preconditioner from hypre.
{
Array<int> blockTrueOffsets;
blockTrueOffsets.SetSize(3);
blockTrueOffsets[0] = 0;
blockTrueOffsets[1] = PCOp.Ptr()->Height();
blockTrueOffsets[2] = PCOp.Ptr()->Height();
blockTrueOffsets[1] = A->Height() / 2;
blockTrueOffsets[2] = A->Height() / 2;
blockTrueOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
@@ -413,25 +421,34 @@ int main(int argc, char *argv[])
Operator * pc_r = NULL;
Operator * pc_i = NULL;
switch (prob)
if (pa)
{
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
break;
case 2:
if (dim == 2 )
{
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
}
else
{
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
switch (prob)
{
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
break;
case 2:
if (dim == 2 )
{
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
}
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
@@ -449,9 +466,11 @@ int main(int argc, char *argv[])
fgmres.SetPrintLevel(1);
fgmres.Mult(B, U);
}
// 13. Recover the parallel grid function corresponding to U. This is the
// 14. Recover the parallel grid function corresponding to U. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(U, b, u);
u.real().SyncMemory(u);
u.imag().SyncMemory(u);
if (exact_sol)
{
@@ -484,7 +503,7 @@ int main(int argc, char *argv[])
}
}
// 14. Save the refined mesh and the solution in parallel. This output can be
// 15. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_r_name, sol_i_name;
@@ -504,7 +523,7 @@ int main(int argc, char *argv[])
u.imag().Save(sol_i_ofs);
}
// 15. Send the solution by socket to a GLVis server.
// 16. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -569,7 +588,7 @@ int main(int argc, char *argv[])
}
}
// 16. Free the used memory.
// 17. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
+87 -8
View File
@@ -7,6 +7,7 @@
// ex24 -m ../data/beam-tet.mesh
// ex24 -m ../data/beam-hex.mesh -o 2 -pa
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 1
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 2
// ex24 -m ../data/escher.mesh
// ex24 -m ../data/escher.mesh -o 2
// ex24 -m ../data/fichera.mesh
@@ -24,12 +25,13 @@
// ex24 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code illustrates usage of mixed finite element
// spaces, with two variants:
// spaces, with three variants:
//
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
// 2) (div v, q) for v in H(div) tested against q in L_2
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
// 3) (div v, q) for v in H(div) tested against q in L_2
//
// Using different approaches, we project the gradient or
// Using different approaches, we project the gradient, curl, or
// divergence to the appropriate space.
//
// We recommend viewing examples 1, 3, and 5 before viewing this
@@ -45,8 +47,11 @@ using namespace mfem;
double p_exact(const Vector &x);
void gradp_exact(const Vector &, Vector &);
double div_gradp_exact(const Vector &x);
void v_exact(const Vector &x, Vector &v);
void curlv_exact(const Vector &x, Vector &cv);
int dim;
double freq = 1.0, kappa;
int main(int argc, char *argv[])
{
@@ -65,7 +70,7 @@ int main(int argc, char *argv[])
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H(Curl) or 1: H(Div)");
"Choose between 0: grad, 1: curl, 2: div");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
@@ -83,6 +88,7 @@ int main(int argc, char *argv[])
return 1;
}
args.PrintOptions(cout);
kappa = freq * M_PI;
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
@@ -119,10 +125,15 @@ int main(int argc, char *argv[])
trial_fec = new H1_FECollection(order, dim);
test_fec = new ND_FECollection(order, dim);
}
else if (prob == 1)
{
trial_fec = new ND_FECollection(order, dim);
test_fec = new RT_FECollection(order-1, dim);
}
else
{
trial_fec = new RT_FECollection(order - 1, dim);
test_fec = new L2_FECollection(order - 1, dim);
trial_fec = new RT_FECollection(order-1, dim);
test_fec = new L2_FECollection(order-1, dim);
}
FiniteElementSpace trial_fes(mesh, trial_fec);
@@ -136,6 +147,12 @@ int main(int argc, char *argv[])
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
}
else if (prob == 1)
{
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
endl;
}
else
{
cout << "Number of Raviart-Thomas finite element unknowns: "
@@ -150,12 +167,18 @@ int main(int argc, char *argv[])
GridFunction x(&test_fes);
FunctionCoefficient p_coef(p_exact);
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
VectorFunctionCoefficient v_coef(sdim, v_exact);
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
FunctionCoefficient divgradp_coef(div_gradp_exact);
if (prob == 0)
{
gftrial.ProjectCoefficient(p_coef);
}
else if (prob == 1)
{
gftrial.ProjectCoefficient(v_coef);
}
else
{
gftrial.ProjectCoefficient(gradp_coef);
@@ -179,6 +202,11 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
}
else if (prob == 1)
{
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
}
else
{
a.AddDomainIntegrator(new MassIntegrator(one));
@@ -244,6 +272,10 @@ int main(int argc, char *argv[])
{
dlo.AddDomainInterpolator(new GradientInterpolator());
}
else if (prob == 1)
{
dlo.AddDomainInterpolator(new CurlInterpolator());
}
else
{
dlo.AddDomainInterpolator(new DivergenceInterpolator());
@@ -258,6 +290,10 @@ int main(int argc, char *argv[])
{
exact_proj.ProjectCoefficient(gradp_coef);
}
else if (prob == 1)
{
exact_proj.ProjectCoefficient(curlv_coef);
}
else
{
exact_proj.ProjectCoefficient(divgradp_coef);
@@ -276,10 +312,23 @@ int main(int argc, char *argv[])
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
"||_{L_2} = " << errInterp << '\n' << endl;
" ||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
else if (prob == 1)
{
double errSol = x.ComputeL2Error(curlv_coef);
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
double errProj = exact_proj.ComputeL2Error(curlv_coef);
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in H(div): "
"|| E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
"||_{L_2} = " << errProj << '\n' << endl;
}
else
{
int order_quad = max(2, 2*order+1);
@@ -295,7 +344,7 @@ int main(int argc, char *argv[])
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v "
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
"||_{L_2} = " << errProj << '\n' << endl;
@@ -371,3 +420,33 @@ double div_gradp_exact(const Vector &x)
return 0.0;
}
void v_exact(const Vector &x, Vector &v)
{
if (dim == 3)
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(2));
v(2) = sin(kappa * x(0));
}
else
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(0));
if (x.Size() == 3) { v(2) = 0.0; }
}
}
void curlv_exact(const Vector &x, Vector &cv)
{
if (dim == 3)
{
cv(0) = -kappa * cos(kappa * x(2));
cv(1) = -kappa * cos(kappa * x(0));
cv(2) = -kappa * cos(kappa * x(1));
}
else
{
cv = 0.0;
}
}
+94 -12
View File
@@ -6,7 +6,8 @@
// mpirun -np 4 ex24p -m ../data/square-disc.mesh -o 2
// mpirun -np 4 ex24p -m ../data/beam-tet.mesh
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -p 1 -pa
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 1
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 2
// mpirun -np 4 ex24p -m ../data/escher.mesh
// mpirun -np 4 ex24p -m ../data/escher.mesh -o 2
// mpirun -np 4 ex24p -m ../data/fichera.mesh
@@ -24,12 +25,13 @@
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code illustrates usage of mixed finite element
// spaces, with two variants:
// spaces, with three variants:
//
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
// 2) (div v, q) for v in H(div) tested against q in L_2
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
// 3) (div v, q) for v in H(div) tested against q in L_2
//
// Using different approaches, we project the gradient or
// Using different approaches, we project the gradient, curl, or
// divergence to the appropriate space.
//
// We recommend viewing examples 1, 3, and 5 before viewing this
@@ -45,8 +47,11 @@ using namespace mfem;
double p_exact(const Vector &x);
void gradp_exact(const Vector &, Vector &);
double div_gradp_exact(const Vector &x);
void v_exact(const Vector &x, Vector &v);
void curlv_exact(const Vector &x, Vector &cv);
int dim;
double freq = 1.0, kappa;
int main(int argc, char *argv[])
{
@@ -71,7 +76,7 @@ int main(int argc, char *argv[])
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H(Curl) or 1: H(Div)");
"Choose between 0: grad, 1: curl, 2: div");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
@@ -96,6 +101,7 @@ int main(int argc, char *argv[])
{
args.PrintOptions(cout);
}
kappa = freq * M_PI;
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
@@ -147,10 +153,15 @@ int main(int argc, char *argv[])
trial_fec = new H1_FECollection(order, dim);
test_fec = new ND_FECollection(order, dim);
}
else if (prob == 1)
{
trial_fec = new ND_FECollection(order, dim);
test_fec = new RT_FECollection(order-1, dim);
}
else
{
trial_fec = new RT_FECollection(order - 1, dim);
test_fec = new L2_FECollection(order - 1, dim);
trial_fec = new RT_FECollection(order-1, dim);
test_fec = new L2_FECollection(order-1, dim);
}
ParFiniteElementSpace trial_fes(pmesh, trial_fec);
@@ -166,6 +177,12 @@ int main(int argc, char *argv[])
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
}
else if (prob == 1)
{
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
endl;
}
else
{
cout << "Number of Raviart-Thomas finite element unknowns: "
@@ -181,12 +198,18 @@ int main(int argc, char *argv[])
ParGridFunction x(&test_fes);
FunctionCoefficient p_coef(p_exact);
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
VectorFunctionCoefficient v_coef(sdim, v_exact);
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
FunctionCoefficient divgradp_coef(div_gradp_exact);
if (prob == 0)
{
gftrial.ProjectCoefficient(p_coef);
}
else if (prob == 1)
{
gftrial.ProjectCoefficient(v_coef);
}
else
{
gftrial.ProjectCoefficient(gradp_coef);
@@ -210,6 +233,11 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
}
else if (prob == 1)
{
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
}
else
{
a.AddDomainIntegrator(new MassIntegrator(one));
@@ -293,6 +321,10 @@ int main(int argc, char *argv[])
{
dlo.AddDomainInterpolator(new GradientInterpolator());
}
else if (prob == 1)
{
dlo.AddDomainInterpolator(new CurlInterpolator());
}
else
{
dlo.AddDomainInterpolator(new DivergenceInterpolator());
@@ -307,6 +339,10 @@ int main(int argc, char *argv[])
{
exact_proj.ProjectCoefficient(gradp_coef);
}
else if (prob == 1)
{
exact_proj.ProjectCoefficient(curlv_coef);
}
else
{
exact_proj.ProjectCoefficient(divgradp_coef);
@@ -324,14 +360,30 @@ int main(int argc, char *argv[])
if (myid == 0)
{
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
"||_{L_2} = " << errInterp << '\n' << endl;
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl)"
": || E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad"
" p ||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
}
else if (prob == 1)
{
double errSol = x.ComputeL2Error(curlv_coef);
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
double errProj = exact_proj.ComputeL2Error(curlv_coef);
if (myid == 0)
{
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in "
"H(div): || E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
"||_{L_2} = " << errProj << '\n' << endl;
}
}
else
{
int order_quad = max(2, 2*order+1);
@@ -350,7 +402,7 @@ int main(int argc, char *argv[])
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
"||_{L_2} = " << errInterp << '\n' << endl;
" ||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
"||_{L_2} = " << errProj << '\n' << endl;
}
@@ -436,3 +488,33 @@ double div_gradp_exact(const Vector &x)
return 0.0;
}
void v_exact(const Vector &x, Vector &v)
{
if (dim == 3)
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(2));
v(2) = sin(kappa * x(0));
}
else
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(0));
if (x.Size() == 3) { v(2) = 0.0; }
}
}
void curlv_exact(const Vector &x, Vector &cv)
{
if (dim == 3)
{
cv(0) = -kappa * cos(kappa * x(2));
cv(1) = -kappa * cos(kappa * x(0));
cv(2) = -kappa * cos(kappa * x(1));
}
else
{
cv = 0.0;
}
}
File diff suppressed because it is too large Load Diff
+116 -92
View File
@@ -10,6 +10,10 @@
// ex25 -o 2 -f 8.0 -ref 3 -prob 4 -m ../data/inline-quad.mesh
// ex25 -o 2 -f 2.0 -ref 1 -prob 4 -m ../data/inline-hex.mesh
//
// Device sample runs:
// ex25 -o 2 -f 8.0 -ref 3 -prob 4 -m ../data/inline-quad.mesh -pa -d cuda
// ex25 -o 2 -f 2.0 -ref 1 -prob 4 -m ../data/inline-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple electromagnetic wave
// propagation problem corresponding to the second order
// indefinite Maxwell equation
@@ -82,24 +86,24 @@ public:
};
// Class for returning the PML coefficients of the bilinear form
class PMLMatrixCoefficient : public MatrixCoefficient
class PMLDiagMatrixCoefficient : public VectorCoefficient
{
private:
CartesianPML * pml = nullptr;
void (*Function)(const Vector &, CartesianPML * , DenseMatrix &);
void (*Function)(const Vector &, CartesianPML * , Vector &);
public:
PMLMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
DenseMatrix &),
CartesianPML * pml_)
: MatrixCoefficient(dim), pml(pml_), Function(F)
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
Vector &),
CartesianPML * pml_)
: VectorCoefficient(dim), pml(pml_), Function(F)
{}
virtual void Eval(DenseMatrix &K, ElementTransformation &T,
virtual void Eval(Vector &K, ElementTransformation &T,
const IntegrationPoint &ip)
{
double x[3];
Vector transip(x, 3);
T.Transform(ip, transip);
K.SetSize(height, width);
K.SetSize(vdim);
(*Function)(transip, pml, K);
}
};
@@ -116,13 +120,13 @@ void source(const Vector &x, Vector & f);
// Functions for computing the necessary coefficients after PML stretching.
// J is the Jacobian matrix of the stretching function
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector &D);
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector &D);
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector &D);
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector &D);
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector &D);
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector &D);
Array2D<double> comp_domain_bdr;
Array2D<double> domain_bdr;
@@ -153,6 +157,8 @@ int main(int argc, char *argv[])
double freq = 5.0;
bool herm_conv = true;
bool visualization = 1;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -174,12 +180,21 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (iprob > 4) { iprob = 4; }
prob = (prob_type)iprob;
// 2. Setup the mesh
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Setup the mesh
if (!mesh_file)
{
exact_known = true;
@@ -220,7 +235,7 @@ int main(int argc, char *argv[])
// Setup PML length
Array2D<double> length(dim, 2); length = 0.0;
// 3. Setup the Cartesian PML region.
// 4. Setup the Cartesian PML region.
switch (prob)
{
case disc:
@@ -246,19 +261,19 @@ int main(int argc, char *argv[])
comp_domain_bdr = pml->GetCompDomainBdr();
domain_bdr = pml->GetDomainBdr();
// 4. Refine the mesh to increase the resolution.
// 5. Refine the mesh to increase the resolution.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 5. Reorient mesh in case of a tet mesh
// 6. Reorient mesh in case of a tet mesh
mesh->ReorientTetMesh();
// Set element attributes in order to distinguish elements in the PML region
pml->SetAttributes(mesh);
// 6. Define a finite element space on the mesh. Here we use the Nedelec
// 7. Define a finite element space on the mesh. Here we use the Nedelec
// finite elements of the specified order.
FiniteElementCollection *fec = new ND_FECollection(order, dim);
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
@@ -266,7 +281,7 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << size << endl;
// 7. Determine the list of true essential boundary dofs. In this example,
// 8. Determine the list of true essential boundary dofs. In this example,
// the boundary conditions are defined based on the specific mesh and the
// problem type.
Array<int> ess_tdof_list;
@@ -308,12 +323,12 @@ int main(int argc, char *argv[])
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
// 8. Setup Complex Operator convention
// 9. Setup Complex Operator convention
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 9. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
// 10. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
VectorFunctionCoefficient f(dim, source);
ComplexLinearForm b(fespace, conv);
if (prob == load_src)
@@ -323,7 +338,7 @@ int main(int argc, char *argv[])
b.Vector::operator=(0.0);
b.Assemble();
// 10. Define the solution vector x as a complex finite element grid function
// 11. Define the solution vector x as a complex finite element grid function
// corresponding to fespace.
ComplexGridFunction x(fespace);
x = 0.0;
@@ -331,7 +346,7 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient E_Im(dim, E_bdr_data_Im);
x.ProjectBdrCoefficientTangent(E_Re, E_Im, ess_bdr);
// 11. Set up the sesquilinear form a(.,.)
// 12. Set up the sesquilinear form a(.,.)
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) - omega^2 * epsilon (E,F)
@@ -365,19 +380,19 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_omeg),NULL);
int cdim = (dim == 2) ? 1 : dim;
PMLMatrixCoefficient pml_c1_Re(cdim,detJ_inv_JT_J_Re, pml);
PMLMatrixCoefficient pml_c1_Im(cdim,detJ_inv_JT_J_Im, pml);
ScalarMatrixProductCoefficient c1_Re(muinv,pml_c1_Re);
ScalarMatrixProductCoefficient c1_Im(muinv,pml_c1_Im);
MatrixRestrictedCoefficient restr_c1_Re(c1_Re,attrPML);
MatrixRestrictedCoefficient restr_c1_Im(c1_Im,attrPML);
PMLDiagMatrixCoefficient pml_c1_Re(cdim,detJ_inv_JT_J_Re, pml);
PMLDiagMatrixCoefficient pml_c1_Im(cdim,detJ_inv_JT_J_Im, pml);
ScalarVectorProductCoefficient c1_Re(muinv,pml_c1_Re);
ScalarVectorProductCoefficient c1_Im(muinv,pml_c1_Im);
VectorRestrictedCoefficient restr_c1_Re(c1_Re,attrPML);
VectorRestrictedCoefficient restr_c1_Im(c1_Im,attrPML);
PMLMatrixCoefficient pml_c2_Re(dim, detJ_JT_J_inv_Re,pml);
PMLMatrixCoefficient pml_c2_Im(dim, detJ_JT_J_inv_Im,pml);
ScalarMatrixProductCoefficient c2_Re(omeg,pml_c2_Re);
ScalarMatrixProductCoefficient c2_Im(omeg,pml_c2_Im);
MatrixRestrictedCoefficient restr_c2_Re(c2_Re,attrPML);
MatrixRestrictedCoefficient restr_c2_Im(c2_Im,attrPML);
PMLDiagMatrixCoefficient pml_c2_Re(dim, detJ_JT_J_inv_Re,pml);
PMLDiagMatrixCoefficient pml_c2_Im(dim, detJ_JT_J_inv_Im,pml);
ScalarVectorProductCoefficient c2_Re(omeg,pml_c2_Re);
ScalarVectorProductCoefficient c2_Im(omeg,pml_c2_Im);
VectorRestrictedCoefficient restr_c2_Re(c2_Re,attrPML);
VectorRestrictedCoefficient restr_c2_Im(c2_Im,attrPML);
// Integrators inside the PML region
a.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_Re),
@@ -385,30 +400,29 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_Re),
new VectorFEMassIntegrator(restr_c2_Im));
// 12. Assemble the bilinear form and the corresponding linear system,
// 13. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: assembly, eliminating
// boundary conditions, applying conforming constraints for
// non-conforming AMR, etc.
a.Assemble();
#ifndef MFEM_USE_SUITESPARSE
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
#endif
a.Assemble(0);
OperatorHandle Ah;
OperatorPtr A;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 13. Transform to monolithic SparseMatrix
SparseMatrix *A = Ah.As<ComplexSparseMatrix>()->GetSystemMatrix();
cout << "Size of linear system: " << A->Height() << endl;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
// 14. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUITESPARSE
{
UMFPackSolver solver(*A);
solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
solver.Mult(B, X);
if (pa) { cout << "PA not available with MFEM_USE_SUITESPARSE" << endl; }
ComplexUMFPackSolver csolver(*A.As<ComplexSparseMatrix>());
csolver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
csolver.SetPrintLevel(1);
csolver.Mult(B, X);
}
#else
// 14a. Set up the Bilinear form a(.,.) for the preconditioner
//
// In Comp
@@ -424,45 +438,64 @@ int main(int argc, char *argv[])
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_muinv));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_absomeg));
PMLMatrixCoefficient pml_c1_abs(cdim,detJ_inv_JT_J_abs, pml);
ScalarMatrixProductCoefficient c1_abs(muinv,pml_c1_abs);
MatrixRestrictedCoefficient restr_c1_abs(c1_abs,attrPML);
PMLDiagMatrixCoefficient pml_c1_abs(cdim,detJ_inv_JT_J_abs, pml);
ScalarVectorProductCoefficient c1_abs(muinv,pml_c1_abs);
VectorRestrictedCoefficient restr_c1_abs(c1_abs,attrPML);
PMLMatrixCoefficient pml_c2_abs(dim, detJ_JT_J_inv_abs,pml);
ScalarMatrixProductCoefficient c2_abs(absomeg,pml_c2_abs);
MatrixRestrictedCoefficient restr_c2_abs(c2_abs,attrPML);
PMLDiagMatrixCoefficient pml_c2_abs(dim, detJ_JT_J_inv_abs,pml);
ScalarVectorProductCoefficient c2_abs(absomeg,pml_c2_abs);
VectorRestrictedCoefficient restr_c2_abs(c2_abs,attrPML);
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_abs));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_abs));
if (pa) { prec.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
prec.Assemble();
OperatorHandle PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 14b. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the Gauss-Seidel sparse smoother.
// preconditioner based on the Gauss-Seidel or Jacobi sparse smoother.
Array<int> offsets(3);
offsets[0] = 0;
offsets[1] = fespace->GetTrueVSize();
offsets[2] = fespace->GetTrueVSize();
offsets.PartialSum();
GSSmoother gs00(*PCOpAh.As<SparseMatrix>());
BlockDiagonalPreconditioner BlockGS(offsets);
ScaledOperator gs11(&gs00,
(conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0);
BlockGS.SetDiagonalBlock(0,&gs00);
BlockGS.SetDiagonalBlock(1,&gs11);
Operator *pc_r = nullptr;
Operator *pc_i = nullptr;
int s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
if (pa)
{
// Jacobi Smoother
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
ScaledOperator *d11 = new ScaledOperator(d00, s);
pc_r = d00;
pc_i = d11;
}
else
{
OperatorPtr PCOpAh;
prec.SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// Gauss-Seidel Smoother
GSSmoother *gs00 = new GSSmoother(*PCOpAh.As<SparseMatrix>());
ScaledOperator *gs11 = new ScaledOperator(gs00, s);
pc_r = gs00;
pc_i = gs11;
}
BlockDiagonalPreconditioner BlockDP(offsets);
BlockDP.SetDiagonalBlock(0, pc_r);
BlockDP.SetDiagonalBlock(1, pc_i);
GMRESSolver gmres;
gmres.SetPrintLevel(1);
gmres.SetKDim(200);
gmres.SetMaxIter(2000);
gmres.SetMaxIter(pa ? 5000 : 2000);
gmres.SetRelTol(1e-5);
gmres.SetAbsTol(0.0);
gmres.SetOperator(*A);
gmres.SetPreconditioner(BlockGS);
gmres.SetPreconditioner(BlockDP);
gmres.Mult(B, X);
}
#endif
@@ -474,10 +507,8 @@ int main(int argc, char *argv[])
// If exact is known compute the error
if (exact_known)
{
ComplexGridFunction x_gf(fespace);
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
int order_quad = max(2, 2 * order + 1);
const IntegrationRule *irs[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; ++i)
@@ -573,7 +604,6 @@ int main(int argc, char *argv[])
}
// 18. Free the used memory.
delete A;
delete pml;
delete fespace;
delete fec;
@@ -771,7 +801,7 @@ void E_bdr_data_Im(const Vector &x, Vector &E)
}
}
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det(1.0, 0.0);
@@ -782,14 +812,13 @@ void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (det / pow(dxs[i], 2)).real();
D(i) = (det / pow(dxs[i], 2)).real();
}
}
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -800,14 +829,13 @@ void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (det / pow(dxs[i], 2)).imag();
D(i) = (det / pow(dxs[i], 2)).imag();
}
}
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -818,14 +846,13 @@ void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = abs(det / pow(dxs[i], 2));
D(i) = abs(det / pow(dxs[i], 2));
}
}
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det(1.0, 0.0);
@@ -839,19 +866,18 @@ void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
// in the 2D case the coefficient is scalar 1/det(J)
if (dim == 2)
{
M = (1.0 / det).real();
D = (1.0 / det).real();
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (pow(dxs[i], 2) / det).real();
D(i) = (pow(dxs[i], 2) / det).real();
}
}
}
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -864,19 +890,18 @@ void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
if (dim == 2)
{
M = (1.0 / det).imag();
D = (1.0 / det).imag();
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (pow(dxs[i], 2) / det).imag();
D(i) = (pow(dxs[i], 2) / det).imag();
}
}
}
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector &D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -889,14 +914,13 @@ void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
if (dim == 2)
{
M = abs(1.0 / det);
D = abs(1.0 / det);
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = abs(pow(dxs[i], 2) / det);
D(i) = abs(pow(dxs[i], 2) / det);
}
}
}
+115 -93
View File
@@ -10,6 +10,10 @@
// mpirun -np 4 ex25p -o 2 -f 8.0 -rs 2 -rp 2 -prob 4 -m ../data/inline-quad.mesh
// mpirun -np 4 ex25p -o 2 -f 2.0 -rs 1 -rp 1 -prob 4 -m ../data/inline-hex.mesh
//
// Device sample runs:
// mpirun -np 4 ex25p -o 1 -f 3.0 -rs 3 -rp 1 -prob 2 -pa -d cuda
// mpirun -np 4 ex25p -o 2 -f 1.0 -rs 1 -rp 1 -prob 3 -pa -d cuda
//
// Description: This example code solves a simple electromagnetic wave
// propagation problem corresponding to the second order
// indefinite Maxwell equation
@@ -82,24 +86,24 @@ public:
};
// Class for returning the PML coefficients of the bilinear form
class PMLMatrixCoefficient : public MatrixCoefficient
class PMLDiagMatrixCoefficient : public VectorCoefficient
{
private:
CartesianPML * pml = nullptr;
void (*Function)(const Vector &, CartesianPML * , DenseMatrix &);
void (*Function)(const Vector &, CartesianPML * , Vector &);
public:
PMLMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
DenseMatrix &),
CartesianPML * pml_)
: MatrixCoefficient(dim), pml(pml_), Function(F)
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
Vector &),
CartesianPML * pml_)
: VectorCoefficient(dim), pml(pml_), Function(F)
{}
virtual void Eval(DenseMatrix &K, ElementTransformation &T,
virtual void Eval(Vector &K, ElementTransformation &T,
const IntegrationPoint &ip)
{
double x[3];
Vector transip(x, 3);
T.Transform(ip, transip);
K.SetSize(height, width);
K.SetSize(vdim);
(*Function)(transip, pml, K);
}
};
@@ -116,13 +120,13 @@ void source(const Vector &x, Vector & f);
// Functions for computing the necessary coefficients after PML stretching.
// J is the Jacobian matrix of the stretching function
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector & D);
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector & D);
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector & D);
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M);
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector & D);
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector & D);
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector & D);
Array2D<double> comp_domain_bdr;
Array2D<double> domain_bdr;
@@ -160,6 +164,8 @@ int main(int argc, char *argv[])
double freq = 5.0;
bool herm_conv = true;
bool visualization = 1;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -183,12 +189,21 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (iprob > 4) { iprob = 4; }
prob = (prob_type)iprob;
// 3. Setup the (serial) mesh on all processors.
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 4. Setup the (serial) mesh on all processors.
if (!mesh_file)
{
exact_known = true;
@@ -236,7 +251,7 @@ int main(int argc, char *argv[])
// Setup PML length
Array2D<double> length(dim, 2); length = 0.0;
// 4. Setup the Cartesian PML region.
// 5. Setup the Cartesian PML region.
switch (prob)
{
case disc:
@@ -262,13 +277,13 @@ int main(int argc, char *argv[])
comp_domain_bdr = pml->GetCompDomainBdr();
domain_bdr = pml->GetDomainBdr();
// 5. Refine the serial mesh on all processors to increase the resolution.
// 6. Refine the serial mesh on all processors to increase the resolution.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 6. Define a parallel mesh by a partitioning of the serial mesh.
// 7. Define a parallel mesh by a partitioning of the serial mesh.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
{
@@ -278,13 +293,13 @@ int main(int argc, char *argv[])
}
}
// 6a. Reorient mesh in case of a tet mesh
// 7a. Reorient mesh in case of a tet mesh
pmesh->ReorientTetMesh();
// 7. Set element attributes in order to distinguish elements in the PML
// 8. Set element attributes in order to distinguish elements in the PML
pml->SetAttributes(pmesh);
// 8. Define a parallel finite element space on the parallel mesh. Here we
// 9. Define a parallel finite element space on the parallel mesh. Here we
// use the Nedelec finite elements of the specified order.
FiniteElementCollection *fec = new ND_FECollection(order, dim);
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
@@ -294,9 +309,9 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << size << endl;
}
// 9. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the specific mesh and the problem type.
// 10. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the specific mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (pmesh->bdr_attributes.Size())
@@ -336,11 +351,11 @@ int main(int argc, char *argv[])
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
// 10. Setup Complex Operator convention
// 11. Setup Complex Operator convention
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 11. Set up the parallel linear form b(.) which corresponds to the
// 12. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system.
VectorFunctionCoefficient f(dim, source);
ParComplexLinearForm b(fespace, conv);
@@ -351,7 +366,7 @@ int main(int argc, char *argv[])
b.Vector::operator=(0.0);
b.Assemble();
// 12. Define the solution vector x as a parallel complex finite element grid
// 13. Define the solution vector x as a parallel complex finite element grid
// function corresponding to fespace.
ParComplexGridFunction x(fespace);
x = 0.0;
@@ -359,7 +374,7 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient E_Im(dim, E_bdr_data_Im);
x.ProjectBdrCoefficientTangent(E_Re, E_Im, ess_bdr);
// 13. Set up the parallel sesquilinear form a(.,.)
// 14. Set up the parallel sesquilinear form a(.,.)
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) - omega^2 * epsilon (E,F)
@@ -393,19 +408,19 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_omeg),NULL);
int cdim = (dim == 2) ? 1 : dim;
PMLMatrixCoefficient pml_c1_Re(cdim,detJ_inv_JT_J_Re, pml);
PMLMatrixCoefficient pml_c1_Im(cdim,detJ_inv_JT_J_Im, pml);
ScalarMatrixProductCoefficient c1_Re(muinv,pml_c1_Re);
ScalarMatrixProductCoefficient c1_Im(muinv,pml_c1_Im);
MatrixRestrictedCoefficient restr_c1_Re(c1_Re,attrPML);
MatrixRestrictedCoefficient restr_c1_Im(c1_Im,attrPML);
PMLDiagMatrixCoefficient pml_c1_Re(cdim,detJ_inv_JT_J_Re, pml);
PMLDiagMatrixCoefficient pml_c1_Im(cdim,detJ_inv_JT_J_Im, pml);
ScalarVectorProductCoefficient c1_Re(muinv,pml_c1_Re);
ScalarVectorProductCoefficient c1_Im(muinv,pml_c1_Im);
VectorRestrictedCoefficient restr_c1_Re(c1_Re,attrPML);
VectorRestrictedCoefficient restr_c1_Im(c1_Im,attrPML);
PMLMatrixCoefficient pml_c2_Re(dim, detJ_JT_J_inv_Re,pml);
PMLMatrixCoefficient pml_c2_Im(dim, detJ_JT_J_inv_Im,pml);
ScalarMatrixProductCoefficient c2_Re(omeg,pml_c2_Re);
ScalarMatrixProductCoefficient c2_Im(omeg,pml_c2_Im);
MatrixRestrictedCoefficient restr_c2_Re(c2_Re,attrPML);
MatrixRestrictedCoefficient restr_c2_Im(c2_Im,attrPML);
PMLDiagMatrixCoefficient pml_c2_Re(dim, detJ_JT_J_inv_Re,pml);
PMLDiagMatrixCoefficient pml_c2_Im(dim, detJ_JT_J_inv_Im,pml);
ScalarVectorProductCoefficient c2_Re(omeg,pml_c2_Re);
ScalarVectorProductCoefficient c2_Im(omeg,pml_c2_Im);
VectorRestrictedCoefficient restr_c2_Re(c2_Re,attrPML);
VectorRestrictedCoefficient restr_c2_Im(c2_Im,attrPML);
// Integrators inside the PML region
a.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_Re),
@@ -413,27 +428,25 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_Re),
new VectorFEMassIntegrator(restr_c2_Im));
// 14. Assemble the parallel bilinear form and the corresponding linear
// 15. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, etc.
#ifndef MFEM_USE_SUPERLU
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
#endif
a.Assemble();
OperatorHandle Ah;
OperatorPtr Ah;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 15. Transform to monolithic HypreParMatrix
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
if (myid == 0)
{
cout << "Size of linear system: " << A->GetGlobalNumRows() << endl;
}
// 16. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUPERLU
{
if (pa) { cout << "PA not available with MFEM_USE_SUPERLU" << endl; }
// Transform to monolithic HypreParMatrix
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
SuperLURowLocMatrix SA(*A);
SuperLUSolver superlu(MPI_COMM_WORLD);
superlu.SetPrintStatistics(false);
@@ -441,9 +454,9 @@ int main(int argc, char *argv[])
superlu.SetColumnPermutation(superlu::PARMETIS);
superlu.SetOperator(SA);
superlu.Mult(B, X);
delete A;
}
#else
// 16a. Set up the parallel Bilinear form a(.,.) for the preconditioner
//
// In Comp
@@ -459,22 +472,20 @@ int main(int argc, char *argv[])
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_muinv));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_absomeg));
PMLMatrixCoefficient pml_c1_abs(cdim,detJ_inv_JT_J_abs, pml);
ScalarMatrixProductCoefficient c1_abs(muinv,pml_c1_abs);
MatrixRestrictedCoefficient restr_c1_abs(c1_abs,attrPML);
PMLDiagMatrixCoefficient pml_c1_abs(cdim,detJ_inv_JT_J_abs, pml);
ScalarVectorProductCoefficient c1_abs(muinv,pml_c1_abs);
VectorRestrictedCoefficient restr_c1_abs(c1_abs,attrPML);
PMLMatrixCoefficient pml_c2_abs(dim, detJ_JT_J_inv_abs,pml);
ScalarMatrixProductCoefficient c2_abs(absomeg,pml_c2_abs);
MatrixRestrictedCoefficient restr_c2_abs(c2_abs,attrPML);
PMLDiagMatrixCoefficient pml_c2_abs(dim, detJ_JT_J_inv_abs,pml);
ScalarVectorProductCoefficient c2_abs(absomeg,pml_c2_abs);
VectorRestrictedCoefficient restr_c2_abs(c2_abs,attrPML);
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_abs));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_abs));
if (pa) { prec.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
prec.Assemble();
OperatorHandle PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 16b. Define and apply a parallel GMRES solver for AU=B with a block
// diagonal preconditioner based on hypre's AMS preconditioner.
Array<int> offsets(3);
@@ -483,21 +494,41 @@ int main(int argc, char *argv[])
offsets[2] = fespace->GetTrueVSize();
offsets.PartialSum();
HypreAMS ams00(*PCOpAh.As<HypreParMatrix>(),fespace);
BlockDiagonalPreconditioner BlockAMS(offsets);
ScaledOperator ams11(&ams00,
(conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0);
BlockAMS.SetDiagonalBlock(0,&ams00);
BlockAMS.SetDiagonalBlock(1,&ams11);
Operator *pc_r = nullptr;
Operator *pc_i = nullptr;
int s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
if (pa)
{
// Jacobi Smoother
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
ScaledOperator *d11 = new ScaledOperator(d00, s);
pc_r = d00;
pc_i = d11;
}
else
{
OperatorPtr PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// Hypre AMS
HypreAMS *ams00 = new HypreAMS(*PCOpAh.As<HypreParMatrix>(), fespace);
ScaledOperator *ams11 = new ScaledOperator(ams00, s);
pc_r = ams00;
pc_i = ams11;
}
BlockDiagonalPreconditioner BlockDP(offsets);
BlockDP.SetDiagonalBlock(0, pc_r);
BlockDP.SetDiagonalBlock(1, pc_i);
GMRESSolver gmres(MPI_COMM_WORLD);
gmres.SetPrintLevel(1);
gmres.SetKDim(200);
gmres.SetMaxIter(2000);
gmres.SetMaxIter(pa ? 5000 : 2000);
gmres.SetRelTol(1e-5);
gmres.SetAbsTol(0.0);
gmres.SetOperator(*A);
gmres.SetPreconditioner(BlockAMS);
gmres.SetOperator(*Ah);
gmres.SetPreconditioner(BlockDP);
gmres.Mult(B, X);
}
#endif
@@ -509,10 +540,8 @@ int main(int argc, char *argv[])
// If exact is known compute the error
if (exact_known)
{
ParComplexGridFunction x_gf(fespace);
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
int order_quad = max(2, 2 * order + 1);
const IntegrationRule *irs[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; ++i)
@@ -629,7 +658,6 @@ int main(int argc, char *argv[])
}
// 20. Free the used memory.
delete A;
delete pml;
delete fespace;
delete fec;
@@ -828,7 +856,7 @@ void E_bdr_data_Im(const Vector &x, Vector &E)
}
}
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det(1.0, 0.0);
@@ -839,14 +867,13 @@ void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (det / pow(dxs[i], 2)).real();
D(i) = (det / pow(dxs[i], 2)).real();
}
}
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -857,14 +884,13 @@ void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (det / pow(dxs[i], 2)).imag();
D(i) = (det / pow(dxs[i], 2)).imag();
}
}
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -875,14 +901,13 @@ void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
det *= dxs[i];
}
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = abs(det / pow(dxs[i], 2));
D(i) = abs(det / pow(dxs[i], 2));
}
}
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det(1.0, 0.0);
@@ -896,19 +921,18 @@ void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, DenseMatrix &M)
// in the 2D case the coefficient is scalar 1/det(J)
if (dim == 2)
{
M = (1.0 / det).real();
D = (1.0 / det).real();
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (pow(dxs[i], 2) / det).real();
D(i) = (pow(dxs[i], 2) / det).real();
}
}
}
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -921,19 +945,18 @@ void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, DenseMatrix &M)
if (dim == 2)
{
M = (1.0 / det).imag();
D = (1.0 / det).imag();
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = (pow(dxs[i], 2) / det).imag();
D(i) = (pow(dxs[i], 2) / det).imag();
}
}
}
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector & D)
{
vector<complex<double>> dxs(dim);
complex<double> det = 1.0;
@@ -946,14 +969,13 @@ void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, DenseMatrix &M)
if (dim == 2)
{
M = abs(1.0 / det);
D = abs(1.0 / det);
}
else
{
M = 0.0;
for (int i = 0; i < dim; ++i)
{
M(i, i) = abs(pow(dxs[i], 2) / det);
D(i) = abs(pow(dxs[i], 2) / det);
}
}
}
+36 -17
View File
@@ -11,6 +11,12 @@
// ex5 -m ../data/escher.mesh
// ex5 -m ../data/fichera.mesh
//
// Device sample runs:
// ex5 -m ../data/star.mesh -pa -d cuda
// ex5 -m ../data/star.mesh -pa -d raja-cuda
// ex5 -m ../data/star.mesh -pa -d raja-omp
// ex5 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple 2D/3D mixed Darcy problem
// corresponding to the saddle point system
// k*u + grad p = f
@@ -50,6 +56,7 @@ int main(int argc, char *argv[])
const char *mesh_file = "../data/star.mesh";
int order = 1;
bool pa = false;
const char *device_config = "cpu";
bool visualization = 1;
OptionsParser args(argc, argv);
@@ -59,6 +66,8 @@ int main(int argc, char *argv[])
"Finite element order (polynomial degree).");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -70,13 +79,18 @@ int main(int argc, char *argv[])
}
args.PrintOptions(cout);
// 2. Read the mesh from the given mesh file. We can handle triangular,
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Refine the mesh to increase the resolution. In this example we do
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
// largest number that gives a final mesh with no more than 10,000
// elements.
@@ -89,7 +103,7 @@ int main(int argc, char *argv[])
}
}
// 4. Define a finite element space on the mesh. Here we use the
// 5. Define a finite element space on the mesh. Here we use the
// Raviart-Thomas finite elements of the specified order.
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
@@ -97,7 +111,7 @@ int main(int argc, char *argv[])
FiniteElementSpace *R_space = new FiniteElementSpace(mesh, hdiv_coll);
FiniteElementSpace *W_space = new FiniteElementSpace(mesh, l2_coll);
// 5. Define the BlockStructure of the problem, i.e. define the array of
// 6. Define the BlockStructure of the problem, i.e. define the array of
// offsets for each variable. The last component of the Array is the sum
// of the dimensions of each block.
Array<int> block_offsets(3); // number of variables + 1
@@ -112,7 +126,7 @@ int main(int argc, char *argv[])
std::cout << "dim(R+W) = " << block_offsets.Last() << "\n";
std::cout << "***********************************************************\n";
// 6. Define the coefficients, analytical solution, and rhs of the PDE.
// 7. Define the coefficients, analytical solution, and rhs of the PDE.
ConstantCoefficient k(1.0);
VectorFunctionCoefficient fcoeff(dim, fFun);
@@ -122,25 +136,28 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
FunctionCoefficient pcoeff(pFun_ex);
// 7. Allocate memory (x, rhs) for the analytical solution and the right hand
// 8. Allocate memory (x, rhs) for the analytical solution and the right hand
// side. Define the GridFunction u,p for the finite element solution and
// linear forms fform and gform for the right hand side. The data
// allocated by x and rhs are passed as a reference to the grid functions
// (u,p) and the linear forms (fform, gform).
BlockVector x(block_offsets), rhs(block_offsets);
MemoryType mt = device.GetMemoryType();
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
LinearForm *fform(new LinearForm);
fform->Update(R_space, rhs.GetBlock(0), 0);
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
fform->Assemble();
fform->SyncAliasMemory(rhs);
LinearForm *gform(new LinearForm);
gform->Update(W_space, rhs.GetBlock(1), 0);
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
gform->Assemble();
gform->SyncAliasMemory(rhs);
// 8. Assemble the finite element matrices for the Darcy operator
// 9. Assemble the finite element matrices for the Darcy operator
//
// D = [ M B^T ]
// [ B 0 ]
@@ -185,7 +202,7 @@ int main(int argc, char *argv[])
darcyOp.SetBlock(1,0, &B);
}
// 9. Construct the operators for preconditioner
// 10. Construct the operators for preconditioner
//
// P = [ diag(M) 0 ]
// [ 0 B diag(M)^-1 B^T ]
@@ -202,10 +219,11 @@ int main(int argc, char *argv[])
if (pa)
{
mVarf->AssembleDiagonal(Md);
auto Md_host = Md.HostRead();
Vector invMd(mVarf->Height());
for (int i=0; i<mVarf->Height(); ++i)
{
invMd(i) = 1.0 / Md(i);
invMd(i) = 1.0 / Md_host[i];
}
Vector BMBt_diag(bVarf->Height());
@@ -246,7 +264,7 @@ int main(int argc, char *argv[])
darcyPrec.SetDiagonalBlock(0, invM);
darcyPrec.SetDiagonalBlock(1, invS);
// 10. Solve the linear system with MINRES.
// 11. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(1000);
double rtol(1.e-6);
@@ -263,6 +281,7 @@ int main(int argc, char *argv[])
solver.SetPrintLevel(1);
x = 0.0;
solver.Mult(rhs, x);
if (device.IsEnabled()) { x.HostRead(); }
chrono.Stop();
if (solver.GetConverged())
@@ -273,7 +292,7 @@ int main(int argc, char *argv[])
<< " iterations. Residual norm is " << solver.GetFinalNorm() << ".\n";
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
// 11. Create the grid functions u and p. Compute the L2 error norms.
// 12. Create the grid functions u and p. Compute the L2 error norms.
GridFunction u, p;
u.MakeRef(R_space, x.GetBlock(0), 0);
p.MakeRef(W_space, x.GetBlock(1), 0);
@@ -293,7 +312,7 @@ int main(int argc, char *argv[])
std::cout << "|| u_h - u_ex || / || u_ex || = " << err_u / norm_u << "\n";
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
// 12. Save the mesh and the solution. This output can be viewed later using
// 13. Save the mesh and the solution. This output can be viewed later using
// GLVis: "glvis -m ex5.mesh -g sol_u.gf" or "glvis -m ex5.mesh -g
// sol_p.gf".
{
@@ -310,13 +329,13 @@ int main(int argc, char *argv[])
p.Save(p_ofs);
}
// 13. Save data in the VisIt format
// 14. Save data in the VisIt format
VisItDataCollection visit_dc("Example5", mesh);
visit_dc.RegisterField("velocity", &u);
visit_dc.RegisterField("pressure", &p);
visit_dc.Save();
// 14. Save data in the ParaView format
// 15. Save data in the ParaView format
ParaViewDataCollection paraview_dc("Example5", mesh);
paraview_dc.SetPrefixPath("ParaView");
paraview_dc.SetLevelsOfDetail(order);
@@ -328,7 +347,7 @@ int main(int argc, char *argv[])
paraview_dc.RegisterField("pressure",&p);
paraview_dc.Save();
// 15. Send the solution by socket to a GLVis server.
// 16. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -341,7 +360,7 @@ int main(int argc, char *argv[])
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
}
// 16. Free the used memory.
// 17. Free the used memory.
delete fform;
delete gform;
delete invM;
+42 -21
View File
@@ -11,6 +11,12 @@
// mpirun -np 4 ex5p -m ../data/escher.mesh
// mpirun -np 4 ex5p -m ../data/fichera.mesh
//
// Device sample runs:
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d cuda
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-cuda
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-omp
// mpirun -np 4 ex5p -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple 2D/3D mixed Darcy problem
// corresponding to the saddle point system
// k*u + grad p = f
@@ -60,6 +66,7 @@ int main(int argc, char *argv[])
int order = 1;
bool par_format = false;
bool pa = false;
const char *device_config = "cpu";
bool visualization = 1;
bool adios2 = false;
@@ -75,6 +82,8 @@ int main(int argc, char *argv[])
"Format to use when saving the results for VisIt.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -96,13 +105,18 @@ int main(int argc, char *argv[])
args.PrintOptions(cout);
}
// 3. Read the (serial) mesh from the given mesh file on all processors. We
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
if (myid == 0) { device.Print(); }
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the serial mesh on all processors to increase the resolution. In
// 5. Refine the serial mesh on all processors to increase the resolution. In
// this example we do 'ref_levels' of uniform refinement. We choose
// 'ref_levels' to be the largest number that gives a final mesh with no
// more than 10,000 elements, unless the user specifies it as input.
@@ -118,7 +132,7 @@ int main(int argc, char *argv[])
}
}
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
@@ -131,7 +145,7 @@ int main(int argc, char *argv[])
}
}
// 6. Define a parallel finite element space on the parallel mesh. Here we
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use the Raviart-Thomas finite elements of the specified order.
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
@@ -151,7 +165,7 @@ int main(int argc, char *argv[])
std::cout << "***********************************************************\n";
}
// 7. Define the two BlockStructure of the problem. block_offsets is used
// 8. Define the two BlockStructure of the problem. block_offsets is used
// for Vector based on dof (like ParGridFunction or ParLinearForm),
// block_trueOffstes is used for Vector based on trueDof (HypreParVector
// for the rhs and solution of the linear system). The offsets computed
@@ -168,7 +182,7 @@ int main(int argc, char *argv[])
block_trueOffsets[2] = W_space->TrueVSize();
block_trueOffsets.PartialSum();
// 8. Define the coefficients, analytical solution, and rhs of the PDE.
// 9. Define the coefficients, analytical solution, and rhs of the PDE.
ConstantCoefficient k(1.0);
VectorFunctionCoefficient fcoeff(dim, fFun);
@@ -178,25 +192,30 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
FunctionCoefficient pcoeff(pFun_ex);
// 9. Define the parallel grid function and parallel linear forms, solution
// vector and rhs.
BlockVector x(block_offsets), rhs(block_offsets);
BlockVector trueX(block_trueOffsets), trueRhs(block_trueOffsets);
// 10. Define the parallel grid function and parallel linear forms, solution
// vector and rhs.
MemoryType mt = device.GetMemoryType();
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
BlockVector trueX(block_trueOffsets, mt), trueRhs(block_trueOffsets, mt);
ParLinearForm *fform(new ParLinearForm);
fform->Update(R_space, rhs.GetBlock(0), 0);
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
fform->Assemble();
fform->SyncAliasMemory(rhs);
fform->ParallelAssemble(trueRhs.GetBlock(0));
trueRhs.GetBlock(0).SyncAliasMemory(trueRhs);
ParLinearForm *gform(new ParLinearForm);
gform->Update(W_space, rhs.GetBlock(1), 0);
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
gform->Assemble();
gform->SyncAliasMemory(rhs);
gform->ParallelAssemble(trueRhs.GetBlock(1));
trueRhs.GetBlock(1).SyncAliasMemory(trueRhs);
// 10. Assemble the finite element matrices for the Darcy operator
// 11. Assemble the finite element matrices for the Darcy operator
//
// D = [ M B^T ]
// [ B 0 ]
@@ -249,7 +268,7 @@ int main(int argc, char *argv[])
darcyOp->SetBlock(1,0, B);
}
// 11. Construct the operators for preconditioner
// 12. Construct the operators for preconditioner
//
// P = [ diag(M) 0 ]
// [ 0 B diag(M)^-1 B^T ]
@@ -266,10 +285,11 @@ int main(int argc, char *argv[])
{
Md_PA.SetSize(R_space->GetTrueVSize());
mVarf->AssembleDiagonal(Md_PA);
auto Md_host = Md_PA.HostRead();
Vector invMd(Md_PA.Size());
for (int i=0; i<Md_PA.Size(); ++i)
{
invMd(i) = 1.0 / Md_PA(i);
invMd(i) = 1.0 / Md_host[i];
}
Vector BMBt_diag(W_space->GetTrueVSize());
@@ -302,7 +322,7 @@ int main(int argc, char *argv[])
darcyPr->SetDiagonalBlock(0, invM);
darcyPr->SetDiagonalBlock(1, invS);
// 12. Solve the linear system with MINRES.
// 13. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(pa ? 1000 : 500);
double rtol(1.e-6);
@@ -319,6 +339,7 @@ int main(int argc, char *argv[])
solver.SetPrintLevel(verbose);
trueX = 0.0;
solver.Mult(trueRhs, trueX);
if (device.IsEnabled()) { trueX.HostRead(); }
chrono.Stop();
if (verbose)
@@ -332,7 +353,7 @@ int main(int argc, char *argv[])
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
}
// 13. Extract the parallel grid function corresponding to the finite element
// 14. Extract the parallel grid function corresponding to the finite element
// approximation X. This is the local solution on each processor. Compute
// L2 error norms.
ParGridFunction *u(new ParGridFunction);
@@ -360,7 +381,7 @@ int main(int argc, char *argv[])
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
}
// 14. Save the refined mesh and the solution in parallel. This output can be
// 15. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol_*".
{
ostringstream mesh_name, u_name, p_name;
@@ -381,7 +402,7 @@ int main(int argc, char *argv[])
p->Save(p_ofs);
}
// 15. Save data in the VisIt format
// 16. Save data in the VisIt format
VisItDataCollection visit_dc("Example5-Parallel", pmesh);
visit_dc.RegisterField("velocity", u);
visit_dc.RegisterField("pressure", p);
@@ -390,7 +411,7 @@ int main(int argc, char *argv[])
DataCollection::PARALLEL_FORMAT);
visit_dc.Save();
// 16. Save data in the ParaView format
// 17. Save data in the ParaView format
ParaViewDataCollection paraview_dc("Example5P", pmesh);
paraview_dc.SetPrefixPath("ParaView");
paraview_dc.SetLevelsOfDetail(order);
@@ -402,7 +423,7 @@ int main(int argc, char *argv[])
paraview_dc.RegisterField("pressure",p);
paraview_dc.Save();
// 17. Optionally output a BP (binary pack) file using ADIOS2. This can be
// 18. Optionally output a BP (binary pack) file using ADIOS2. This can be
// visualized with the ParaView VTX reader.
#ifdef MFEM_USE_ADIOS2
if (adios2)
@@ -422,7 +443,7 @@ int main(int argc, char *argv[])
}
#endif
// 18. Send the solution by socket to a GLVis server.
// 19. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -442,7 +463,7 @@ int main(int argc, char *argv[])
<< endl;
}
// 19. Free the used memory.
// 20. Free the used memory.
delete fform;
delete gform;
delete u;
+19 -17
View File
@@ -20,8 +20,11 @@
// Device sample runs:
// ex9 -pa
// ex9 -ea
// ex9 -fa
// ex9 -pa -m ../data/periodic-cube.mesh
// ex9 -pa -m ../data/periodic-cube.mesh -d cuda
// ex9 -ea -m ../data/periodic-cube.mesh -d cuda
// ex9 -fa -m ../data/periodic-cube.mesh -d cuda
//
// Description: This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
@@ -144,6 +147,7 @@ int main(int argc, char *argv[])
int order = 3;
bool pa = false;
bool ea = false;
bool fa = false;
const char *device_config = "cpu";
int ode_solver_type = 4;
double t_final = 10.0;
@@ -170,6 +174,8 @@ int main(int argc, char *argv[])
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
"--no-element-assembly", "Enable Element Assembly.");
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
"--no-full-assembly", "Enable Full Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
@@ -254,7 +260,7 @@ int main(int argc, char *argv[])
// 5. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
DG_FECollection fec(order, dim, BasisType::Positive);
DG_FECollection fec(order, dim, BasisType::GaussLobatto);
FiniteElementSpace fes(&mesh, &fec);
cout << "Number of unknowns: " << fes.GetVSize() << endl;
@@ -278,6 +284,11 @@ int main(int argc, char *argv[])
m.SetAssemblyLevel(AssemblyLevel::ELEMENT);
k.SetAssemblyLevel(AssemblyLevel::ELEMENT);
}
else if (fa)
{
m.SetAssemblyLevel(AssemblyLevel::FULL);
k.SetAssemblyLevel(AssemblyLevel::FULL);
}
m.AddDomainIntegrator(new MassIntegrator);
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k.AddInteriorFaceIntegrator(
@@ -378,10 +389,6 @@ int main(int argc, char *argv[])
// iterations, ti, with a time-step dt).
FE_Evolution adv(m, k, b);
Vector masses(u.Size());
m.SpMat().Mult(u, masses);
double mass = masses.Sum();
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
@@ -428,9 +435,6 @@ int main(int argc, char *argv[])
u.Save(osol);
}
m.SpMat().Mult(u, masses);
cout << "Mass difference:" << abs(mass - masses.Sum()) << endl;
// 10. Free the used memory.
delete ode_solver;
delete pd;
@@ -444,21 +448,19 @@ int main(int argc, char *argv[])
FE_Evolution::FE_Evolution(BilinearForm &_M, BilinearForm &_K, const Vector &_b)
: TimeDependentOperator(_M.Height()), M(_M), K(_K), b(_b), z(_M.Height())
{
bool pa = M.GetAssemblyLevel() == AssemblyLevel::PARTIAL;
bool ea = M.GetAssemblyLevel() == AssemblyLevel::ELEMENT;
Array<int> ess_tdof_list;
if (pa || ea)
if (M.GetAssemblyLevel() == AssemblyLevel::LEGACYFULL)
{
M_prec = new DSmoother(M.SpMat());
M_solver.SetOperator(M.SpMat());
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
}
else
{
M_prec = new OperatorJacobiSmoother(M, ess_tdof_list);
M_solver.SetOperator(M);
dg_solver = NULL;
}
else
{
M_prec = new DSmoother(M.SpMat());
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
M_solver.SetOperator(M.SpMat());
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
+24 -15
View File
@@ -21,8 +21,11 @@
// Device sample runs:
// mpirun -np 4 ex9p -pa
// mpirun -np 4 ex9p -ea
// mpirun -np 4 ex9p -fa
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh -d cuda
// mpirun -np 4 ex9p -ea -m ../data/periodic-cube.mesh -d cuda
// mpirun -np 4 ex9p -fa -m ../data/periodic-cube.mesh -d cuda
//
// Description: This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
@@ -164,6 +167,7 @@ int main(int argc, char *argv[])
int order = 3;
bool pa = false;
bool ea = false;
bool fa = false;
const char *device_config = "cpu";
int ode_solver_type = 4;
double t_final = 10.0;
@@ -193,6 +197,8 @@ int main(int argc, char *argv[])
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
"--no-element-assembly", "Enable Element Assembly.");
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
"--no-full-assembly", "Enable Full Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
@@ -329,6 +335,12 @@ int main(int argc, char *argv[])
m->SetAssemblyLevel(AssemblyLevel::ELEMENT);
k->SetAssemblyLevel(AssemblyLevel::ELEMENT);
}
else if (fa)
{
m->SetAssemblyLevel(AssemblyLevel::FULL);
k->SetAssemblyLevel(AssemblyLevel::FULL);
}
m->AddDomainIntegrator(new MassIntegrator);
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k->AddInteriorFaceIntegrator(
@@ -565,29 +577,21 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
M_solver(_M.ParFESpace()->GetComm()),
z(_M.Height())
{
bool pa = _M.GetAssemblyLevel()==AssemblyLevel::PARTIAL;
bool ea = _M.GetAssemblyLevel()==AssemblyLevel::ELEMENT;
if (pa || ea)
{
M.Reset(&_M, false);
K.Reset(&_K, false);
}
else
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
{
M.Reset(_M.ParallelAssemble(), true);
K.Reset(_K.ParallelAssemble(), true);
}
else
{
M.Reset(&_M, false);
K.Reset(&_K, false);
}
M_solver.SetOperator(*M);
Array<int> ess_tdof_list;
if (pa || ea)
{
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
dg_solver = NULL;
}
else
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
{
HypreParMatrix &M_mat = *M.As<HypreParMatrix>();
HypreParMatrix &K_mat = *K.As<HypreParMatrix>();
@@ -596,6 +600,11 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
dg_solver = new DG_Solver(M_mat, K_mat, *_M.FESpace());
}
else
{
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
dg_solver = NULL;
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
+16 -14
View File
@@ -76,7 +76,7 @@ BilinearForm::BilinearForm(FiniteElementSpace * f)
precompute_sparsity = 0;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::FULL;
assembly = AssemblyLevel::LEGACYFULL;
batch = 1;
ext = NULL;
}
@@ -94,7 +94,7 @@ BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
precompute_sparsity = ps;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::FULL;
assembly = AssemblyLevel::LEGACYFULL;
batch = 1;
ext = NULL;
@@ -121,9 +121,10 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::LEGACYFULL:
break;
case AssemblyLevel::FULL:
// ext = new FABilinearFormExtension(this);
// Use the original BilinearForm implementation for now
ext = new FABilinearFormExtension(this);
break;
case AssemblyLevel::ELEMENT:
ext = new EABilinearFormExtension(this);
@@ -143,7 +144,7 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
void BilinearForm::EnableStaticCondensation()
{
delete static_cond;
if (assembly != AssemblyLevel::FULL)
if (assembly != AssemblyLevel::LEGACYFULL)
{
static_cond = NULL;
MFEM_WARNING("Static condensation not supported for this assembly level");
@@ -168,7 +169,7 @@ void BilinearForm::EnableHybridization(FiniteElementSpace *constr_space,
const Array<int> &ess_tdof_list)
{
delete hybridization;
if (assembly != AssemblyLevel::FULL)
if (assembly != AssemblyLevel::LEGACYFULL)
{
delete constr_integ;
hybridization = NULL;
@@ -223,7 +224,7 @@ MatrixInverse * BilinearForm::Inverse() const
void BilinearForm::Finalize (int skip_zeros)
{
if (assembly == AssemblyLevel::FULL)
if (assembly == AssemblyLevel::LEGACYFULL)
{
if (!static_cond) { mat->Finalize(skip_zeros); }
if (mat_e) { mat_e->Finalize(skip_zeros); }
@@ -639,8 +640,7 @@ void BilinearForm::AssembleDiagonal(Vector &diag) const
}
else
{
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
"matrix and use SparseMatrix::GetDiag?");
mat->GetDiag(diag);
}
}
@@ -1083,7 +1083,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
mat = NULL;
mat_e = NULL;
extern_bfs = 0;
assembly = AssemblyLevel::FULL;
assembly = AssemblyLevel::LEGACYFULL;
ext = NULL;
}
@@ -1108,7 +1108,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
bbfi_marker = mbf->bbfi_marker;
btfbfi_marker = mbf->btfbfi_marker;
assembly = AssemblyLevel::FULL;
assembly = AssemblyLevel::LEGACYFULL;
ext = NULL;
}
@@ -1121,6 +1121,8 @@ void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::LEGACYFULL:
break;
case AssemblyLevel::FULL:
// ext = new FAMixedBilinearFormExtension(this);
// Use the original BilinearForm implementation for now
@@ -1191,7 +1193,7 @@ void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
MatrixInverse * MixedBilinearForm::Inverse() const
{
if (assembly != AssemblyLevel::FULL)
if (assembly != AssemblyLevel::LEGACYFULL)
{
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
return NULL;
@@ -1204,7 +1206,7 @@ MatrixInverse * MixedBilinearForm::Inverse() const
void MixedBilinearForm::Finalize (int skip_zeros)
{
if (assembly == AssemblyLevel::FULL)
if (assembly == AssemblyLevel::LEGACYFULL)
{
mat -> Finalize (skip_zeros);
}
@@ -1481,7 +1483,7 @@ void MixedBilinearForm::AssembleDiagonal_ADAt(const Vector &D,
void MixedBilinearForm::ConformingAssemble()
{
if (assembly != AssemblyLevel::FULL)
if (assembly != AssemblyLevel::LEGACYFULL)
{
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
return;
+6 -3
View File
@@ -29,8 +29,11 @@ namespace mfem
form classes derived from Operator. */
enum class AssemblyLevel
{
/// Fully assembled form, i.e. a global sparse matrix in MFEM, Hypre or PETSC
/// format.
/// Legacy fully assembled form, i.e. a global sparse matrix in MFEM, Hypre
/// or PETSC format. This assembly is ALWAYS performed on the host.
LEGACYFULL = 0,
/// Fully assembled form, i.e. a global sparse matrix in MFEM format. This
/// assembly is compatible with device execution.
FULL,
/// Form assembled at element level, which computes and stores dense element
/// matrices.
@@ -119,7 +122,7 @@ protected:
static_cond = NULL; hybridization = NULL;
precompute_sparsity = 0;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::FULL;
assembly = AssemblyLevel::LEGACYFULL;
batch = 1;
ext = NULL;
}
+187 -35
View File
@@ -15,6 +15,7 @@
#include "../general/forall.hpp"
#include "bilinearform.hpp"
#include "libceed/ceed.hpp"
#include "pgridfunc.hpp"
namespace mfem
{
@@ -292,7 +293,8 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
// Data and methods for element-assembled bilinear forms
EABilinearFormExtension::EABilinearFormExtension(BilinearForm *form)
: PABilinearFormExtension(form)
: PABilinearFormExtension(form),
factorize_face_terms(form->FESpace()->IsDGSpace())
{
}
@@ -347,6 +349,17 @@ void EABilinearFormExtension::Assemble()
{
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr);
}
if (factorize_face_terms && int_face_restrict_lex)
{
auto restFint = dynamic_cast<const L2FaceRestriction&>(*int_face_restrict_lex);
restFint.AddFaceMatricesToElementMatrices(ea_data_int, ea_data);
}
if (factorize_face_terms && bdr_face_restrict_lex)
{
auto restFbdr = dynamic_cast<const L2FaceRestriction&>(*bdr_face_restrict_lex);
restFbdr.AddFaceMatricesToElementMatrices(ea_data_bdr, ea_data);
}
}
void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
@@ -399,24 +412,27 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
const int NDOFS = faceDofs;
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
if (!factorize_face_terms)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
res += A_int(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
@@ -443,7 +459,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
// Treatment of boundary faces
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
const int bFISz = bdrFaceIntegrators.Size();
if (bdr_face_restrict_lex && bFISz>0)
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
{
// Apply the Boundary Face Restriction
bdr_face_restrict_lex->Mult(x, faceBdrX);
@@ -522,24 +538,27 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
const int NDOFS = faceDofs;
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
if (!factorize_face_terms)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
@@ -566,7 +585,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
// Treatment of boundary faces
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
const int bFISz = bdrFaceIntegrators.Size();
if (bdr_face_restrict_lex && bFISz>0)
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
{
// Apply the Boundary Face Restriction
bdr_face_restrict_lex->Mult(x, faceBdrX);
@@ -595,6 +614,139 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
}
}
// Data and methods for fully-assembled bilinear forms
FABilinearFormExtension::FABilinearFormExtension(BilinearForm *form)
: EABilinearFormExtension(form),
mat(form->FESpace()->GetVSize(),form->FESpace()->GetVSize(),0),
face_mat(form->FESpace()->GetVSize(),0,0),
use_face_mat(false)
{
#ifdef MFEM_USE_MPI
if ( ParFiniteElementSpace* pfes =
dynamic_cast<ParFiniteElementSpace*>(form->FESpace()) )
{
if (pfes->IsDGSpace())
{
use_face_mat = true;
pfes->ExchangeFaceNbrData();
face_mat.SetWidth(pfes->GetFaceNbrVSize());
}
}
#endif
}
void FABilinearFormExtension::Assemble()
{
EABilinearFormExtension::Assemble();
FiniteElementSpace &fes = *a->FESpace();
if (fes.IsDGSpace())
{
const L2ElementRestriction *restE =
static_cast<const L2ElementRestriction*>(elem_restrict);
const L2FaceRestriction *restF =
static_cast<const L2FaceRestriction*>(int_face_restrict_lex);
// 1. Fill I
// 1.1 Increment with restE
restE->FillI(mat);
// 1.2 Increment with restF
if (restF) { restF->FillI(mat, face_mat); }
// 1.3 Sum the non-zeros in I
auto h_I = mat.HostReadWriteI();
int cpt = 0;
const int vd = fes.GetVDim();
const int ndofs = ne*elemDofs*vd;
for (int i = 0; i < ndofs; i++)
{
const int nnz = h_I[i];
h_I[i] = cpt;
cpt += nnz;
}
const int nnz = cpt;
h_I[ndofs] = nnz;
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
if (use_face_mat && restF)
{
auto h_I_face = face_mat.HostReadWriteI();
int cpt = 0;
for (int i = 0; i < ndofs; i++)
{
const int nnz = h_I_face[i];
h_I_face[i] = cpt;
cpt += nnz;
}
const int nnz_face = cpt;
h_I_face[ndofs] = nnz_face;
face_mat.GetMemoryJ().New(nnz_face,
face_mat.GetMemoryJ().GetMemoryType());
face_mat.GetMemoryData().New(nnz_face,
face_mat.GetMemoryData().GetMemoryType());
}
// 2. Fill J and Data
// 2.1 Fill J and Data with Elem ea_data
restE->FillJAndData(ea_data, mat);
// 2.2 Fill J and Data with Face ea_data_ext
if (restF) { restF->FillJAndData(ea_data_ext, mat, face_mat); }
// 2.3 Shift indirections in I back to original
auto I = mat.HostReadWriteI();
for (int i = ndofs; i > 0; i--)
{
I[i] = I[i-1];
}
I[0] = 0;
if (use_face_mat && restF)
{
auto I_face = face_mat.HostReadWriteI();
for (int i = ndofs; i > 0; i--)
{
I_face[i] = I_face[i-1];
}
I_face[0] = 0;
}
}
else // continuous Galerkin case
{
const ElementRestriction &rest =
static_cast<const ElementRestriction&>(*elem_restrict);
rest.FillSparseMatrix(ea_data, mat);
}
}
void FABilinearFormExtension::Mult(const Vector &x, Vector &y) const
{
mat.Mult(x, y);
#ifdef MFEM_USE_MPI
if (const ParFiniteElementSpace *pfes =
dynamic_cast<const ParFiniteElementSpace*>(testFes))
{
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
const_cast<Vector&>(x),0);
x_gf.ExchangeFaceNbrData();
Vector &shared_x = x_gf.FaceNbrData();
if (shared_x.Size()) { face_mat.AddMult(shared_x, y); }
}
#endif
}
void FABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
{
mat.MultTranspose(x, y);
#ifdef MFEM_USE_MPI
if (const ParFiniteElementSpace *pfes =
dynamic_cast<const ParFiniteElementSpace*>(testFes))
{
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
const_cast<Vector&>(x),0);
x_gf.ExchangeFaceNbrData();
Vector &shared_x = x_gf.FaceNbrData();
if (shared_x.Size()) { face_mat.AddMultTranspose(shared_x, y); }
}
#endif
}
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
: Operator(form->Height(), form->Width()), a(form)
{
+19 -21
View File
@@ -62,27 +62,6 @@ public:
virtual void Update() = 0;
};
/** @brief Data and methods for fully-assembled bilinear forms.
Not yet implemented! Use the BilinearForm Class instead. */
class FABilinearFormExtension : public BilinearFormExtension
{
public:
FABilinearFormExtension(BilinearForm *form)
: BilinearFormExtension(form) { }
/// TODO
void Assemble() {}
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A) {}
void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0) {}
void Mult(const Vector &x, Vector &y) const {}
void MultTranspose(const Vector &x, Vector &y) const {}
void Update() {}
~FABilinearFormExtension() {}
};
/// Data and methods for partially-assembled bilinear forms
class PABilinearFormExtension : public BilinearFormExtension
{
@@ -119,10 +98,12 @@ class EABilinearFormExtension : public PABilinearFormExtension
protected:
int ne;
int elemDofs;
// The element matrices are stored row major
Vector ea_data;
int nf_int, nf_bdr;
int faceDofs;
Vector ea_data_int, ea_data_ext, ea_data_bdr;
bool factorize_face_terms;
public:
EABilinearFormExtension(BilinearForm *form);
@@ -132,6 +113,23 @@ public:
void MultTranspose(const Vector &x, Vector &y) const;
};
/// Data and methods for fully-assembled bilinear forms
class FABilinearFormExtension : public EABilinearFormExtension
{
private:
SparseMatrix mat;
/// face_mat handles parallelism for DG face terms.
SparseMatrix face_mat;
bool use_face_mat;
public:
FABilinearFormExtension(BilinearForm *form);
void Assemble();
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
};
/// Data and methods for matrix-free bilinear forms NOT YET IMPLEMENTED.
class MFBilinearFormExtension : public BilinearFormExtension
{
+8
View File
@@ -1522,6 +1522,7 @@ void CurlCurlIntegrator::AssembleElementMatrix
double w;
#ifdef MFEM_THREAD_SAFE
Vector D;
DenseMatrix curlshape(nd,dimc), curlshape_dFt(nd,dimc), M;
#else
curlshape.SetSize(nd,dimc);
@@ -1529,6 +1530,7 @@ void CurlCurlIntegrator::AssembleElementMatrix
#endif
elmat.SetSize(nd);
if (MQ) { M.SetSize(dimc); }
if (DQ) { D.SetSize(dimc); }
const IntegrationRule *ir = IntRule;
if (ir == NULL)
@@ -1572,6 +1574,12 @@ void CurlCurlIntegrator::AssembleElementMatrix
Mult(curlshape_dFt, M, curlshape);
AddMultABt(curlshape, curlshape_dFt, elmat);
}
else if (DQ)
{
DQ->Eval(D, Trans, ip);
D *= w;
AddMultADAt(curlshape_dFt, D, elmat);
}
else if (Q)
{
w *= Q->Eval(Trans, ip);
+53 -4
View File
@@ -20,6 +20,13 @@
namespace mfem
{
// Local maximum size of dofs and quads in 1D
constexpr int HCURL_MAX_D1D = 5;
constexpr int HCURL_MAX_Q1D = 6;
constexpr int HDIV_MAX_D1D = 5;
constexpr int HDIV_MAX_Q1D = 6;
/// Abstract base class BilinearFormIntegrator
class BilinearFormIntegrator : public NonlinearFormIntegrator
{
@@ -1685,6 +1692,22 @@ protected:
{
trial_fe.CalcPhysCurlShape(Trans, shape);
}
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector&, Vector&) const;
private:
// PA extension
Vector pa_data;
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const DofToQuad *mapsOtest; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsCtest; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, dofs1Dtest,quad1D, testType, trialType, coeffDim;
};
/** Class for integrating the bilinear form a(u,v) := (Q u, curl v) in 3D and
@@ -1724,6 +1747,20 @@ protected:
{
test_fe.CalcPhysCurlShape(Trans, shape);
}
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector&, Vector&) const;
private:
// PA extension
Vector pa_data;
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D, testType, trialType, coeffDim;
};
/** Class for integrating the bilinear form a(u,v) := - (Q u, grad v) in either
@@ -2263,12 +2300,14 @@ class CurlCurlIntegrator: public BilinearFormIntegrator
private:
Vector vec, pointflux;
#ifndef MFEM_THREAD_SAFE
Vector D;
DenseMatrix curlshape, curlshape_dFt, M;
DenseMatrix vshape, projcurl;
#endif
protected:
Coefficient *Q;
VectorCoefficient *DQ;
MatrixCoefficient *MQ;
// PA extension
@@ -2277,12 +2316,17 @@ protected:
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
public:
CurlCurlIntegrator() { Q = NULL; MQ = NULL; }
CurlCurlIntegrator() { Q = NULL; DQ = NULL; MQ = NULL; }
/// Construct a bilinear form integrator for Nedelec elements
CurlCurlIntegrator(Coefficient &q) : Q(&q) { MQ = NULL; }
CurlCurlIntegrator(MatrixCoefficient &m) : MQ(&m) { Q = NULL; }
CurlCurlIntegrator(Coefficient &q, const IntegrationRule *ir = NULL) :
BilinearFormIntegrator(ir), Q(&q) { DQ = NULL; MQ = NULL; }
CurlCurlIntegrator(VectorCoefficient &dq, const IntegrationRule *ir = NULL) :
BilinearFormIntegrator(ir), DQ(&dq) { Q = NULL; MQ = NULL; }
CurlCurlIntegrator(MatrixCoefficient &mq, const IntegrationRule *ir = NULL) :
BilinearFormIntegrator(ir), MQ(&mq) { Q = NULL; DQ = NULL; }
/* Given a particular Finite Element, compute the
element curl-curl matrix elmat */
@@ -2360,8 +2404,11 @@ protected:
Vector pa_data;
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const DofToQuad *mapsOtest; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsCtest; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D, fetype;
int dim, ne, nq, dofs1D, dofs1Dtest, quad1D, trial_fetype, test_fetype;
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
public:
VectorFEMassIntegrator() { Init(NULL, NULL, NULL); }
@@ -2382,6 +2429,8 @@ public:
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual void AssembleDiagonalPA(Vector& diag);
};
+6 -6
View File
@@ -32,7 +32,7 @@ static void EAConvectionAssemble1D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -54,7 +54,7 @@ static void EAConvectionAssemble1D(const int NE,
{
val += r_Bj[k1] * D(k1, e) * r_Gi[k1];
}
A(i1, j1, e) = val;
A(i1, j1, e) += val;
}
}
});
@@ -76,7 +76,7 @@ static void EAConvectionAssemble2D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -121,7 +121,7 @@ static void EAConvectionAssemble2D(const int NE,
* r_B[k1][j1]* r_B[k2][j2];
}
}
A(i1, i2, j1, j2, e) = val;
A(i1, i2, j1, j2, e) += val;
}
}
}
@@ -145,7 +145,7 @@ static void EAConvectionAssemble3D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 3, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -191,7 +191,7 @@ static void EAConvectionAssemble3D(const int NE,
}
}
}
A(i1, i2, i3, j1, j2, j3, e) = val;
A(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
+14
View File
@@ -788,6 +788,20 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
vel = cQ->GetVec();
}
else if (VectorQuadratureFunctionCoefficient* cQ =
dynamic_cast<VectorQuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == dim * nq * ne,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
vel.SetSize(dim * nq * ne);
+27
View File
@@ -167,6 +167,19 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
r.SetSize(1);
r(0) = c_rho->constant;
}
else if (QuadratureFunctionCoefficient* c_rho =
dynamic_cast<QuadratureFunctionCoefficient*>(rho))
{
const QuadratureFunction &qFun = c_rho->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == nq * nf,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
r.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
r.SetSize(nq * nf);
@@ -200,6 +213,20 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
{
vel = c_u->GetVec();
}
else if (VectorQuadratureFunctionCoefficient* c_u =
dynamic_cast<VectorQuadratureFunctionCoefficient*>(u))
{
// Assumed to be in lexicographical ordering
const QuadratureFunction &qFun = c_u->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == dim * nq * nf,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
vel.SetSize(dim * nq * nf);
+6 -6
View File
@@ -31,7 +31,7 @@ static void EADiffusionAssemble1D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -53,7 +53,7 @@ static void EADiffusionAssemble1D(const int NE,
{
val += r_Gj[k1] * D(k1, e) * r_Gi[k1];
}
A(i1, j1, e) = val;
A(i1, j1, e) += val;
}
}
});
@@ -75,7 +75,7 @@ static void EADiffusionAssemble2D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, 3, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -120,7 +120,7 @@ static void EADiffusionAssemble2D(const int NE,
+ gbi * D11 * gbj;
}
}
A(i1, i2, j1, j2, e) = val;
A(i1, i2, j1, j2, e) += val;
}
}
}
@@ -144,7 +144,7 @@ static void EADiffusionAssemble3D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 6, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -208,7 +208,7 @@ static void EADiffusionAssemble3D(const int NE,
}
}
}
A(i1, i2, i3, j1, j2, j3, e) = val;
A(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
+251 -157
View File
@@ -296,6 +296,19 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
coeff.SetSize(1);
coeff(0) = cQ->constant;
}
else if (QuadratureFunctionCoefficient* cQ =
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == ne*nq,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
coeff.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
coeff.SetSize(nq * ne);
@@ -1307,7 +1320,33 @@ static void PADiffusionApply3D(const int NE,
});
}
// Shared memory PA Diffusion Apply 3D kernel
// Half of B and G are stored in shared to get B, Bt, G and Gt.
// Indices computation for SmemPADiffusionApply3D.
static MFEM_HOST_DEVICE inline int qi(const int q, const int d, const int Q)
{
return (q<=d) ? q : Q-1-q;
}
static MFEM_HOST_DEVICE inline int dj(const int q, const int d, const int D)
{
return (q<=d) ? d : D-1-d;
}
static MFEM_HOST_DEVICE inline int qk(const int q, const int d, const int Q)
{
return (q<=d) ? Q-1-q : q;
}
static MFEM_HOST_DEVICE inline int dl(const int q, const int d, const int D)
{
return (q<=d) ? D-1-d : d;
}
static MFEM_HOST_DEVICE inline double sign(const int q, const int d)
{
return (q<=d) ? -1.0 : 1.0;
}
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPADiffusionApply3D(const int NE,
const Array<double> &b_,
@@ -1320,28 +1359,27 @@ static void SmemPADiffusionApply3D(const int NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= M1D, "");
MFEM_VERIFY(Q1D <= M1Q, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, 6, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
MFEM_FORALL_3D(e, NE, Q1D, Q1D, 1,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
MFEM_SHARED double sBG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
double (*Gt)[MQ1] = (double (*)[MQ1]) (sBG+1);
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
MFEM_SHARED double sBG[MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) sBG;
double (*G)[MD1] = (double (*)[MD1]) sBG;
double (*Bt)[MQ1] = (double (*)[MQ1]) sBG;
double (*Gt)[MQ1] = (double (*)[MQ1]) sBG;
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
@@ -1359,108 +1397,127 @@ static void SmemPADiffusionApply3D(const int NE,
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
}
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
const int i = qi(qx,dy,Q1D);
const int j = dj(qx,dy,D1D);
const int k = qk(qx,dy,Q1D);
const int l = dl(qx,dy,D1D);
B[i][j] = b(qx,dy);
G[k][l] = g(qx,dy) * sign(qx,dy);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
double u[D1D], v[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dx = 0; dx < D1D; ++dx)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double coords = X[dz][dy][dx];
u += coords * B[qx][dx];
v += coords * G[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
const int i = qi(qx,dx,Q1D);
const int j = dj(qx,dx,D1D);
const int k = qk(qx,dx,Q1D);
const int l = dl(qx,dx,D1D);
const double s = sign(qx,dx);
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
const double coords = X[dz][dy][dx];
u[dz] += coords * B[i][j];
v[dz] += coords * G[k][l] * s;
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
DDQ0[dz][dy][qx] = u[dz];
DDQ1[dz][dy][qx] = v[dz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
double u[D1D], v[D1D], w[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = w[dz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dy = 0; dy < D1D; ++dy)
{
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
const double O11 = d(q,0,e);
const double O12 = d(q,1,e);
const double O13 = d(q,2,e);
const double O22 = d(q,3,e);
const double O23 = d(q,4,e);
const double O33 = d(q,5,e);
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
const int i = qi(qy,dy,Q1D);
const int j = dj(qy,dy,D1D);
const int k = qk(qy,dy,Q1D);
const int l = dl(qy,dy,D1D);
const double s = sign(qy,dy);
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++)
{
u[dz] += DDQ1[dz][dy][qx] * B[i][j];
v[dz] += DDQ0[dz][dy][qx] * G[k][l] * s;
w[dz] += DDQ0[dz][dy][qx] * B[i][j];
}
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++)
{
DQQ0[dz][qy][qx] = u[dz];
DQQ1[dz][qy][qx] = v[dz];
DQQ2[dz][qy][qx] = w[dz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
{
const int i = qi(qz,dz,Q1D);
const int j = dj(qz,dz,D1D);
const int k = qk(qz,dz,Q1D);
const int l = dl(qz,dz,D1D);
const double s = sign(qz,dz);
u[qz] += DQQ0[dz][qy][qx] * B[i][j];
v[qz] += DQQ1[dz][qy][qx] * B[i][j];
w[qz] += DQQ2[dz][qy][qx] * G[k][l] * s;
}
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
{
const double O11 = d(qx,qy,qz,0,e);
const double O12 = d(qx,qy,qz,1,e);
const double O13 = d(qx,qy,qz,2,e);
const double O22 = d(qx,qy,qz,3,e);
const double O23 = d(qx,qy,qz,4,e);
const double O33 = d(qx,qy,qz,5,e);
const double gX = u[qz];
const double gY = v[qz];
const double gZ = w[qz];
QQQ0[qz][qy][qx] = (O11*gX) + (O12*gY) + (O13*gZ);
QQQ1[qz][qy][qx] = (O12*gX) + (O22*gY) + (O23*gZ);
QQQ2[qz][qy][qx] = (O13*gX) + (O23*gY) + (O33*gZ);
@@ -1468,78 +1525,112 @@ static void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(q,x,Q1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = b(q,d);
Gt[d][q] = g(q,d);
}
const int i = qi(q,d,Q1D);
const int j = dj(q,d,D1D);
const int k = qk(q,d,Q1D);
const int l = dl(q,d,D1D);
Bt[j][i] = b(q,d);
Gt[l][k] = g(q,d) * sign(q,d);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qx = 0; qx < Q1D; ++qx)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += QQQ0[qz][qy][qx] * Gt[dx][qx];
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
QQD1[qz][qy][dx] = v;
QQD2[qz][qy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
v += QQD1[qz][qy][dx] * Gt[dy][qy];
w += QQD2[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
QDD1[qz][dy][dx] = v;
QDD2[qz][dy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
const int i = qi(qx,dx,Q1D);
const int j = dj(qx,dx,D1D);
const int k = qk(qx,dx,Q1D);
const int l = dl(qx,dx,D1D);
const double s = sign(qx,dx);
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
v += QDD1[qz][dy][dx] * Bt[dz][qz];
w += QDD2[qz][dy][dx] * Gt[dz][qz];
u[qz] += QQQ0[qz][qy][qx] * Gt[l][k] * s;
v[qz] += QQQ1[qz][qy][qx] * Bt[j][i];
w[qz] += QQQ2[qz][qy][qx] * Bt[j][i];
}
y(dx,dy,dz,e) += (u + v + w);
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
QQD0[qz][qy][dx] = u[qz];
QQD1[qz][qy][dx] = v[qz];
QQD2[qz][qy][dx] = w[qz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qy = 0; qy < Q1D; ++qy)
{
const int i = qi(qy,dy,Q1D);
const int j = dj(qy,dy,D1D);
const int k = qk(qy,dy,Q1D);
const int l = dl(qy,dy,D1D);
const double s = sign(qy,dy);
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
u[qz] += QQD0[qz][qy][dx] * Bt[j][i];
v[qz] += QQD1[qz][qy][dx] * Gt[l][k] * s;
w[qz] += QQD2[qz][qy][dx] * Bt[j][i];
}
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
QDD0[qz][dy][dx] = u[qz];
QDD1[qz][dy][dx] = v[qz];
QDD2[qz][dy][dx] = w[qz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u[D1D], v[D1D], w[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz) { u[dz] = v[dz] = w[dz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
const int i = qi(qz,dz,Q1D);
const int j = dj(qz,dz,D1D);
const int k = qk(qz,dz,Q1D);
const int l = dl(qz,dz,D1D);
const double s = sign(qz,dz);
u[dz] += QDD0[qz][dy][dx] * Bt[j][i];
v[dz] += QDD1[qz][dy][dx] * Bt[j][i];
w[dz] += QDD2[qz][dy][dx] * Gt[l][k] * s;
}
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
y(dx,dy,dz,e) += (u[dz] + v[dz] + w[dz]);
}
}
}
@@ -1574,9 +1665,11 @@ static void PADiffusionApply(const int dim,
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
}
#endif // MFEM_USE_OCCA
const int ID = (D1D << 4 ) | Q1D;
if (dim == 2)
{
switch ((D1D << 4 ) | Q1D)
switch (ID)
{
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,D,X,Y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,D,X,Y);
@@ -1589,9 +1682,10 @@ static void PADiffusionApply(const int dim,
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
}
}
else if (dim == 3)
if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
switch (ID)
{
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,D,X,Y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,D,X,Y);
+1583 -75
View File
File diff suppressed because it is too large Load Diff
+11 -6
View File
@@ -23,11 +23,6 @@ using namespace std;
namespace mfem
{
// Local maximum size of dofs and quads in 1D
constexpr int HDIV_MAX_D1D = 5;
constexpr int HDIV_MAX_Q1D = 6;
// PA H(div) Mass Assemble 2D kernel
void PAHdivSetup2D(const int Q1D,
const int NE,
@@ -114,6 +109,8 @@ void PAHdivMassApply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
@@ -238,6 +235,7 @@ void PAHdivMassAssembleDiagonal2D(const int D1D,
Vector &_diag)
{
constexpr static int VDIM = 2;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
@@ -614,6 +612,8 @@ static void PADivDivApply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bot = Reshape(_Bot.Read(), D1D-1, Q1D);
@@ -977,6 +977,7 @@ static void PADivDivAssembleDiagonal2D(const int D1D,
Vector &_diag)
{
constexpr static int VDIM = 2;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
@@ -1400,6 +1401,8 @@ static void PAHdivL2Apply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
@@ -1666,6 +1669,8 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto L2Bo = Reshape(_L2Bo.Read(), Q1D, L2D1D);
auto Gct = Reshape(_Gct.Read(), D1D, Q1D);
@@ -1724,7 +1729,7 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
for (int qy = 0; qy < Q1D; ++qy)
{
double aX[HDIV_MAX_D1D];
double aX[MAX_D1D];
int osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y components
+6 -6
View File
@@ -30,7 +30,7 @@ static void EAMassAssemble1D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -52,7 +52,7 @@ static void EAMassAssemble1D(const int NE,
{
val += r_Bi[k1] * r_Bj[k1] * D(k1, e);
}
M(i1, j1, e) = val;
M(i1, j1, e) += val;
}
}
});
@@ -72,7 +72,7 @@ static void EAMassAssemble2D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -114,7 +114,7 @@ static void EAMassAssemble2D(const int NE,
* s_D[k1][k2];
}
}
M(i1, i2, j1, j2, e) = val;
M(i1, i2, j1, j2, e) += val;
}
}
}
@@ -136,7 +136,7 @@ static void EAMassAssemble3D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -189,7 +189,7 @@ static void EAMassAssemble3D(const int NE,
}
}
}
M(i1, i2, i3, j1, j2, j3, e) = val;
M(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
+17
View File
@@ -41,6 +41,8 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
InitCeedCoeff(Q, ptr);
return CeedPAMassAssemble(fes, *ir, *ptr);
}
#else
MFEM_CONTRACT_VAR(force);
#endif
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
@@ -62,6 +64,19 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
coeff.SetSize(1);
coeff(0) = cQ->constant;
}
else if (QuadratureFunctionCoefficient* cQ =
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == nq * ne,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
coeff.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
coeff.SetSize(nq * ne);
@@ -639,6 +654,7 @@ static void SmemPAMassApply2D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
@@ -902,6 +918,7 @@ static void SmemPAMassApply3D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
+1 -1
View File
@@ -25,7 +25,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
if (ne == 0) { return; }
const int dofs = fes.GetFE(0)->GetDof();
auto A = Reshape(ea_data_tmp.Write(), dofs, dofs, ne);
auto AT = Reshape(ea_data.Write(), dofs, dofs, ne);
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
MFEM_FORALL(e, ne,
{
for (int i = 0; i < dofs; i++)
+737 -39
View File
@@ -9,12 +9,14 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
namespace mfem
{
void PAHcurlSetup2D(const int Q1D,
const int coeffDim,
const int NE,
const Array<double> &w,
const Vector &j,
@@ -22,6 +24,7 @@ void PAHcurlSetup2D(const int Q1D,
Vector &op);
void PAHcurlSetup3D(const int Q1D,
const int coeffDim,
const int NE,
const Array<double> &w,
const Vector &j,
@@ -31,6 +34,7 @@ void PAHcurlSetup3D(const int Q1D,
void PAHcurlMassAssembleDiagonal2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Vector &_op,
@@ -39,6 +43,7 @@ void PAHcurlMassAssembleDiagonal2D(const int D1D,
void PAHcurlMassAssembleDiagonal3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Vector &_op,
@@ -47,6 +52,7 @@ void PAHcurlMassAssembleDiagonal3D(const int D1D,
void PAHcurlMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
@@ -58,6 +64,7 @@ void PAHcurlMassApply2D(const int D1D,
void PAHcurlMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
@@ -140,20 +147,573 @@ void PAHdivMassApply3D(const int D1D,
const Vector &_x,
Vector &_y);
void PAHcurlL2Setup(const int NQ,
const int coeffDim,
const int NE,
const Array<double> &w,
Vector &_coeff,
Vector &op);
// PA H(curl) x H(div) mass assemble 3D kernel, with factor
// dF^{-1} C dF for a vector or matrix coefficient C.
// If transpose, use dF^T C dF^{-T} for H(div) x H(curl).
void PAHcurlHdivSetup3D(const int Q1D,
const int coeffDim,
const int NE,
const bool transpose,
const Array<double> &_w,
const Vector &j,
Vector &_coeff,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
const bool symmetric = (coeffDim != 9);
auto W = _w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto coeff = Reshape(_coeff.Read(), coeffDim, NQ, NE);
auto y = Reshape(op.Write(), 9, NQ, NE);
const int i11 = 0;
const int i12 = transpose ? 3 : 1;
const int i13 = transpose ? 6 : 2;
const int i21 = transpose ? 1 : 3;
const int i22 = 4;
const int i23 = transpose ? 7 : 5;
const int i31 = transpose ? 2 : 6;
const int i32 = transpose ? 5 : 7;
const int i33 = 8;
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double w_detJ = W[q] / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
if (coeffDim == 6 || coeffDim == 9) // Matrix coefficient version
{
// First compute entries of R = MJ
const double M11 = (!symmetric) ? coeff(i11, q, e) : coeff(0, q, e);
const double M12 = (!symmetric) ? coeff(i12, q, e) : coeff(1, q, e);
const double M13 = (!symmetric) ? coeff(i13, q, e) : coeff(2, q, e);
const double M21 = (!symmetric) ? coeff(i21, q, e) : M12;
const double M22 = (!symmetric) ? coeff(i22, q, e) : coeff(3, q, e);
const double M23 = (!symmetric) ? coeff(i23, q, e) : coeff(4, q, e);
const double M31 = (!symmetric) ? coeff(i31, q, e) : M13;
const double M32 = (!symmetric) ? coeff(i32, q, e) : M23;
const double M33 = (!symmetric) ? coeff(i33, q, e) : coeff(5, q, e);
const double R11 = M11*J11 + M12*J12 + M13*J13;
const double R12 = M11*J21 + M12*J22 + M13*J23;
const double R13 = M11*J31 + M12*J32 + M13*J33;
const double R21 = M21*J11 + M22*J12 + M23*J13;
const double R22 = M21*J21 + M22*J22 + M23*J23;
const double R23 = M21*J31 + M22*J32 + M23*J33;
const double R31 = M31*J11 + M32*J12 + M33*J13;
const double R32 = M31*J21 + M32*J22 + M33*J23;
const double R33 = M31*J31 + M32*J32 + M33*J33;
// Now set y to detJ J^{-1} R = adj(J) R
y(i11,q,e) = w_detJ * (A11*R11 + A12*R21 + A13*R31); // 1,1
y(i12,q,e) = w_detJ * (A11*R12 + A12*R22 + A13*R32); // 1,2
y(i13,q,e) = w_detJ * (A11*R13 + A12*R23 + A13*R33); // 1,3
y(i21,q,e) = w_detJ * (A21*R11 + A22*R21 + A23*R31); // 2,1
y(i22,q,e) = w_detJ * (A21*R12 + A22*R22 + A23*R32); // 2,2
y(i23,q,e) = w_detJ * (A21*R13 + A22*R23 + A23*R33); // 2,3
y(i31,q,e) = w_detJ * (A31*R11 + A32*R21 + A33*R31); // 3,1
y(i32,q,e) = w_detJ * (A31*R12 + A32*R22 + A33*R32); // 3,2
y(i33,q,e) = w_detJ * (A31*R13 + A32*R23 + A33*R33); // 3,3
}
else if (coeffDim == 3) // Vector coefficient version
{
const double D1 = coeff(0, q, e);
const double D2 = coeff(1, q, e);
const double D3 = coeff(2, q, e);
// detJ J^{-1} DJ = adj(J) DJ
y(i11,q,e) = w_detJ * (D1*A11*J11 + D2*A12*J21 + D3*A13*J31); // 1,1
y(i12,q,e) = w_detJ * (D1*A11*J12 + D2*A12*J22 + D3*A13*J32); // 1,2
y(i13,q,e) = w_detJ * (D1*A11*J13 + D2*A12*J23 + D3*A13*J33); // 1,3
y(i21,q,e) = w_detJ * (D1*A21*J11 + D2*A22*J21 + D3*A23*J31); // 2,1
y(i22,q,e) = w_detJ * (D1*A21*J12 + D2*A22*J22 + D3*A23*J32); // 2,2
y(i23,q,e) = w_detJ * (D1*A21*J13 + D2*A22*J23 + D3*A23*J33); // 2,3
y(i31,q,e) = w_detJ * (D1*A31*J11 + D2*A32*J21 + D3*A33*J31); // 3,1
y(i32,q,e) = w_detJ * (D1*A31*J12 + D2*A32*J22 + D3*A33*J32); // 3,2
y(i33,q,e) = w_detJ * (D1*A31*J13 + D2*A32*J23 + D3*A33*J33); // 3,3
}
}
});
}
// PA H(curl) x H(div) mass assemble 2D kernel, with factor
// dF^{-1} C dF for a vector or matrix coefficient C.
// If transpose, use dF^T C dF^{-T} for H(div) x H(curl).
void PAHcurlHdivSetup2D(const int Q1D,
const int coeffDim,
const int NE,
const bool transpose,
const Array<double> &_w,
const Vector &j,
Vector &_coeff,
Vector &op)
{
const int NQ = Q1D*Q1D;
const bool symmetric = (coeffDim != 4);
auto W = _w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto coeff = Reshape(_coeff.Read(), coeffDim, NQ, NE);
auto y = Reshape(op.Write(), 4, NQ, NE);
const int i11 = 0;
const int i12 = transpose ? 2 : 1;
const int i21 = transpose ? 1 : 2;
const int i22 = 3;
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double w_detJ = W[q] / (J11*J22) - (J21*J12);
if (coeffDim == 3 || coeffDim == 4) // Matrix coefficient version
{
// First compute entries of R = MJ
const double M11 = coeff(i11, q, e);
const double M12 = (!symmetric) ? coeff(i12, q, e) : coeff(1, q, e);
const double M21 = (!symmetric) ? coeff(i21, q, e) : M12;
const double M22 = (!symmetric) ? coeff(i22, q, e) : coeff(2, q, e);
const double R11 = M11*J11 + M12*J21;
const double R12 = M11*J12 + M12*J22;
const double R21 = M21*J11 + M22*J21;
const double R22 = M21*J12 + M22*J22;
// Now set y to J^{-1} R
y(i11,q,e) = w_detJ * ( J22*R11 - J12*R21); // 1,1
y(i12,q,e) = w_detJ * ( J22*R12 - J12*R22); // 1,2
y(i21,q,e) = w_detJ * (-J21*R11 + J11*R21); // 2,1
y(i22,q,e) = w_detJ * (-J21*R12 + J11*R22); // 2,2
}
else if (coeffDim == 2) // Vector coefficient version
{
const double D1 = coeff(0, q, e);
const double D2 = coeff(1, q, e);
const double R11 = D1*J11;
const double R12 = D1*J12;
const double R21 = D2*J21;
const double R22 = D2*J22;
y(i11,q,e) = w_detJ * ( J22*R11 - J12*R21); // 1,1
y(i12,q,e) = w_detJ * ( J22*R12 - J12*R22); // 1,2
y(i21,q,e) = w_detJ * (-J21*R11 + J11*R21); // 2,1
y(i22,q,e) = w_detJ * (-J21*R12 + J11*R22); // 2,2
}
}
});
}
// Mass operator for H(curl) and H(div) functions, using Piola transformations
// u = dF^{-T} \hat{u} in H(curl), v = (1 / det dF) dF \hat{v} in H(div).
void PAHcurlHdivMassApply3D(const int D1D,
const int D1Dtest,
const int Q1D,
const int NE,
const bool scalarCoeff,
const bool trialHcurl,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y)
{
constexpr static int MAX_D1D = HCURL_MAX_D1D;
constexpr static int MAX_Q1D = HCURL_MAX_Q1D;
MFEM_VERIFY(D1D <= MAX_D1D, "Error: D1D > MAX_D1D");
MFEM_VERIFY(Q1D <= MAX_Q1D, "Error: Q1D > MAX_Q1D");
constexpr static int VDIM = 3;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
auto Bot = Reshape(_Bot.Read(), D1Dtest-1, Q1D);
auto Bct = Reshape(_Bct.Read(), D1Dtest, Q1D);
auto op = Reshape(_op.Read(), scalarCoeff ? 1 : 9, Q1D, Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), 3*(D1D-1)*D1D*(trialHcurl ? D1D : D1D-1), NE);
auto y = Reshape(_y.ReadWrite(), 3*(D1Dtest-1)*D1Dtest*
(trialHcurl ? D1Dtest-1 : D1Dtest), NE);
MFEM_FORALL(e, NE,
{
double mass[MAX_Q1D][MAX_Q1D][MAX_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
for (int c = 0; c < VDIM; ++c)
{
mass[qz][qy][qx][c] = 0.0;
}
}
}
}
int osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y, z trial components
{
const int D1Dz = trialHcurl ? ((c == 2) ? D1D - 1 : D1D) :
((c == 2) ? D1D : D1D - 1);
const int D1Dy = trialHcurl ? ((c == 1) ? D1D - 1 : D1D) :
((c == 1) ? D1D : D1D - 1);
const int D1Dx = trialHcurl ? ((c == 0) ? D1D - 1 : D1D) :
((c == 0) ? D1D : D1D - 1);
for (int dz = 0; dz < D1Dz; ++dz)
{
double massXY[MAX_Q1D][MAX_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
massXY[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1Dy; ++dy)
{
double massX[MAX_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
massX[qx] = 0.0;
}
for (int dx = 0; dx < D1Dx; ++dx)
{
const double t = x(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e);
for (int qx = 0; qx < Q1D; ++qx)
{
massX[qx] += t * (trialHcurl ? ((c == 0) ? Bo(qx,dx) : Bc(qx,dx)) :
((c == 0) ? Bc(qx,dx) : Bo(qx,dx)));
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = trialHcurl ? ((c == 1) ? Bo(qy,dy) : Bc(qy,dy)) :
((c == 1) ? Bc(qy,dy) : Bo(qy,dy));
for (int qx = 0; qx < Q1D; ++qx)
{
const double wx = massX[qx];
massXY[qy][qx] += wx * wy;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = trialHcurl ? ((c == 2) ? Bo(qz,dz) : Bc(qz,dz)) :
((c == 2) ? Bc(qz,dz) : Bo(qz,dz));
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
mass[qz][qy][qx][c] += massXY[qy][qx] * wz;
}
}
}
}
osc += D1Dx * D1Dy * D1Dz;
} // loop (c) over components
// Apply D operator.
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double O11 = op(0,qx,qy,qz,e);
const double O12 = scalarCoeff ? 0.0 : op(1,qx,qy,qz,e);
const double O13 = scalarCoeff ? 0.0 : op(2,qx,qy,qz,e);
const double O21 = scalarCoeff ? 0.0 : op(3,qx,qy,qz,e);
const double O22 = scalarCoeff ? O11 : op(4,qx,qy,qz,e);
const double O23 = scalarCoeff ? 0.0 : op(5,qx,qy,qz,e);
const double O31 = scalarCoeff ? 0.0 : op(6,qx,qy,qz,e);
const double O32 = scalarCoeff ? 0.0 : op(7,qx,qy,qz,e);
const double O33 = scalarCoeff ? O11 : op(8,qx,qy,qz,e);
const double massX = mass[qz][qy][qx][0];
const double massY = mass[qz][qy][qx][1];
const double massZ = mass[qz][qy][qx][2];
mass[qz][qy][qx][0] = (O11*massX)+(O12*massY)+(O13*massZ);
mass[qz][qy][qx][1] = (O21*massX)+(O22*massY)+(O23*massZ);
mass[qz][qy][qx][2] = (O31*massX)+(O32*massY)+(O33*massZ);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double massXY[HDIV_MAX_D1D][HDIV_MAX_D1D];
osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y, z test components
{
const int D1Dz = trialHcurl ? ((c == 2) ? D1Dtest : D1Dtest - 1) :
((c == 2) ? D1Dtest - 1 : D1Dtest);
const int D1Dy = trialHcurl ? ((c == 1) ? D1Dtest : D1Dtest - 1) :
((c == 1) ? D1Dtest - 1 : D1Dtest);
const int D1Dx = trialHcurl ? ((c == 0) ? D1Dtest : D1Dtest - 1) :
((c == 0) ? D1Dtest - 1 : D1Dtest);
for (int dy = 0; dy < D1Dy; ++dy)
{
for (int dx = 0; dx < D1Dx; ++dx)
{
massXY[dy][dx] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double massX[HDIV_MAX_D1D];
for (int dx = 0; dx < D1Dx; ++dx)
{
massX[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dx = 0; dx < D1Dx; ++dx)
{
massX[dx] += mass[qz][qy][qx][c] * (trialHcurl ?
((c == 0) ? Bct(dx,qx) : Bot(dx,qx)) :
((c == 0) ? Bot(dx,qx) : Bct(dx,qx)));
}
}
for (int dy = 0; dy < D1Dy; ++dy)
{
const double wy = trialHcurl ? ((c == 1) ? Bct(dy,qy) : Bot(dy,qy)) :
((c == 1) ? Bot(dy,qy) : Bct(dy,qy));
for (int dx = 0; dx < D1Dx; ++dx)
{
massXY[dy][dx] += massX[dx] * wy;
}
}
}
for (int dz = 0; dz < D1Dz; ++dz)
{
const double wz = trialHcurl ? ((c == 2) ? Bct(dz,qz) : Bot(dz,qz)) :
((c == 2) ? Bot(dz,qz) : Bct(dz,qz));
for (int dy = 0; dy < D1Dy; ++dy)
{
for (int dx = 0; dx < D1Dx; ++dx)
{
y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc, e) +=
massXY[dy][dx] * wz;
}
}
}
osc += D1Dx * D1Dy * D1Dz;
} // loop c
} // loop qz
}); // end of element loop
}
// Mass operator for H(curl) and H(div) functions, using Piola transformations
// u = dF^{-T} \hat{u} in H(curl), v = (1 / det dF) dF \hat{v} in H(div).
void PAHcurlHdivMassApply2D(const int D1D,
const int D1Dtest,
const int Q1D,
const int NE,
const bool scalarCoeff,
const bool trialHcurl,
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y)
{
constexpr static int MAX_D1D = HCURL_MAX_D1D;
constexpr static int MAX_Q1D = HCURL_MAX_Q1D;
MFEM_VERIFY(D1D <= MAX_D1D, "Error: D1D > MAX_D1D");
MFEM_VERIFY(Q1D <= MAX_Q1D, "Error: Q1D > MAX_Q1D");
constexpr static int VDIM = 2;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
auto Bot = Reshape(_Bot.Read(), D1Dtest-1, Q1D);
auto Bct = Reshape(_Bct.Read(), D1Dtest, Q1D);
auto op = Reshape(_op.Read(), scalarCoeff ? 1 : 4, Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), 2*(D1D-1)*D1D, NE);
auto y = Reshape(_y.ReadWrite(), 2*(D1Dtest-1)*D1Dtest, NE);
MFEM_FORALL(e, NE,
{
double mass[MAX_Q1D][MAX_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
for (int c = 0; c < VDIM; ++c)
{
mass[qy][qx][c] = 0.0;
}
}
}
int osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y trial components
{
const int D1Dy = trialHcurl ? ((c == 1) ? D1D - 1 : D1D) :
((c == 1) ? D1D : D1D - 1);
const int D1Dx = trialHcurl ? ((c == 0) ? D1D - 1 : D1D) :
((c == 0) ? D1D : D1D - 1);
for (int dy = 0; dy < D1Dy; ++dy)
{
double massX[MAX_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
massX[qx] = 0.0;
}
for (int dx = 0; dx < D1Dx; ++dx)
{
const double t = x(dx + (dy * D1Dx) + osc, e);
for (int qx = 0; qx < Q1D; ++qx)
{
massX[qx] += t * (trialHcurl ? ((c == 0) ? Bo(qx,dx) : Bc(qx,dx)) :
((c == 0) ? Bc(qx,dx) : Bo(qx,dx)));
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = trialHcurl ? ((c == 1) ? Bo(qy,dy) : Bc(qy,dy)) :
((c == 1) ? Bc(qy,dy) : Bo(qy,dy));
for (int qx = 0; qx < Q1D; ++qx)
{
mass[qy][qx][c] += massX[qx] * wy;
}
}
}
osc += D1Dx * D1Dy;
} // loop (c) over components
// Apply D operator.
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double O11 = op(0,qx,qy,e);
const double O12 = scalarCoeff ? 0.0 : op(1,qx,qy,e);
const double O21 = scalarCoeff ? 0.0 : op(2,qx,qy,e);
const double O22 = scalarCoeff ? O11 : op(3,qx,qy,e);
const double massX = mass[qy][qx][0];
const double massY = mass[qy][qx][1];
mass[qy][qx][0] = (O11*massX)+(O12*massY);
mass[qy][qx][1] = (O21*massX)+(O22*massY);
}
}
osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y test components
{
const int D1Dy = trialHcurl ? ((c == 1) ? D1Dtest : D1Dtest - 1) :
((c == 1) ? D1Dtest - 1 : D1Dtest);
const int D1Dx = trialHcurl ? ((c == 0) ? D1Dtest : D1Dtest - 1) :
((c == 0) ? D1Dtest - 1 : D1Dtest);
for (int qy = 0; qy < Q1D; ++qy)
{
double massX[HDIV_MAX_D1D];
for (int dx = 0; dx < D1Dx; ++dx)
{
massX[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dx = 0; dx < D1Dx; ++dx)
{
massX[dx] += mass[qy][qx][c] * (trialHcurl ?
((c == 0) ? Bct(dx,qx) : Bot(dx,qx)) :
((c == 0) ? Bot(dx,qx) : Bct(dx,qx)));
}
}
for (int dy = 0; dy < D1Dy; ++dy)
{
const double wy = trialHcurl ? ((c == 1) ? Bct(dy,qy) : Bot(dy,qy)) :
((c == 1) ? Bot(dy,qy) : Bct(dy,qy));
for (int dx = 0; dx < D1Dx; ++dx)
{
y(dx + (dy * D1Dx) + osc, e) += massX[dx] * wy;
}
}
}
osc += D1Dx * D1Dy;
} // loop c
}); // end of element loop
}
void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement *fel = fes.GetFE(0);
AssemblePA(fes, fes);
}
const VectorTensorFiniteElement *el =
dynamic_cast<const VectorTensorFiniteElement*>(fel);
MFEM_VERIFY(el != NULL, "Only VectorTensorFiniteElement is supported!");
void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes)
{
// Assumes tensor-product elements
Mesh *mesh = trial_fes.GetMesh();
const FiniteElement *trial_fel = trial_fes.GetFE(0);
const VectorTensorFiniteElement *trial_el =
dynamic_cast<const VectorTensorFiniteElement*>(trial_fel);
MFEM_VERIFY(trial_el != NULL, "Only VectorTensorFiniteElement is supported!");
const FiniteElement *test_fel = test_fes.GetFE(0);
const VectorTensorFiniteElement *test_el =
dynamic_cast<const VectorTensorFiniteElement*>(test_fel);
MFEM_VERIFY(test_el != NULL, "Only VectorTensorFiniteElement is supported!");
const IntegrationRule *ir
= IntRule ? IntRule : &MassIntegrator::GetRule(*el, *el,
= IntRule ? IntRule : &MassIntegrator::GetRule(*trial_el, *trial_el,
*mesh->GetElementTransformation(0));
const int dims = el->GetDim();
const int dims = trial_el->GetDim();
MFEM_VERIFY(dims == 2 || dims == 3, "");
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
@@ -161,53 +721,160 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
dim = mesh->Dimension();
MFEM_VERIFY(dim == 2 || dim == 3, "");
ne = fes.GetNE();
ne = trial_fes.GetNE();
MFEM_VERIFY(ne == test_fes.GetNE(),
"Different meshes for test and trial spaces");
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
mapsC = &el->GetDofToQuad(*ir, DofToQuad::TENSOR);
mapsO = &el->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
mapsC = &trial_el->GetDofToQuad(*ir, DofToQuad::TENSOR);
mapsO = &trial_el->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
dofs1D = mapsC->ndof;
quad1D = mapsC->nqpt;
mapsCtest = &test_el->GetDofToQuad(*ir, DofToQuad::TENSOR);
mapsOtest = &test_el->GetDofToQuadOpen(*ir, DofToQuad::TENSOR);
dofs1Dtest = mapsCtest->ndof;
MFEM_VERIFY(dofs1D == mapsO->ndof + 1 && quad1D == mapsO->nqpt, "");
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
trial_fetype = trial_el->GetDerivType();
test_fetype = test_el->GetDerivType();
Vector coeff(ne * nq);
const int MQsymmDim = MQ ? (MQ->GetWidth() * (MQ->GetWidth() + 1)) / 2 : 0;
const int MQfullDim = MQ ? (MQ->GetHeight() * MQ->GetWidth()) : 0;
const int MQdim = MQ ? (MQ->IsSymmetric() ? MQsymmDim : MQfullDim) : 0;
const int coeffDim = MQ ? MQdim : (VQ ? VQ->GetVDim() : 1);
symmetric = MQ ? MQ->IsSymmetric() : true;
if ((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL))
pa_data.SetSize((coeffDim == 1 ? 1 : dim*dim) * nq * ne,
Device::GetMemoryType());
else
pa_data.SetSize((symmetric ? symmDims : MQfullDim) * nq * ne,
Device::GetMemoryType());
Vector coeff(coeffDim * ne * nq);
coeff = 1.0;
if (Q)
auto coeffh = Reshape(coeff.HostWrite(), coeffDim, nq, ne);
if (Q || VQ || MQ)
{
Vector D(VQ ? coeffDim : 0);
DenseMatrix M;
Vector Msymm;
if (MQ)
{
if (symmetric)
{
Msymm.SetSize(MQsymmDim);
}
else
{
M.SetSize(dim);
}
}
if (VQ)
{
MFEM_VERIFY(coeffDim == dim, "");
}
if (MQ)
{
MFEM_VERIFY(coeffDim == MQdim, "");
MFEM_VERIFY(MQ->GetHeight() == dim && MQ->GetWidth() == dim, "");
}
for (int e=0; e<ne; ++e)
{
ElementTransformation *tr = mesh->GetElementTransformation(e);
for (int p=0; p<nq; ++p)
{
coeff[p + (e * nq)] = Q->Eval(*tr, ir->IntPoint(p));
if (MQ)
{
if (MQ->IsSymmetric())
{
MQ->EvalSymmetric(Msymm, *tr, ir->IntPoint(p));
for (int i=0; i<MQsymmDim; ++i)
{
coeffh(i, p, e) = Msymm[i];
}
}
else
{
MQ->Eval(M, *tr, ir->IntPoint(p));
for (int i=0; i<dim; ++i)
for (int j=0; j<dim; ++j)
{
coeffh(j+(i*dim), p, e) = M(i,j);
}
}
}
else if (VQ)
{
VQ->Eval(D, *tr, ir->IntPoint(p));
for (int i=0; i<coeffDim; ++i)
{
coeffh(i, p, e) = D[i];
}
}
else
{
coeffh(0, p, e) = Q->Eval(*tr, ir->IntPoint(p));
}
}
}
}
fetype = el->GetDerivType();
if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype
&& dim == 3)
{
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
PAHcurlSetup3D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
else if (trial_fetype == mfem::FiniteElement::CURL
&& test_fetype == trial_fetype && dim == 2)
{
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
PAHcurlSetup2D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (el->GetDerivType() == mfem::FiniteElement::DIV && dim == 3)
else if (trial_fetype == mfem::FiniteElement::DIV
&& test_fetype == trial_fetype && dim == 3)
{
PAHdivSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (el->GetDerivType() == mfem::FiniteElement::DIV && dim == 2)
else if (trial_fetype == mfem::FiniteElement::DIV
&& test_fetype == trial_fetype && dim == 2)
{
PAHdivSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL)) &&
test_fel->GetOrder() == trial_fel->GetOrder())
{
if (coeffDim == 1)
{
PAHcurlL2Setup(nq, coeffDim, ne, ir->GetWeights(), coeff, pa_data);
}
else
{
const bool tr = (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL);
if (dim == 3)
PAHcurlHdivSetup3D(quad1D, coeffDim, ne, tr, ir->GetWeights(),
geom->J, coeff, pa_data);
else
PAHcurlHdivSetup2D(quad1D, coeffDim, ne, tr, ir->GetWeights(),
geom->J, coeff, pa_data);
}
}
else
{
MFEM_ABORT("Unknown kernel.");
@@ -218,12 +885,13 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
{
if (dim == 3)
{
if (fetype == mfem::FiniteElement::CURL)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
PAHcurlMassAssembleDiagonal3D(dofs1D, quad1D, ne,
PAHcurlMassAssembleDiagonal3D(dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, pa_data, diag);
}
else if (fetype == mfem::FiniteElement::DIV)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassAssembleDiagonal3D(dofs1D, quad1D, ne,
mapsO->B, mapsC->B, pa_data, diag);
@@ -235,12 +903,13 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
}
else
{
if (fetype == mfem::FiniteElement::CURL)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
PAHcurlMassAssembleDiagonal2D(dofs1D, quad1D, ne,
PAHcurlMassAssembleDiagonal2D(dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, pa_data, diag);
}
else if (fetype == mfem::FiniteElement::DIV)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassAssembleDiagonal2D(dofs1D, quad1D, ne,
mapsO->B, mapsC->B, pa_data, diag);
@@ -256,16 +925,33 @@ void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
if (dim == 3)
{
if (fetype == mfem::FiniteElement::CURL)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
PAHcurlMassApply3D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (fetype == mfem::FiniteElement::DIV)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassApply3D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
else if (trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV)
{
const bool scalarCoeff = !(VQ || MQ);
PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
true, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL)
{
const bool scalarCoeff = !(VQ || MQ);
PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
false, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
@@ -273,16 +959,28 @@ void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
else
{
if (fetype == mfem::FiniteElement::CURL)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
PAHcurlMassApply2D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (fetype == mfem::FiniteElement::DIV)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassApply2D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
else if ((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL))
{
const bool scalarCoeff = !(VQ || MQ);
const bool trialHcurl = (trial_fetype == mfem::FiniteElement::CURL);
PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
trialHcurl, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
@@ -348,12 +1046,12 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
// Use the same setup functions as VectorFEMassIntegrator.
if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
{
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
PAHcurlSetup3D(quad1D, 1, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
{
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
PAHcurlSetup2D(quad1D, 1, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else
+25
View File
@@ -319,6 +319,31 @@ void MatrixFunctionCoefficient::Eval(DenseMatrix &K, ElementTransformation &T,
}
}
void MatrixFunctionCoefficient::EvalSymmetric(Vector &K,
ElementTransformation &T,
const IntegrationPoint &ip)
{
MFEM_VERIFY(symmetric && height == width && height < 4 && SymmFunction,
"MatrixFunctionCoefficient is not symmetric");
double x[3];
Vector transip(x, 3);
T.Transform(ip, transip);
K.SetSize((width * (width + 1)) / 2); // 1x1: 1, 2x2: 3, 3x3: 6
if (SymmFunction)
{
(*SymmFunction)(transip, K);
}
if (Q)
{
K *= Q->Eval(T, ip, GetTime());
}
}
MatrixArrayCoefficient::MatrixArrayCoefficient (int dim)
: MatrixCoefficient (dim)
{
+38 -3
View File
@@ -30,7 +30,10 @@ class ParMesh;
/** @brief Base class Coefficients that optionally depend on space and time.
These are used by the BilinearFormIntegrator, LinearFormIntegrator, and
NonlinearFormIntegrator classes to represent the physical coefficients in
the PDEs that are being discretized. */
the PDEs that are being discretized. This class can also be used in a more
general way to represent functions that don't necessarily belong to a FE
space, e.g., to project onto GridFunctions to use as initial conditions,
exact solutions, etc. See, e.g., ex4 or ex22 for these uses. */
class Coefficient
{
protected:
@@ -692,13 +695,16 @@ class MatrixCoefficient
protected:
int height, width;
double time;
bool symmetric;
public:
/// Construct a dim x dim matrix coefficient.
explicit MatrixCoefficient(int dim) { height = width = dim; time = 0.; }
explicit MatrixCoefficient(int dim, bool symm=false)
{ height = width = dim; time = 0.; symmetric = symm; }
/// Construct a h x w matrix coefficient.
MatrixCoefficient(int h, int w) : height(h), width(w), time(0.) { }
MatrixCoefficient(int h, int w, bool symm=false) :
height(h), width(w), time(0.), symmetric(symm) { }
/// Set the time for time dependent coefficients
void SetTime(double t) { time = t; }
@@ -715,6 +721,9 @@ public:
/// For backward compatibility get the width of the matrix.
int GetVDim() const { return width; }
void SetSymmetric(bool s) { symmetric = s; }
bool IsSymmetric() const { return symmetric; }
/** @brief Evaluate the matrix coefficient in the element described by @a T
at the point @a ip, storing the result in @a K. */
/** @note When this method is called, the caller must make sure that the
@@ -723,6 +732,15 @@ public:
virtual void Eval(DenseMatrix &K, ElementTransformation &T,
const IntegrationPoint &ip) = 0;
/** @brief Evaluate the upper triangular entries of the matrix coefficient
in the symmetric case, similarly to Eval. Matrix entry (i,j) is stored
in K[j - i + os_i] for 0 <= i <= j < width, os_0 = 0,
os_{i+1} = os_i + width - i. That is, K = {M(0,0), ..., M(0,w-1),
M(1,1), ..., M(1,w-1), ..., M(w-1,w-1) with w = width. */
virtual void EvalSymmetric(Vector &K, ElementTransformation &T,
const IntegrationPoint &ip)
{ mfem_error("MatrixCoefficient::EvalSymmetric"); }
virtual ~MatrixCoefficient() { }
};
@@ -750,6 +768,7 @@ class MatrixFunctionCoefficient : public MatrixCoefficient
{
private:
void (*Function)(const Vector &, DenseMatrix &);
void (*SymmFunction)(const Vector &, Vector &);
void (*TDFunction)(const Vector &, double, DenseMatrix &);
Coefficient *Q;
DenseMatrix mat;
@@ -787,10 +806,26 @@ public:
mat.SetSize(0);
}
/// Construct a symmetric square matrix coefficient from a C-function
/// defining a vector function used by EvalSymmetric
MatrixFunctionCoefficient(int dim, void (*F)(const Vector &, Vector &),
Coefficient *q = NULL)
: MatrixCoefficient(dim, true), Q(q)
{
SymmFunction = F;
Function = NULL;
TDFunction = NULL;
mat.SetSize(0);
}
/// Evaluate the matrix coefficient at @a ip.
virtual void Eval(DenseMatrix &K, ElementTransformation &T,
const IntegrationPoint &ip);
/// Evaluate the symmetric matrix coefficient at @a ip.
virtual void EvalSymmetric(Vector &K, ElementTransformation &T,
const IntegrationPoint &ip);
virtual ~MatrixFunctionCoefficient() { }
};
+446 -197
View File
File diff suppressed because it is too large Load Diff
+39 -9
View File
@@ -99,8 +99,8 @@ public:
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a ComplexLinearForm on the FiniteElementSpace @a f, using
the same integrators as the LinearForms @a lfr (real) and @a lfi (imag) .
/** @brief Create a ComplexLinearForm on the FiniteElementSpace @a fes, using
the same integrators as the LinearForms @a lf_r (real) and @a lf_i (imag).
The pointer @a fes is not owned by the newly constructed object.
@@ -195,8 +195,8 @@ private:
BilinearForm *blfr;
BilinearForm *blfi;
/* These methods check if the real/imag parts of the sesqulinear form are not
empty */
/* These methods check if the real/imag parts of the sesquilinear form are
not empty */
bool RealInteg();
bool ImagInteg();
@@ -204,7 +204,7 @@ public:
SesquilinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
/** @brief Create a SesquilinearForm on the FiniteElementSpace @a f, using
/** @brief Create a SesquilinearForm on the FiniteElementSpace @a fes, using
the same integrators as the BilinearForms @a bfr and @a bfi .
The pointer @a fes is not owned by the newly constructed object.
@@ -219,6 +219,21 @@ public:
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::FULL (default)
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
blfr->SetAssemblyLevel(assembly_level);
blfi->SetAssemblyLevel(assembly_level);
}
BilinearForm & real() { return *blfr; }
BilinearForm & imag() { return *blfi; }
const BilinearForm & real() const { return *blfr; }
@@ -309,7 +324,7 @@ protected:
public:
/* @brief Construct a ParComplexGridFunction associated with the
ParFiniteElementSpace @a *f. */
ParFiniteElementSpace @a *pf. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
void Update();
@@ -401,8 +416,8 @@ public:
convention = ComplexOperator::HERMITIAN);
/** @brief Create a ParComplexLinearForm on the ParFiniteElementSpace @a pf,
using the same integrators as the LinearForms @a plfr (real) and @a plfi
(imag) .
using the same integrators as the LinearForms @a plf_r (real) and
@a plf_i (imag).
The pointer @a fes is not owned by the newly constructed object.
@@ -478,7 +493,7 @@ public:
/** Class for a parallel sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but but the
fields. Sesquilinear forms are linear in the second argument but the
first argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
@@ -524,6 +539,21 @@ public:
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::FULL (default)
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
pblfr->SetAssemblyLevel(assembly_level);
pblfi->SetAssemblyLevel(assembly_level);
}
ParBilinearForm & real() { return *pblfr; }
ParBilinearForm & imag() { return *pblfi; }
const ParBilinearForm & real() const { return *pblfr; }
+40 -8
View File
@@ -415,9 +415,6 @@ void VisItDataCollection::SetMesh(MPI_Comm comm, Mesh *new_mesh)
void VisItDataCollection::RegisterField(const std::string& name,
GridFunction *gf)
{
DataCollection::RegisterField(name, gf);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim());
int LOD = 1;
if (gf->FESpace()->GetNURBSext())
{
@@ -431,6 +428,27 @@ void VisItDataCollection::RegisterField(const std::string& name,
}
}
DataCollection::RegisterField(name, gf);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD);
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
void VisItDataCollection::RegisterQField(const std::string& name,
QuadratureFunction *qf)
{
int LOD = -1;
Mesh *mesh = qf->GetSpace()->GetMesh();
for (int e=0; e<qf->GetSpace()->GetNE(); e++)
{
int locLOD = GlobGeometryRefiner.GetRefinementLevelFromElems(
mesh->GetElementBaseGeometry(e),
qf->GetElementIntRule(e).GetNPoints());
LOD = std::max(LOD,locLOD);
}
DataCollection::RegisterQField(name, qf);
field_info_map[name] = VisItFieldInfo("elements", 1, LOD);
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
@@ -598,14 +616,28 @@ void VisItDataCollection::LoadFields()
// TODO: 1) load parallel GridFunction on one processor
if (serial)
{
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
if ((it->second).association == "nodes")
{
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
}
else if ((it->second).association == "elements")
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
}
else
{
#ifdef MFEM_USE_MPI
field_map.Register(
it->first,
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
if ((it->second).association == "nodes")
{
field_map.Register(
it->first,
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
}
else if ((it->second).association == "elements")
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
#else
error = READ_ERROR;
MFEM_WARNING("Reading parallel format in serial is not supported");
@@ -640,7 +672,7 @@ std::string VisItDataCollection::GetVisItRootString()
{
ftags["assoc"] = picojson::value((it->second).association);
ftags["comps"] = picojson::value(to_string((it->second).num_components));
ftags["lod"] = picojson::value(to_string(visit_levels_of_detail));
ftags["lod"] = picojson::value(to_string((it->second).lod));
field["path"] = picojson::value(path_str + it->first + file_ext_format);
field["tags"] = picojson::value(ftags);
fields[it->first] = picojson::value(field);
+10 -3
View File
@@ -391,9 +391,10 @@ class VisItFieldInfo
public:
std::string association;
int num_components;
VisItFieldInfo() { association = ""; num_components = 0; }
VisItFieldInfo(std::string _association, int _num_components)
{ association = _association; num_components = _num_components; }
int lod;
VisItFieldInfo() { association = ""; num_components = 0; lod = 1;}
VisItFieldInfo(std::string _association, int _num_components, int _lod = 1)
{ association = _association; num_components = _num_components; lod =_lod;}
};
/// Data collection with VisIt I/O routines
@@ -445,6 +446,12 @@ public:
/// Add a grid function to the collection and update the root file
virtual void RegisterField(const std::string& field_name, GridFunction *gf);
/// Add a quadrature function to the collection and update the root file.
/** Visualization of quadrature function is not supported in VisIt(3.12).
A patch has been sent to VisIt developers in June 2020. */
virtual void RegisterQField(const std::string& q_field_name,
QuadratureFunction *qf);
/// Set VisIt parameter: default levels of detail for the MultiresControl
void SetLevelsOfDetail(int levels_of_detail);
+4 -1
View File
@@ -37,7 +37,8 @@ public:
ClosedUniform = 4, ///< Nodes: x_i = i/(n-1), i=0,...,n-1
OpenHalfUniform = 5, ///< Nodes: x_i = (i+1/2)/n, i=0,...,n-1
Serendipity = 6, ///< Serendipity basis (squares / cubes)
NumBasisTypes = 7 /**< Keep track of maximum types to prevent
ClosedGL = 7, ///< Closed GaussLegendre
NumBasisTypes = 8 /**< Keep track of maximum types to prevent
hard-coding */
};
/** @brief If the input does not represents a valid BasisType, abort with an
@@ -69,6 +70,7 @@ public:
case ClosedUniform: return Quadrature1D::ClosedUniform;
case OpenHalfUniform: return Quadrature1D::OpenHalfUniform;
case Serendipity: return Quadrature1D::GaussLobatto;
case ClosedGL: return Quadrature1D::ClosedGL;
}
return Quadrature1D::Invalid;
}
@@ -82,6 +84,7 @@ public:
case Quadrature1D::OpenUniform: return OpenUniform;
case Quadrature1D::ClosedUniform: return ClosedUniform;
case Quadrature1D::OpenHalfUniform: return OpenHalfUniform;
case Quadrature1D::ClosedGL: return ClosedGL;
}
return Invalid;
}
+6
View File
@@ -756,6 +756,12 @@ public:
/// Return the total number of quadrature points.
int GetSize() const { return size; }
/// Returns the mesh
inline Mesh *GetMesh() const { return mesh; }
/// Returns number of elements in the mesh.
inline int GetNE() const { return mesh->GetNE(); }
/// Get the IntegrationRule associated with mesh element @a idx.
const IntegrationRule &GetElementIntRule(int idx) const
{ return *int_rule[mesh->GetElementBaseGeometry(idx)]; }
+141 -33
View File
@@ -1344,15 +1344,14 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, Times-1);
if (ir == NULL)
if (ir) { return ir; }
ir = new IntegrationRule(Times-1);
for (int i = 1; i < Times; i++)
{
ir = new IntegrationRule(Times-1);
for (int i = 1; i < Times; i++)
{
IntegrationPoint &ip = ir->IntPoint(i-1);
ip.x = double(i) / Times;
ip.y = ip.z = 0.0;
}
IntegrationPoint &ip = ir->IntPoint(i-1);
ip.x = double(i) / Times;
ip.y = ip.z = 0.0;
}
}
break;
@@ -1364,18 +1363,17 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, ((Times-1)*(Times-2))/2);
if (ir == NULL)
{
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
for (int k = 0, j = 1; j < Times-1; j++)
for (int i = 1; i < Times-j; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
if (ir) { return ir; }
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
for (int k = 0, j = 1; j < Times-1; j++)
for (int i = 1; i < Times-j; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
break;
@@ -1386,18 +1384,17 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, (Times-1)*(Times-1));
if (ir == NULL)
{
ir = new IntegrationRule((Times-1)*(Times-1));
for (int k = 0, j = 1; j < Times; j++)
for (int i = 1; i < Times; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
if (ir) { return ir; }
ir = new IntegrationRule((Times-1)*(Times-1));
for (int k = 0, j = 1; j < Times; j++)
for (int i = 1; i < Times; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
break;
@@ -1405,10 +1402,121 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
mfem_error("GeometryRefiner::RefineInterior(...)");
}
if (ir) { IntPts[Geom].Append(ir); }
MFEM_ASSERT(ir != NULL, "Failed to construct the refined IntegrationRule.");
IntPts[Geom].Append(ir);
return ir;
}
int GeometryRefiner::GetRefinementLevelFromPoints(Geometry::Type geom, int Npts)
{
switch (geom)
{
case Geometry::POINT:
{
return -1;
}
case Geometry::SEGMENT:
{
return Npts -1;
}
case Geometry::TRIANGLE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+2)/2;
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::SQUARE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1);
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::CUBE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1)*(n+1);
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::TETRAHEDRON:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+3)*(n+2)*(n+1)/6;
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::PRISM:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1)*(n+2)/2;
if (np == Npts) { return n; }
}
return -1;
}
default:
{
mfem_error("Non existing Geometry.");
}
}
return -1;
}
int GeometryRefiner::GetRefinementLevelFromElems(Geometry::Type geom, int Nels)
{
switch (geom)
{
case Geometry::POINT:
{
return -1;
}
case Geometry::SEGMENT:
{
return Nels;
}
case Geometry::TRIANGLE:
case Geometry::SQUARE:
{
for (int n = 0; (n < 15) && (n*n < Nels+1) ; n++)
{
if (n*n == Nels) { return n-1; }
}
return -1;
}
case Geometry::CUBE:
case Geometry::TETRAHEDRON:
case Geometry::PRISM:
{
for (int n = 0; (n < 15) && (n*n*n < Nels+1) ; n++)
{
if (n*n*n == Nels) { return n-1; }
}
return -1;
}
default:
{
mfem_error("Non existing Geometry.");
}
}
return -1;
}
GeometryRefiner GlobGeometryRefiner;
}
+6
View File
@@ -273,6 +273,12 @@ public:
/// @note This method always uses Quadrature1D::OpenUniform points.
const IntegrationRule *RefineInterior(Geometry::Type Geom, int Times);
/// Get the Refinement level based on number of points
virtual int GetRefinementLevelFromPoints(Geometry::Type Geom, int Npts);
/// Get the Refinement level based on number of elements
virtual int GetRefinementLevelFromElems(Geometry::Type geom, int Npts);
~GeometryRefiner();
};
+3 -4
View File
@@ -199,8 +199,7 @@ void GridFunction::MakeRef(FiniteElementSpace *f, Vector &v, int v_offset)
if (f != fes) { Destroy(); }
fes = f;
v.UseDevice(true);
NewMemoryAndSize(Memory<double>(v.GetMemory(), v_offset, fes->GetVSize()),
fes->GetVSize(), true);
this->Vector::MakeRef(v, v_offset, fes->GetVSize());
sequence = fes->GetSequence();
}
@@ -2984,7 +2983,7 @@ double GridFunction::ComputeLpError(const double p, Coefficient &exsol,
}
else
{
int intorder = 2*fe->GetOrder() + 1; // <----------
int intorder = 2*fe->GetOrder() + 3; // <----------
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
GetValues(i, *ir, vals);
@@ -3117,7 +3116,7 @@ double GridFunction::ComputeLpError(const double p, VectorCoefficient &exsol,
}
else
{
int intorder = 2*fe->GetOrder() + 1; // <----------
int intorder = 2*fe->GetOrder() + 3; // <----------
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
T = fes->GetElementTransformation(i);
+2 -2
View File
@@ -105,7 +105,7 @@ public:
have the same size.
@note Defining this method overwrites the implicitly defined copy
assignemnt operator. */
assignment operator. */
GridFunction &operator=(const GridFunction &rhs)
{ return operator=((const Vector &)rhs); }
@@ -714,7 +714,7 @@ public:
the same size.
@note Defining this method overwrites the implicitly defined copy
assignemnt operator. */
assignment operator. */
QuadratureFunction &operator=(const QuadratureFunction &v);
/// Get the IntegrationRule associated with mesh element @a idx.
+25
View File
@@ -618,6 +618,26 @@ void QuadratureFunctions1D::OpenHalfUniform(const int np, IntegrationRule* ir)
CalculateUniformWeights(ir, Quadrature1D::OpenHalfUniform);
}
void QuadratureFunctions1D::ClosedGL(const int np, IntegrationRule* ir)
{
ir->SetSize(np);
ir->IntPoint(0).x = 0.0;
ir->IntPoint(np-1).x = 1.0;
if ( np > 2 )
{
IntegrationRule gl_ir;
GaussLegendre(np-1, &gl_ir);
for (int i = 1; i < np-1; ++i)
{
ir->IntPoint(i).x = (gl_ir.IntPoint(i-1).x + gl_ir.IntPoint(i).x)/2;
}
}
CalculateUniformWeights(ir, Quadrature1D::ClosedGL);
}
void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
const int type)
{
@@ -650,6 +670,11 @@ void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
OpenHalfUniform(np, &ir);
break;
}
case Quadrature1D::ClosedGL:
{
ClosedGL(np, &ir);
break;
}
default:
{
MFEM_ABORT("Asking for an unknown type of 1D Quadrature points, "
+3 -1
View File
@@ -272,6 +272,7 @@ public:
void OpenUniform(const int np, IntegrationRule *ir);
void ClosedUniform(const int np, IntegrationRule *ir);
void OpenHalfUniform(const int np, IntegrationRule *ir);
void ClosedGL(const int np, IntegrationRule *ir);
///@}
/// A helper function that will play nice with Poly_1D::OpenPoints and
@@ -293,7 +294,8 @@ public:
GaussLobatto = 1,
OpenUniform = 2, ///< aka open Newton-Cotes
ClosedUniform = 3, ///< aka closed Newton-Cotes
OpenHalfUniform = 4 ///< aka "open half" Newton-Cotes
OpenHalfUniform = 4, ///< aka "open half" Newton-Cotes
ClosedGL = 5 ///< aka closed Gauss Legendre
};
/** @brief If the Quadrature1D type is not closed return Invalid; otherwise
return type. */
+10 -1
View File
@@ -199,10 +199,19 @@ void LinearForm::Assemble()
void LinearForm::Update(FiniteElementSpace *f, Vector &v, int v_offset)
{
fes = f;
NewDataAndSize((double *)v + v_offset, fes->GetVSize());
NewMemoryAndSize(Memory<double>(v.GetMemory(), v_offset, f->GetVSize()),
f->GetVSize(), false);
ResetDeltaLocations();
}
void LinearForm::MakeRef(FiniteElementSpace *f, Vector &v, int v_offset)
{
MFEM_ASSERT(v.Size() >= v_offset + f->GetVSize(), "");
fes = f;
v.UseDevice(true);
this->Vector::MakeRef(v, v_offset, fes->GetVSize());
}
void LinearForm::AssembleDelta()
{
if (dlfi_delta.Size() == 0) { return; }
+10 -1
View File
@@ -26,7 +26,7 @@ protected:
/// FE space on which the LinearForm lives. Not owned.
FiniteElementSpace *fes;
/** @brief Indicates the LinerFormIntegrator%s stored in #dlfi, #dlfi_delta,
/** @brief Indicates the LinearFormIntegrator%s stored in #dlfi, #dlfi_delta,
#blfi, and #flfi are owned by another LinearForm. */
int extern_lfs;
@@ -175,6 +175,15 @@ public:
@note This method does not perform assembly. */
void Update(FiniteElementSpace *f, Vector &v, int v_offset);
/** @brief Make the LinearForm reference external data on a new
FiniteElementSpace. */
/** This method changes the FiniteElementSpace associated with the LinearForm
@a *f and sets the data of the Vector @a v (plus the @a v_offset)
as external data in the LinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
virtual void MakeRef(FiniteElementSpace *f, Vector &v, int v_offset);
/// Return the action of the LinearForm as a linear mapping.
/** Linear forms are linear functionals which map GridFunctions to
the real numbers. This method performs this mapping which in
+11 -4
View File
@@ -135,11 +135,18 @@ void Multigrid::SetOperator(const Operator& op)
MFEM_ABORT("SetOperator not supported in Multigrid");
}
void Multigrid::SmoothingStep(int level) const
void Multigrid::SmoothingStep(int level, bool transpose) const
{
GetOperatorAtLevel(level)->Mult(*Y[level], *R[level]); // r = A x
subtract(*X[level], *R[level], *R[level]); // r = b - A x
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
if (transpose)
{
GetSmootherAtLevel(level)->MultTranspose(*R[level], *Z[level]); // z = S r
}
else
{
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
}
add(*Y[level], 1.0, *Z[level], *Y[level]); // x = x + S (b - A x)
}
@@ -153,7 +160,7 @@ void Multigrid::Cycle(int level) const
for (int i = 0; i < preSmoothingSteps; i++)
{
SmoothingStep(level);
SmoothingStep(level, false);
}
// Compute residual
@@ -187,7 +194,7 @@ void Multigrid::Cycle(int level) const
// Post-smooth
for (int i = 0; i < postSmoothingSteps; i++)
{
SmoothingStep(level);
SmoothingStep(level, true);
}
}
+1 -1
View File
@@ -108,7 +108,7 @@ public:
private:
/// Application of a smoothing step at particular level
void SmoothingStep(int level) const;
void SmoothingStep(int level, bool transpose) const;
/// Application of a cycle at particular level
void Cycle(int level) const;
+11 -11
View File
@@ -933,6 +933,17 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
}
}
if (!Grads(0,0)->Finalized())
{
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
{
Grads(i,j)->Finalize(skip_zeros);
}
}
}
for (int s=0; s<fes.Size(); ++s)
{
for (int i = 0; i < ess_vdofs[s]->Size(); ++i)
@@ -952,17 +963,6 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
}
}
if (!Grads(0,0)->Finalized())
{
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
{
Grads(i,j)->Finalize(skip_zeros);
}
}
}
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
+1 -1
View File
@@ -241,7 +241,7 @@ void ParBilinearForm::Assemble(int skip_zeros)
BilinearForm::Assemble(skip_zeros);
if (fbfi.Size() > 0)
if (!ext && fbfi.Size() > 0)
{
AssembleSharedFaces(skip_zeros);
}
+4 -2
View File
@@ -711,7 +711,9 @@ void ParGridFunction::SaveAsOne(std::ostream &out)
int *nfdofs = new int[NRanks];
int *nrdofs = new int[NRanks];
values[0] = data;
double * h_data = const_cast<double *>(this->HostRead());
values[0] = h_data;
nv[0] = pfes -> GetVSize();
nvdofs[0] = pfes -> GetNVDofs();
nedofs[0] = pfes -> GetNEDofs();
@@ -812,7 +814,7 @@ void ParGridFunction::SaveAsOne(std::ostream &out)
MPI_Send(&nvdofs[0], 1, MPI_INT, 0, 456, MyComm);
MPI_Send(&nedofs[0], 1, MPI_INT, 0, 457, MyComm);
MPI_Send(&nfdofs[0], 1, MPI_INT, 0, 458, MyComm);
MPI_Send(data, nv[0], MPI_DOUBLE, 0, 460, MyComm);
MPI_Send(h_data, nv[0], MPI_DOUBLE, 0, 460, MyComm);
}
delete [] values;
+12 -1
View File
@@ -21,7 +21,6 @@ namespace mfem
void ParLinearForm::Update(ParFiniteElementSpace *pf)
{
if (pf) { pfes = pf; }
LinearForm::Update(pfes);
}
@@ -31,6 +30,18 @@ void ParLinearForm::Update(ParFiniteElementSpace *pf, Vector &v, int v_offset)
LinearForm::Update(pf,v,v_offset);
}
void ParLinearForm::MakeRef(FiniteElementSpace *f, Vector &v, int v_offset)
{
LinearForm::MakeRef(f, v, v_offset);
pfes = dynamic_cast<ParFiniteElementSpace*>(f);
}
void ParLinearForm::MakeRef(ParFiniteElementSpace *pf, Vector &v, int v_offset)
{
LinearForm::MakeRef(pf, v, v_offset);
pfes = pf;
}
void ParLinearForm::ParallelAssemble(Vector &tv)
{
const Operator* prolong = pfes->GetProlongationMatrix();
+19
View File
@@ -92,6 +92,25 @@ public:
@note This method does not perform assembly. */
void Update(ParFiniteElementSpace *pf, Vector &v, int v_offset);
/** @brief Make the ParLinearForm reference external data on a new
FiniteElementSpace. */
/** This method changes the FiniteElementSpace associated with the ParLinearForm
to @a *f and sets the data of the Vector @a v (plus the @a v_offset) as external
data in the ParLinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
virtual void MakeRef(FiniteElementSpace *f, Vector &v, int v_offset);
/** @brief Make the ParLinearForm reference external data on a new
ParFiniteElementSpace. */
/** This method changes the ParFiniteElementSpace associated with the ParLinearForm
to @a *pf and sets the data of the Vector @a v (plus the @a v_offset) as external
data in the ParLinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
void MakeRef(ParFiniteElementSpace *pf, Vector &v, int v_offset);
/// Assemble the vector on the true dofs, i.e. P^t v.
void ParallelAssemble(Vector &tv);
+136 -57
View File
@@ -27,35 +27,24 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
ElementDofOrdering e_ordering,
FaceType type,
L2FaceValues m)
: fes(fes),
nf(fes.GetNFbyType(type)),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndofs(fes.GetNDofs()),
dof(nf>0 ?
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
: 0),
m(m),
nfdofs(nf*dof),
scatter_indices1(nf*dof),
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
offsets(ndofs+1),
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
: L2FaceRestriction(fes, type, m)
{
if (nf==0) { return; }
// If fespace == L2
const FiniteElement *fe = fes.GetFE(0);
const ParFiniteElementSpace &pfes =
static_cast<const ParFiniteElementSpace&>(this->fes);
const FiniteElement *fe = pfes.GetFE(0);
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement*>(fe);
MFEM_VERIFY(tfe != NULL &&
(tfe->GetBasisType()==BasisType::GaussLobatto ||
tfe->GetBasisType()==BasisType::Positive),
"Only Gauss-Lobatto and Bernstein basis are supported in "
"ParL2FaceRestriction.");
MFEM_VERIFY(fes.GetMesh()->Conforming(),
MFEM_VERIFY(pfes.GetMesh()->Conforming(),
"Non-conforming meshes not yet supported with partial assembly.");
// Assuming all finite elements are using Gauss-Lobatto dofs
height = (m==L2FaceValues::DoubleValued? 2 : 1)*vdim*nf*dof;
width = fes.GetVSize();
width = pfes.GetVSize();
const bool dof_reorder = (e_ordering == ElementDofOrdering::LEXICOGRAPHIC);
if (!dof_reorder)
{
@@ -63,32 +52,32 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
}
if (dof_reorder && nf > 0)
{
for (int f = 0; f < fes.GetNF(); ++f)
for (int f = 0; f < pfes.GetNF(); ++f)
{
const FiniteElement *fe =
fes.GetTraceElement(f, fes.GetMesh()->GetFaceBaseGeometry(f));
pfes.GetTraceElement(f, pfes.GetMesh()->GetFaceBaseGeometry(f));
const TensorBasisElement* el =
dynamic_cast<const TensorBasisElement*>(fe);
if (el) { continue; }
mfem_error("Finite element not suitable for lexicographic ordering");
}
}
const Table& e2dTable = fes.GetElementToDofTable();
const Table& e2dTable = pfes.GetElementToDofTable();
const int* elementMap = e2dTable.GetJ();
Array<int> faceMap1(dof), faceMap2(dof);
int e1, e2;
int inf1, inf2;
int face_id1, face_id2;
int orientation;
const int dof1d = fes.GetFE(0)->GetOrder()+1;
const int elem_dofs = fes.GetFE(0)->GetDof();
const int dim = fes.GetMesh()->SpaceDimension();
const int dof1d = pfes.GetFE(0)->GetOrder()+1;
const int elem_dofs = pfes.GetFE(0)->GetDof();
const int dim = pfes.GetMesh()->SpaceDimension();
// Computation of scatter indices
int f_ind=0;
for (int f = 0; f < fes.GetNF(); ++f)
for (int f = 0; f < pfes.GetNF(); ++f)
{
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if (dof_reorder)
{
orientation = inf1 % 64;
@@ -136,7 +125,7 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
{
const int se2 = -1 - e2;
Array<int> sharedDofs;
fes.GetFaceNbrElementVDofs(se2, sharedDofs);
pfes.GetFaceNbrElementVDofs(se2, sharedDofs);
for (int d = 0; d < dof; ++d)
{
const int pd = PermuteFaceL2(dim, face_id1, face_id2,
@@ -180,10 +169,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
offsets[i] = 0;
}
f_ind = 0;
for (int f = 0; f < fes.GetNF(); ++f)
for (int f = 0; f < pfes.GetNF(); ++f)
{
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
(type==FaceType::Boundary && e2<0 && inf2<0) )
{
@@ -222,10 +211,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
offsets[i] += offsets[i - 1];
}
f_ind = 0;
for (int f = 0; f < fes.GetNF(); ++f)
for (int f = 0; f < pfes.GetNF(); ++f)
{
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
(type==FaceType::Boundary && e2<0 && inf2<0) )
{
@@ -272,8 +261,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
{
const ParFiniteElementSpace &pfes =
static_cast<const ParFiniteElementSpace&>(this->fes);
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&fes),
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&pfes),
const_cast<Vector&>(x), 0);
x_gf.ExchangeFaceNbrData();
@@ -337,34 +328,122 @@ void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
}
}
void ParL2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
{
// Assumes all elements have the same number of dofs
const int nd = dof;
const int vd = vdim;
const bool t = byvdim;
const int dofs = nfdofs;
auto d_offsets = offsets.Read();
auto d_indices = gather_indices.Read();
auto d_x = Reshape(x.Read(), nd, vd, 2, nf);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
int val = AtomicAdd(I[iE],dofs);
return val;
}
void ParL2FaceRestriction::FillI(SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
const int Ndofs = ndofs;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
auto I_face = face_mat.ReadWriteI();
MFEM_FORALL(i, ne*elemDofs*vdim+1,
{
const int offset = d_offsets[i];
const int nextOffset = d_offsets[i + 1];
for (int c = 0; c < vd; ++c)
I_face[i] = 0;
});
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
if (iE1 < Ndofs)
{
double dofValue = 0;
for (int j = offset; j < nextOffset; ++j)
for (int jF = 0; jF < face_dofs; jF++)
{
int idx_j = d_indices[j];
bool isE1 = idx_j < dofs;
idx_j = isE1 ? idx_j : idx_j - dofs;
dofValue += isE1 ?
d_x(idx_j % nd, c, 0, idx_j / nd)
:d_x(idx_j % nd, c, 1, idx_j / nd);
const int jE2 = d_indices2[f*face_dofs+jF];
if (jE2 < Ndofs)
{
AddNnz(iE1,I,1);
}
else
{
AddNnz(iE1,I_face,1);
}
}
}
const int iE2 = d_indices2[f*face_dofs+iF];
if (iE2 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
if (jE1 < Ndofs)
{
AddNnz(iE2,I,1);
}
else
{
AddNnz(iE2,I_face,1);
}
}
}
});
}
void ParL2FaceRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
const int Ndofs = ndofs;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
auto I = mat.ReadWriteI();
auto I_face = face_mat.ReadWriteI();
auto J = mat.WriteJ();
auto J_face = face_mat.WriteJ();
auto Data = mat.WriteData();
auto Data_face = face_mat.WriteData();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
if (iE1 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE2 = d_indices2[f*face_dofs+jF];
if (jE2 < Ndofs)
{
const int offset = AddNnz(iE1,I,1);
J[offset] = jE2;
Data[offset] = mat_fea(jF,iF,1,f);
}
else
{
const int offset = AddNnz(iE1,I_face,1);
J_face[offset] = jE2-Ndofs;
Data_face[offset] = mat_fea(jF,iF,1,f);
}
}
}
const int iE2 = d_indices2[f*face_dofs+iF];
if (iE2 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
if (jE1 < Ndofs)
{
const int offset = AddNnz(iE2,I,1);
J[offset] = jE1;
Data[offset] = mat_fea(jF,iF,0,f);
}
else
{
const int offset = AddNnz(iE2,I_face,1);
J_face[offset] = jE1-Ndofs;
Data_face[offset] = mat_fea(jF,iF,0,f);
}
}
d_y(t?c:i,t?i:c) += dofValue;
}
});
}
+9 -16
View File
@@ -26,28 +26,21 @@ class ParFiniteElementSpace;
/// Operator that extracts Face degrees of freedom in parallel.
/** Objects of this type are typically created and owned by FiniteElementSpace
objects, see FiniteElementSpace::GetFaceRestriction(). */
class ParL2FaceRestriction : public Operator
class ParL2FaceRestriction : public L2FaceRestriction
{
protected:
const ParFiniteElementSpace &fes;
const int nf;
const int vdim;
const bool byvdim;
const int ndofs;
const int dof;
const L2FaceValues m;
const int nfdofs;
Array<int> scatter_indices1;
Array<int> scatter_indices2;
Array<int> offsets;
Array<int> gather_indices;
public:
ParL2FaceRestriction(const ParFiniteElementSpace&, ElementDofOrdering,
FaceType type,
L2FaceValues m = L2FaceValues::DoubleValued);
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ParL2FaceRestriction. */
void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this ParL2FaceRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const;
};
}
+421 -52
View File
@@ -17,55 +17,6 @@
namespace mfem
{
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
: ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
ndofs(fes.GetNDofs())
{
height = vdim*ne*ndof;
width = vdim*ne*ndof;
}
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
auto d_y = Reshape(y.Write(), nd, vd, ne);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
}
});
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), nd, vd, ne);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
}
});
}
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
ElementDofOrdering e_ordering)
: fes(f),
@@ -281,7 +232,309 @@ void ElementRestriction::BooleanMask(Vector& y) const
}
}
/// Return the face degrees of freedom returned in Lexicographic order.
void ElementRestriction::FillSparseMatrix(const Vector &mat_ea,
SparseMatrix &mat) const
{
mat.GetMemoryI().New(mat.Height()+1, mat.GetMemoryI().GetMemoryType());
const int nnz = FillI(mat);
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
FillJAndData(mat_ea, mat);
}
template <int MaxNbNbr>
static MFEM_HOST_DEVICE int GetMinElt(const int *my_elts, const int nbElts,
const int *nbr_elts, const int nbrNbElts)
{
// Building the intersection
int inter[MaxNbNbr];
int cpt = 0;
for (int i = 0; i < nbElts; i++)
{
const int e_i = my_elts[i];
for (int j = 0; j < nbrNbElts; j++)
{
if (e_i==nbr_elts[j])
{
inter[cpt] = e_i;
cpt++;
}
}
}
// Finding the minimum
int min = inter[0];
for (int i = 1; i < cpt; i++)
{
if (inter[i] < min)
{
min = inter[i];
}
}
return min;
}
/** Returns the index where a non-zero entry should be added and increment the
number of non-zeros for the row i_L. */
static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
{
int ind = AtomicAdd(I[i_L],1);
return ind;
}
int ElementRestriction::FillI(SparseMatrix &mat) const
{
static constexpr int Max = MaxNbNbr;
const int all_dofs = ndofs;
const int vd = vdim;
const int elt_dofs = dof;
auto I = mat.ReadWriteI();
auto d_offsets = offsets.Read();
auto d_indices = indices.Read();
auto d_gatherMap = gatherMap.Read();
MFEM_FORALL(i_L, vd*all_dofs+1,
{
I[i_L] = 0;
});
MFEM_FORALL(e, ne,
{
for (int i = 0; i < elt_dofs; i++)
{
int i_elts[Max];
const int i_E = e*elt_dofs + i;
const int i_L = d_gatherMap[i_E];
const int i_offset = d_offsets[i_L];
const int i_nextOffset = d_offsets[i_L+1];
const int i_nbElts = i_nextOffset - i_offset;
for (int e_i = 0; e_i < i_nbElts; ++e_i)
{
const int i_E = d_indices[i_offset+e_i];
i_elts[e_i] = i_E/elt_dofs;
}
for (int j = 0; j < elt_dofs; j++)
{
const int j_E = e*elt_dofs + j;
const int j_L = d_gatherMap[j_E];
const int j_offset = d_offsets[j_L];
const int j_nextOffset = d_offsets[j_L+1];
const int j_nbElts = j_nextOffset - j_offset;
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
{
GetAndIncrementNnzIndex(i_L, I);
}
else // assembly required
{
int j_elts[Max];
for (int e_j = 0; e_j < j_nbElts; ++e_j)
{
const int j_E = d_indices[j_offset+e_j];
const int elt = j_E/elt_dofs;
j_elts[e_j] = elt;
}
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
if (e == min_e) // add the nnz only once
{
GetAndIncrementNnzIndex(i_L, I);
}
}
}
}
});
// We need to sum the entries of I, we do it on CPU as it is very sequential.
auto h_I = mat.HostReadWriteI();
const int nTdofs = vd*all_dofs;
int sum = 0;
for (int i = 0; i < nTdofs; i++)
{
const int nnz = h_I[i];
h_I[i] = sum;
sum+=nnz;
}
h_I[nTdofs] = sum;
// We return the number of nnz
return h_I[nTdofs];
}
void ElementRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat) const
{
static constexpr int Max = MaxNbNbr;
const int all_dofs = ndofs;
const int vd = vdim;
const int elt_dofs = dof;
auto I = mat.ReadWriteI();
auto J = mat.WriteJ();
auto Data = mat.WriteData();
auto d_offsets = offsets.Read();
auto d_indices = indices.Read();
auto d_gatherMap = gatherMap.Read();
auto mat_ea = Reshape(ea_data.Read(), elt_dofs, elt_dofs, ne);
MFEM_FORALL(e, ne,
{
for (int i = 0; i < elt_dofs; i++)
{
int i_elts[Max];
int i_B[Max];
const int i_E = e*elt_dofs + i;
const int i_L = d_gatherMap[i_E];
const int i_offset = d_offsets[i_L];
const int i_nextOffset = d_offsets[i_L+1];
const int i_nbElts = i_nextOffset - i_offset;
for (int e_i = 0; e_i < i_nbElts; ++e_i)
{
const int i_E = d_indices[i_offset+e_i];
i_elts[e_i] = i_E/elt_dofs;
i_B[e_i] = i_E%elt_dofs;
}
for (int j = 0; j < elt_dofs; j++)
{
const int j_E = e*elt_dofs + j;
const int j_L = d_gatherMap[j_E];
const int j_offset = d_offsets[j_L];
const int j_nextOffset = d_offsets[j_L+1];
const int j_nbElts = j_nextOffset - j_offset;
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
{
const int nnz = GetAndIncrementNnzIndex(i_L, I);
J[nnz] = j_L;
Data[nnz] = mat_ea(j,i,e);
}
else // assembly required
{
int j_elts[Max];
int j_B[Max];
for (int e_j = 0; e_j < j_nbElts; ++e_j)
{
const int j_E = d_indices[j_offset+e_j];
const int elt = j_E/elt_dofs;
j_elts[e_j] = elt;
j_B[e_j] = j_E%elt_dofs;
}
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
if (e == min_e) // add the nnz only once
{
double val = 0.0;
for (int i = 0; i < i_nbElts; i++)
{
const int e_i = i_elts[i];
const int i_Bloc = i_B[i];
for (int j = 0; j < j_nbElts; j++)
{
const int e_j = j_elts[j];
const int j_Bloc = j_B[j];
if (e_i == e_j)
{
val += mat_ea(j_Bloc, i_Bloc, e_i);
}
}
}
const int nnz = GetAndIncrementNnzIndex(i_L, I);
J[nnz] = j_L;
Data[nnz] = val;
}
}
}
}
});
// We need to shift again the entries of I, we do it on CPU as it is very
// sequential.
auto h_I = mat.HostReadWriteI();
const int size = vd*all_dofs;
for (int i = 0; i < size; i++)
{
h_I[size-i] = h_I[size-(i+1)];
}
h_I[0] = 0;
}
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
: ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
ndofs(fes.GetNDofs())
{
height = vdim*ne*ndof;
width = vdim*ne*ndof;
}
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
auto d_y = Reshape(y.Write(), nd, vd, ne);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
}
});
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), nd, vd, ne);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
}
});
}
void L2ElementRestriction::FillI(SparseMatrix &mat) const
{
const int elem_dofs = ndof;
const int vd = vdim;
auto I = mat.WriteI();
MFEM_FORALL(dof, ne*elem_dofs*vd,
{
I[dof] = elem_dofs;
});
}
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
{
int val = AtomicAdd(I[iE],dofs);
return val;
}
void L2ElementRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat) const
{
const int elem_dofs = ndof;
const int vd = vdim;
auto I = mat.ReadWriteI();
auto J = mat.WriteJ();
auto Data = mat.WriteData();
auto mat_ea = Reshape(ea_data.Read(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(iE, ne*elem_dofs*vd,
{
const int offset = AddNnz(iE,I,elem_dofs);
const int e = iE/elem_dofs;
const int i = iE%elem_dofs;
for (int j = 0; j < elem_dofs; j++)
{
J[offset+j] = e*elem_dofs+j;
Data[offset+j] = mat_ea(j,i,e);
}
});
}
// Return the face degrees of freedom returned in Lexicographic order.
void GetFaceDofs(const int dim, const int face_id,
const int dof1d, Array<int> &faceMap)
{
@@ -700,7 +953,7 @@ static int PermuteFace3D(const int face_id1, const int face_id2,
return ToLexOrdering3D(face_id2, size1d, new_i, new_j);
}
/// Permute dofs or quads on a face for e2 to match with the ordering of e1
// Permute dofs or quads on a face for e2 to match with the ordering of e1
int PermuteFaceL2(const int dim, const int face_id1,
const int face_id2, const int orientation,
const int size1d, const int index)
@@ -720,23 +973,32 @@ int PermuteFaceL2(const int dim, const int face_id1,
}
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
const ElementDofOrdering e_ordering,
const FaceType type,
const L2FaceValues m)
: fes(fes),
nf(fes.GetNFbyType(type)),
ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndofs(fes.GetNDofs()),
dof(nf > 0 ?
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
: 0),
elemDofs(fes.GetFE(0)->GetDof()),
m(m),
nfdofs(nf*dof),
scatter_indices1(nf*dof),
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
offsets(ndofs+1),
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
{
}
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
const ElementDofOrdering e_ordering,
const FaceType type,
const L2FaceValues m)
: L2FaceRestriction(fes, type, m)
{
// If fespace == L2
const FiniteElement *fe = fes.GetFE(0);
@@ -1034,6 +1296,113 @@ void L2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
}
}
void L2FaceRestriction::FillI(SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int iE1 = d_indices1[fdof];
const int iE2 = d_indices2[fdof];
AddNnz(iE1,I,face_dofs);
AddNnz(iE2,I,face_dofs);
});
}
void L2FaceRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
auto J = mat.WriteJ();
auto Data = mat.WriteData();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
const int iE2 = d_indices2[f*face_dofs+iF];
const int offset1 = AddNnz(iE1,I,face_dofs);
const int offset2 = AddNnz(iE2,I,face_dofs);
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
const int jE2 = d_indices2[f*face_dofs+jF];
J[offset2+jF] = jE1;
J[offset1+jF] = jE2;
Data[offset2+jF] = mat_fea(jF,iF,0,f);
Data[offset1+jF] = mat_fea(jF,iF,1,f);
}
});
}
void L2FaceRestriction::AddFaceMatricesToElementMatrices(Vector &fea_data,
Vector &ea_data) const
{
const int face_dofs = dof;
const int elem_dofs = elemDofs;
const int NE = ne;
if (m==L2FaceValues::DoubleValued)
{
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, 2, nf);
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(f, nf,
{
const int e1 = d_indices1[f*face_dofs]/elem_dofs;
const int e2 = d_indices2[f*face_dofs]/elem_dofs;
for (int j = 0; j < face_dofs; j++)
{
const int jB1 = d_indices1[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iB1 = d_indices1[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iB1,jB1,e1), mat_fea(i,j,0,f));
}
}
if (e2 < NE)
{
for (int j = 0; j < face_dofs; j++)
{
const int jB2 = d_indices2[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iB2 = d_indices2[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iB2,jB2,e2), mat_fea(i,j,1,f));
}
}
}
});
}
else
{
auto d_indices = scatter_indices1.Read();
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, nf);
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(f, nf,
{
const int e = d_indices[f*face_dofs]/elem_dofs;
for (int j = 0; j < face_dofs; j++)
{
const int jE = d_indices[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iE = d_indices[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iE,jE,e), mat_fea(i,j,f));
}
}
});
}
}
int ToLexOrdering(const int dim, const int face_id, const int size1d,
const int index)
{
+39 -1
View File
@@ -30,6 +30,11 @@ enum class L2FaceValues : bool {SingleValued, DoubleValued};
objects, see FiniteElementSpace::GetElementRestriction(). */
class ElementRestriction : public Operator
{
private:
/** This number defines the maximum number of elements any dof can belong to
for the FillSparseMatrix method. */
static const int MaxNbNbr = 16;
protected:
const FiniteElementSpace &fes;
const int ne;
@@ -59,6 +64,16 @@ public:
emulate SetSubVector and its transpose on GPUs. This method is running on
the host, since the `processed` array requires a large shared memory. */
void BooleanMask(Vector& y) const;
/// Fill a Sparse Matrix with Element Matrices.
void FillSparseMatrix(const Vector &mat_ea, SparseMatrix &mat) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ElementRestriction. */
int FillI(SparseMatrix &mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this ElementRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
};
/// Operator that converts L2 FiniteElementSpace L-vectors to E-vectors.
@@ -77,6 +92,12 @@ public:
L2ElementRestriction(const FiniteElementSpace&);
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ElementRestriction. */
void FillI(SparseMatrix &mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this L2FaceRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
};
/// Operator that extracts Face degrees of freedom.
@@ -111,10 +132,12 @@ class L2FaceRestriction : public Operator
protected:
const FiniteElementSpace &fes;
const int nf;
const int ne;
const int vdim;
const bool byvdim;
const int ndofs;
const int dof;
const int elemDofs;
const L2FaceValues m;
const int nfdofs;
Array<int> scatter_indices1;
@@ -122,12 +145,27 @@ protected:
Array<int> offsets;
Array<int> gather_indices;
L2FaceRestriction(const FiniteElementSpace&,
const FaceType,
const L2FaceValues m = L2FaceValues::DoubleValued);
public:
L2FaceRestriction(const FiniteElementSpace&, const ElementDofOrdering,
const FaceType,
const L2FaceValues m = L2FaceValues::DoubleValued);
void Mult(const Vector &x, Vector &y) const;
virtual void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this L2FaceRestriction. */
virtual void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this L2FaceRestriction, and the values of ea_data. */
virtual void FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const;
/// This methods adds the DG face matrices to the element matrices.
void AddFaceMatricesToElementMatrices(Vector &fea_data,
Vector &ea_data) const;
};
// Return the face degrees of freedom returned in Lexicographic order.
+493 -54
View File
@@ -927,9 +927,20 @@ void TargetConstructor::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void TargetConstructor::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
// TODO: Compute derivative for targets with GIVEN_SHAPE or/and GIVEN_SIZE
for (int i = 0; i < Tpr.GetFE()->GetDim()*ir.GetNPoints(); i++) { dJtr(i) = 0.; }
}
void AnalyticAdaptTC::SetAnalyticTargetSpec(Coefficient *sspec,
VectorCoefficient *vspec,
MatrixCoefficient *mspec)
TMOPMatrixCoefficient *mspec)
{
scalar_tspec = sspec;
vector_tspec = vspec;
@@ -970,6 +981,39 @@ void AnalyticAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void AnalyticAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
const FiniteElement *fe = Tpr.GetFE();
DenseMatrix point_mat;
point_mat.UseExternalData(elfun.GetData(), fe->GetDof(), fe->GetDim());
switch (target_type)
{
case GIVEN_FULL:
{
MFEM_VERIFY(matrix_tspec != NULL,
"Target type GIVEN_FULL requires a TMOPMatrixCoefficient.");
for (int d = 0; d < fe->GetDim(); d++)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
Tpr.SetIntPoint(&ip);
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
matrix_tspec->EvalGrad(dJtr_i, Tpr, ip, d);
}
}
break;
}
default:
MFEM_ABORT("Incompatible target type for analytic adaptation!");
}
}
#ifdef MFEM_USE_MPI
void DiscreteAdaptTC::FinalizeParDiscreteTargetSpec(const ParGridFunction
&tspec_)
@@ -1168,7 +1212,7 @@ void DiscreteAdaptTC::UpdateTargetSpecificationAtNode(const FiniteElement &el,
Array<int> dofs;
tspec_fes->GetElementDofs(T.ElementNo, dofs);
const int cnt = tspec.Size()/ncomp; //dofs per scalar-field
const int cnt = tspec.Size()/ncomp; // dofs per scalar-field
for (int i = 0; i < ncomp; i++)
{
@@ -1196,6 +1240,9 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
DenseTensor &Jtr) const
{
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
const int dim = fe.GetDim(),
nqp = ir.GetNPoints();
Jtrcomp.SetSize(dim, dim, 4*nqp);
switch (target_type)
{
@@ -1205,7 +1252,7 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
const DenseMatrix &Wideal =
Geometries.GetGeomToPerfGeomJac(fe.GetGeomType());
const int dim = Wideal.Height(),
ndofs = tspec_fes->GetFE(0)->GetDof(),
ndofs = tspec_fes->GetFE(e_id)->GetDof(),
ntspec_dofs = ndofs*ncomp;
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
@@ -1216,25 +1263,32 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
tspec_fesv->GetElementVDofs(e_id, dofs);
tspec.GetSubVector(dofs, tspec_vals);
for (int i = 0; i < ir.GetNPoints(); i++)
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
const IntegrationPoint &ip = ir.IntPoint(q);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
Jtr(i) = Wideal; //Initialize to identity
Jtr(q) = Wideal; // Initialize to identity
for (int d = 0; d < 4; d++)
{
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(d + 4*q), dim, dim);
Jtrcomp_q = Wideal; // Initialize to identity
}
if (sizeidx != -1) //Set size
if (sizeidx != -1) // Set size
{
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
const double min_size = par_vals.Min();
MFEM_VERIFY(min_size > 0.0,
"Non-positive size propagated in the target definition.");
const double size = std::max(shape * par_vals, min_size);
Jtr(i).Set(std::pow(size, 1.0/dim), Jtr(i));
} //Done size
Jtr(q).Set(std::pow(size, 1.0/dim), Jtr(q));
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(0 + 4*q), dim, dim);
Jtrcomp_q = Jtr(q);
} // Done size
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
if (aspectratioidx != -1) //Set aspect ratio
if (aspectratioidx != -1) // Set aspect ratio
{
if (dim == 2)
{
@@ -1262,12 +1316,13 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
D_rho(1,1) = pow(rho2,2./3.);
D_rho(2,2) = pow(rho3,2./3.);
}
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(1 + 4*q), dim, dim);
Jtrcomp_q = D_rho;
DenseMatrix Temp = Jtr(q);
Mult(D_rho, Temp, Jtr(q));
} // Done aspect ratio
DenseMatrix Temp = Jtr(i);
Mult(D_rho, Temp, Jtr(i));
} //Done aspect ratio
if (skewidx != -1) //Set skew
if (skewidx != -1) // Set skew
{
if (dim == 2)
{
@@ -1303,12 +1358,13 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
Q_phi(2,2) = sin(phi13)*sin(chi);
}
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*q), dim, dim);
Jtrcomp_q = Q_phi;
DenseMatrix Temp = Jtr(q);
Mult(Q_phi, Temp, Jtr(q));
} // Done skew
DenseMatrix Temp = Jtr(i);
Mult(Q_phi, Temp, Jtr(i));
} // done skew
if (orientationidx != -1) //Set orientation
if (orientationidx != -1) // Set orientation
{
if (dim == 2)
{
@@ -1333,33 +1389,28 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
const double psi = shape * par_vals_c2;
const double beta = shape * par_vals_c3;
DenseMatrix R_tp(dim), R_beta(dim), R_theta(dim);
double ct = cos(theta), st = sin(theta),
cp = cos(psi), sp = sin(psi);
R_tp(0,0) = ct*sp;
R_tp(1,0) = st*sp;
R_tp(2,0) = cp;
cp = cos(psi), sp = sin(psi),
cb = cos(beta), sb = sin(beta);
R_tp(0,1) = -(ct*st*sp*sp)/(1+cp);
R_tp(1,1) = cp+(pow(ct,2.)*pow(sp,2.))/(1+cp);
R_tp(2,1) = -st*sp;
R_theta = 0.;
R_theta(0,0) = ct*sp;
R_theta(1,0) = st*sp;
R_theta(2,0) = cp;
R_tp(0,2) = -cp-(pow(st,2.)*pow(sp,2.))/(1+cp);
R_tp(1,2) = -R_tp(0,1);
R_tp(2,2) = ct*sp;
R_theta(0,1) = -st*cb + ct*cp*sb;
R_theta(1,1) = ct*cb + st*cp*sb;
R_theta(2,1) = -sp*sb;
R_beta = 0.;
R_beta(0,0) = 1.;
R_beta(1,1) = cos(beta);
R_beta(1,2) = -sin(beta);
R_beta(2,1) = sin(beta);
R_beta(2,2) = cos(beta);
Mult(R_tp, R_beta, R_theta);
R_theta(0,0) = -st*sb - ct*cp*cb;
R_theta(1,0) = ct*sb - st*cp*cb;
R_theta(2,0) = sp*cb;
}
DenseMatrix Temp = Jtr(i);
Mult(R_theta, Temp, Jtr(i));
} // done orientation
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(3 + 4*q), dim, dim);
Jtrcomp_q = R_theta;
DenseMatrix Temp = Jtr(q);
Mult(R_theta, Temp, Jtr(q));
} // Done orientation
}
break;
}
@@ -1368,6 +1419,353 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void DiscreteAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
dJtr = 0.;
const int e_id = Tpr.ElementNo;
const FiniteElement *fe = Tpr.GetFE();
switch (target_type)
{
case IDEAL_SHAPE_GIVEN_SIZE:
case GIVEN_SHAPE_AND_SIZE:
{
const DenseMatrix &Wideal =
Geometries.GetGeomToPerfGeomJac(fe->GetGeomType());
const int dim = Wideal.Height(),
ndofs = fe->GetDof(),
ntspec_dofs = ndofs*ncomp;
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
par_vals_c1(ndofs), par_vals_c2(ndofs), par_vals_c3(ndofs);
Array<int> dofs;
DenseMatrix dD_rho(dim), dQ_phi(dim), dR_theta(dim);
DenseMatrix dQ_phi13(dim), dQ_phichi(dim); // dQ_phi is used for dQ/dphi12 in 3D
DenseMatrix dR_psi(dim), dR_beta(dim);
tspec_fesv->GetElementVDofs(e_id, dofs);
tspec.GetSubVector(dofs, tspec_vals);
DenseMatrix grad_e_c1(ndofs, dim),
grad_e_c2(ndofs, dim),
grad_e_c3(ndofs, dim);
Vector grad_ptr_c1(grad_e_c1.GetData(), ndofs*dim),
grad_ptr_c2(grad_e_c2.GetData(), ndofs*dim),
grad_ptr_c3(grad_e_c3.GetData(), ndofs*dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
fe->ProjectGrad(*fe, Tpr, grad_phys);
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
DenseMatrix Jtrcomp_s(Jtrcomp.GetData(0 + 4*i), dim, dim); // size
DenseMatrix Jtrcomp_d(Jtrcomp.GetData(1 + 4*i), dim, dim); // aspect-ratio
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*i), dim, dim); // skew
DenseMatrix Jtrcomp_r(Jtrcomp.GetData(3 + 4*i), dim, dim); // orientation
DenseMatrix work1(dim), work2(dim), work3(dim);
if (sizeidx != -1) // Set size
{
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double min_size = par_vals.Min();
MFEM_VERIFY(min_size > 0.0,
"Non-positive size propagated in the target definition.");
const double size = std::max(shape * par_vals, min_size);
double dz_dsize = (1./dim)*pow(size, 1./dim - 1.);
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_r, work1, work2); // R*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = Wideal;
work1.Set(dz_dsize, work1); // dz/dsize
work1 *= grad_q(d); // dz/dsize*dsize/dx
AddMult(work1, work2, dJtr_i); // dz/dx*R*Q*D
}
} // Done size
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
if (aspectratioidx != -1) // Set aspect ratio
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
aspectratioidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double aspectratio = shape * par_vals;
dD_rho = 0.;
dD_rho(0,0) = -0.5*pow(aspectratio,-1.5);
dD_rho(1,1) = 0.5*pow(aspectratio,-0.5);
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
Mult(work1, Jtrcomp_q, work2); // z*R*Q
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dD_rho;
work1 *= grad_q(d); // work1 = dD/drho*drho/dx
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
}
}
else // 3D
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
aspectratioidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double rho1 = shape * par_vals_c1;
const double rho2 = shape * par_vals_c2;
const double rho3 = shape * par_vals_c3;
dD_rho = 0.;
dD_rho(0,0) = (2./3.)*pow(rho1,-1./3.);
dD_rho(1,1) = (2./3.)*pow(rho2,-1./3.);
dD_rho(2,2) = (2./3.)*pow(rho3,-1./3.);
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
Mult(work1, Jtrcomp_q, work2); // z*R*Q
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dD_rho;
work1(0,0) *= grad_q1(d);
work1(1,2) *= grad_q2(d);
work1(2,2) *= grad_q3(d);
// work1 = dD/dx = dD/drho1*drho1/dx + dD/drho2*drho2/dx
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
}
}
} // Done aspect ratio
if (skewidx != -1) // Set skew
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
skewidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double skew = shape * par_vals;
dQ_phi = 0.;
dQ_phi(0,0) = 1.;
dQ_phi(0,1) = -sin(skew);
dQ_phi(1,1) = cos(skew);
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dQ_phi;
work1 *= grad_q(d); // work1 = dQ/dphi*dphi/dx
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
}
}
else
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
skewidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double phi12 = shape * par_vals_c1;
const double phi13 = shape * par_vals_c2;
const double chi = shape * par_vals_c3;
dQ_phi = 0.;
dQ_phi(0,0) = 1.;
dQ_phi(0,1) = -sin(phi12);
dQ_phi(1,1) = cos(phi12);
dQ_phi13 = 0.;
dQ_phi13(0,2) = -sin(phi13);
dQ_phi13(1,2) = cos(phi13)*cos(chi);
dQ_phi13(2,2) = cos(phi13)*sin(chi);
dQ_phichi = 0.;
dQ_phichi(1,2) = -sin(phi13)*sin(chi);
dQ_phichi(2,2) = sin(phi13)*cos(chi);
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dQ_phi;
work1 *= grad_q1(d); // work1 = dQ/dphi12*dphi12/dx
work1.Add(grad_q2(d), dQ_phi13); // + dQ/dphi13*dphi13/dx
work1.Add(grad_q3(d), dQ_phichi); // + dQ/dchi*dchi/dx
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
}
}
} // Done skew
if (orientationidx != -1) // Set orientation
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
orientationidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double theta = shape * par_vals;
dR_theta(0,0) = -sin(theta);
dR_theta(0,1) = -cos(theta);
dR_theta(1,0) = cos(theta);
dR_theta(1,1) = -sin(theta);
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_s, work1, work2); // z*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dR_theta;
work1 *= grad_q(d); // work1 = dR/dtheta*dtheta/dx
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
}
}
else
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
orientationidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double theta = shape * par_vals_c1;
const double psi = shape * par_vals_c2;
const double beta = shape * par_vals_c3;
const double ct = cos(theta), st = sin(theta),
cp = cos(psi), sp = sin(psi),
cb = cos(beta), sb = sin(beta);
dR_theta = 0.;
dR_theta(0,0) = -st*sp;
dR_theta(1,0) = ct*sp;
dR_theta(2,0) = 0;
dR_theta(0,1) = -ct*cb - st*cp*sb;
dR_theta(1,1) = -st*cb + ct*cp*sb;
dR_theta(2,1) = 0.;
dR_theta(0,0) = -ct*sb + st*cp*cb;
dR_theta(1,0) = -st*sb - ct*cp*cb;
dR_theta(2,0) = 0.;
dR_beta = 0.;
dR_beta(0,0) = 0.;
dR_beta(1,0) = 0.;
dR_beta(2,0) = 0.;
dR_beta(0,1) = st*sb + ct*cp*cb;
dR_beta(1,1) = -ct*sb + st*cp*cb;
dR_beta(2,1) = -sp*cb;
dR_beta(0,0) = -st*cb + ct*cp*sb;
dR_beta(1,0) = ct*cb + st*cp*sb;
dR_beta(2,0) = 0.;
dR_psi = 0.;
dR_psi(0,0) = ct*cp;
dR_psi(1,0) = st*cp;
dR_psi(2,0) = -sp;
dR_psi(0,1) = 0. - ct*sp*sb;
dR_psi(1,1) = 0. + st*sp*sb;
dR_psi(2,1) = -cp*sb;
dR_psi(0,0) = 0. + ct*sp*cb;
dR_psi(1,0) = 0. + st*sp*cb;
dR_psi(2,0) = cp*cb;
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_s, work1, work2); // z*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dR_theta;
work1 *= grad_q1(d); // work1 = dR/dtheta*dtheta/dx
work1.Add(grad_q2(d), dR_psi); // +dR/dpsi*dpsi/dx
work1.Add(grad_q3(d), dR_beta); // +dR/dbeta*dbeta/dx
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
}
}
} // Done orientation
}
break;
}
default:
MFEM_ABORT("Incompatible target type for discrete adaptation!");
}
Jtrcomp.Clear();
}
void DiscreteAdaptTC::UpdateGradientTargetSpecification(const Vector &x,
const double dx,
bool use_flag)
@@ -1697,6 +2095,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
{
const int dof = el.GetDof(), dim = el.GetDim();
DenseMatrix Amat(dim), work1(dim), work2(dim);
DSh.SetSize(dof, dim);
DS.SetSize(dof, dim);
Jrt.SetSize(dim);
@@ -1712,14 +2111,15 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
elvect = 0.0;
Vector weights(nqp);
DenseTensor Jtr(dim, dim, nqp);
DenseTensor dJtr(dim, dim, dim*nqp);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
// Limited case.
DenseMatrix pos0;
Vector shape, p, p0, d_vals, grad;
shape.SetSize(dof);
if (coeff0)
{
shape.SetSize(dof);
p.SetSize(dim);
p0.SetSize(dim);
pos0.SetSize(dof, dim);
@@ -1739,16 +2139,23 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (coeff1 || coeff0 || zeta)
if (coeff1 || coeff0 || zeta || exact_action)
{
Tpr = new IsoparametricTransformation;
Tpr->SetFE(&el);
Tpr->ElementNo = T.ElementNo;
Tpr->ElementType = ElementTransformation::ELEMENT;
Tpr->Attribute = T.Attribute;
Tpr->GetPointMat().Transpose(PMatI); // PointMat = PMatI^T
if (exact_action)
{
targetC->ComputeElementTargetsGradient(*ir, elfun, *Tpr, dJtr);
}
}
Vector d_detW_dx(dim);
Vector d_Winv_dx(dim);
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir->IntPoint(q);
@@ -1767,13 +2174,44 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
if (coeff1) { weight_m *= coeff1->Eval(*Tpr, ip); }
P *= weight_m;
AddMultABt(DS, P, PMatO);
AddMultABt(DS, P, PMatO); // w_q det(W) dmu/dx : dA/dx Winv
// TODO: derivatives of adaptivity-based targets.
if (exact_action)
{
el.CalcShape(ip, shape);
// Derivatives of adaptivity-based targets.
// First term: w_q d*(Det W)/dx * mu(T)
// d(Det W)/dx = det(W)*Tr[Winv*dW/dx]
DenseMatrix dwdx(dim);
for (int d = 0; d < dim; d++)
{
const DenseMatrix &dJtr_q = dJtr(q + d*ir->GetNPoints());
Mult(Jrt, dJtr_q, dwdx );
d_detW_dx(d) = dwdx.Trace();
}
d_detW_dx *= weight_m*metric->EvalW(Jpt); // *[w_q*det(W)]*mu(T)
// Second term: w_q det(W) dmu/dx : AdWinv/dx
// dWinv/dx = -Winv*dW/dx*Winv
MultAtB(PMatI, DSh, Amat);
for (int d = 0; d < dim; d++)
{
const DenseMatrix &dJtr_q = dJtr(q + d*nqp);
Mult(Jrt, dJtr_q, work1); // Winv*dw/dx
Mult(work1, Jrt, work2); // Winv*dw/dx*Winv
Mult(Amat, work2, work1); // A*Winv*dw/dx*Winv
MultAtB(P, work1, work2); // dmu/dT^T*A*Winv*dw/dx*Winv
d_Winv_dx(d) = work2.Trace(); // Tr[dmu/dT : AWinv*dw/dx*Winv]
}
d_Winv_dx *= -weight_m; // Include (-) factor as well
d_detW_dx += d_Winv_dx;
AddMultVWt(shape, d_detW_dx, PMatO);
}
if (coeff0)
{
el.CalcShape(ip, shape);
if (!exact_action) { el.CalcShape(ip, shape); }
PMatI.MultTranspose(shape, p);
pos0.MultTranspose(shape, p0);
lim_func->Eval_d1(p, p0, d_vals(q), grad);
@@ -2214,7 +2652,8 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
Jpt.SetSize(dim);
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
DenseTensor Jtr(dim, dim, ir->GetNPoints());
const int nqp = ir->GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
metric_energy = 0.0;
lim_energy = 0.0;
@@ -2227,12 +2666,12 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
targetC->ComputeElementTargets(i, *fe, *ir, x_vals, Jtr);
for (int i = 0; i < ir->GetNPoints(); i++)
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
metric->SetTargetJacobian(Jtr(i));
CalcInverse(Jtr(i), Jrt);
const double weight = ip.weight * Jtr(i).Det();
const IntegrationPoint &ip = ir->IntPoint(q);
metric->SetTargetJacobian(Jtr(q));
CalcInverse(Jtr(q), Jrt);
const double weight = ip.weight * Jtr(q).Det();
fe->CalcDShape(ip, DSh);
MultAtB(PMatI, DSh, Jpr);
+44 -5
View File
@@ -677,6 +677,25 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
class TMOPMatrixCoefficient : public MatrixCoefficient
{
public:
explicit TMOPMatrixCoefficient(int dim) : MatrixCoefficient(dim, dim) { }
/** @brief Evaluate the derivative of the matrix coefficient with respect to
@a comp in the element described by @a T at the point @a ip, storing the
result in @a K. */
virtual void EvalGrad(DenseMatrix &K, ElementTransformation &T,
const IntegrationPoint &ip, int comp) = 0;
virtual ~TMOPMatrixCoefficient() { }
};
class AnalyticAdaptTC : public TargetConstructor
@@ -685,7 +704,7 @@ protected:
// Analytic target specification.
Coefficient *scalar_tspec;
VectorCoefficient *vector_tspec;
MatrixCoefficient *matrix_tspec;
TMOPMatrixCoefficient *matrix_tspec;
public:
AnalyticAdaptTC(TargetType ttype)
@@ -694,7 +713,7 @@ public:
virtual void SetAnalyticTargetSpec(Coefficient *sspec,
VectorCoefficient *vspec,
MatrixCoefficient *mspec);
TMOPMatrixCoefficient *mspec);
/** @brief Given an element and quadrature rule, computes ref->target
transformation Jacobians for each quadrature point in the element.
@@ -703,6 +722,11 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
#ifdef MFEM_USE_MPI
@@ -724,13 +748,17 @@ protected:
// eta1(x+h,y), eta2(x+h,y) ... etan(x+h,y), eta1(x,y+h), eta2(x,y+h) ...
// same for tspec_pert2h and tspec_pertmix.
// Components of Target Jacobian at each quadrature point of an element. This
// is required for computation of the derivative using chain rule.
mutable DenseTensor Jtrcomp;
// Note: do not use the Nodes of this space as they may not be on the
// positions corresponding to the values of tspec.
const FiniteElementSpace *tspec_fes;
const FiniteElementSpace *tspec_fesv;
// These flags can be used by outside functions to avoid recomputing
// the tspec and tspec_perth fields again on the same mesh.
// These flags can be used by outside functions to avoid recomputing the
// tspec and tspec_perth fields again on the same mesh.
bool good_tspec, good_tspec_grad, good_tspec_hess;
// Evaluation of the discrete target specification on different meshes.
@@ -837,6 +865,11 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
class TMOPNewtonSolver;
@@ -889,6 +922,9 @@ protected:
// Specifies that ComputeElementTargets is being called by a FD function.
// It's used to skip terms that have exact derivative calculations.
bool fd_call_flag;
// Compute the exact action of the Integrator (includes derivative of the
// target with respect to spatial position)
bool exact_action;
Array <Vector *> ElemDer; //f'(x)
Array <Vector *> ElemPertEnergy; //f(x+h)
@@ -978,7 +1014,7 @@ public:
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
zeta_0(NULL), zeta(NULL), coeff_zeta(NULL), adapt_eval(NULL),
discr_tc(dynamic_cast<DiscreteAdaptTC *>(tc)),
fdflag(false), dxscale(1.0e3), fd_call_flag(false)
fdflag(false), dxscale(1.0e3), fd_call_flag(false), exact_action(false)
{ }
~TMOP_Integrator();
@@ -1066,6 +1102,9 @@ public:
void SetFDhScale(double _dxscale) { dxscale = _dxscale; }
bool GetFDFlag() const { return fdflag; }
double GetFDh() const { return dx; }
/** @brief Flag to control if exact action of Integration is effected. */
void SetExactActionFlag(bool flag_) { exact_action = flag_; }
};
class TMOPComboIntegrator : public NonlinearFormIntegrator
+1
View File
@@ -32,6 +32,7 @@ list(APPEND SRCS
list(APPEND HDRS
array.hpp
backends.hpp
binaryio.hpp
cuda.hpp
device.hpp
+3 -3
View File
@@ -36,9 +36,9 @@ void Swap(Array<T> &, Array<T> &);
Abstract data type Array.
Array<T> is an automatically increasing array containing elements of the
generic type T. The allocated size may be larger then the logical size
of the array.
The elements can be accessed by the [] operator, the range is 0 to size-1.
generic type T, which must be a POD (plain old data) type. The allocated size
may be larger then the logical size of the array. The elements can be
accessed by the [] operator, the range is 0 to size-1.
*/
template <class T>
class Array
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_BACKENDS_HPP
#define MFEM_BACKENDS_HPP
#include "../config/config.hpp"
#ifdef MFEM_USE_CUDA
#include <cuda_runtime.h>
#include <cuda.h>
#endif
#include "cuda.hpp"
#ifdef MFEM_USE_HIP
#include <hip/hip_runtime.h>
#endif
#include "hip.hpp"
#ifdef MFEM_USE_OCCA
#include <occa.hpp>
#include "occa.hpp"
#endif
#ifdef MFEM_USE_RAJA
#include "RAJA/RAJA.hpp"
#if defined(RAJA_ENABLE_CUDA) && !defined(MFEM_USE_CUDA)
#error When RAJA is built with CUDA, MFEM_USE_CUDA=YES is required
#endif
#endif
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
#define MFEM_DEVICE
#define MFEM_LAMBDA
#define MFEM_HOST_DEVICE
// MFEM_DEVICE_SYNC is made available for debugging purposes
#define MFEM_DEVICE_SYNC
// MFEM_STREAM_SYNC is used for UVM and MPI GPU-Aware kernels
#define MFEM_STREAM_SYNC
#endif
#if !((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
#define MFEM_SHARED
#define MFEM_SYNC_THREAD
#define MFEM_THREAD_ID(k) 0
#define MFEM_THREAD_SIZE(k) 1
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
#endif
template <typename T>
MFEM_HOST_DEVICE T AtomicAdd(T &add, const T val)
{
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
return atomicAdd(&add,val);
#else
T old = add;
add += val;
return old;
#endif
}
#endif // MFEM_BACKENDS_HPP
+1 -1
View File
@@ -9,7 +9,7 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "cuda.hpp"
#include "backends.hpp"
#include "globals.hpp"
namespace mfem
+3 -23
View File
@@ -15,17 +15,15 @@
#include "../config/config.hpp"
#include "error.hpp"
#ifdef MFEM_USE_CUDA
#include <cuda_runtime.h>
#include <cuda.h>
#endif
// CUDA block size used by MFEM.
#define MFEM_CUDA_BLOCKS 256
#ifdef MFEM_USE_CUDA
#define MFEM_DEVICE __device__
#define MFEM_LAMBDA __host__
#define MFEM_HOST_DEVICE __host__ __device__
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
// Define a CUDA error check macro, MFEM_GPU_CHECK(x), where x returns/is of
// type 'cudaError_t'. This macro evaluates 'x' and raises an error if the
// result is not cudaSuccess.
@@ -39,8 +37,6 @@
} \
} \
while (0)
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
#endif // MFEM_USE_CUDA
// Define the MFEM inner threading macros
@@ -52,22 +48,6 @@
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=threadIdx.k; i<N; i+=blockDim.k)
#endif
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
#define MFEM_DEVICE
#define MFEM_HOST_DEVICE
#define MFEM_DEVICE_SYNC
#define MFEM_STREAM_SYNC
#endif
#if !((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__ROCM_ARCH__)))
#define MFEM_SHARED
#define MFEM_SYNC_THREAD
#define MFEM_THREAD_ID(k) 0
#define MFEM_THREAD_SIZE(k) 1
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
#endif
namespace mfem
{
+11
View File
@@ -144,4 +144,15 @@ void mfem_warning(const char *msg = NULL);
"invalid index " #i << " = " << (i) << \
", valid range is [" << (imin) << ',' << (imax) << ')')
// Abort inside a device kernel
#if defined(__CUDA_ARCH__)
#define MFEM_ABORT_KERNEL(msg) \
{ \
printf(msg); \
asm("trap;"); \
}
#else
#define MFEM_ABORT_KERNEL(msg) MFEM_ABORT(msg)
#endif
#endif
+5 -14
View File
@@ -14,20 +14,11 @@
#include "../config/config.hpp"
#include "error.hpp"
#include "cuda.hpp"
#include "hip.hpp"
#include "occa.hpp"
#include "backends.hpp"
#include "device.hpp"
#include "mem_manager.hpp"
#include "../linalg/dtensor.hpp"
#ifdef MFEM_USE_RAJA
#include "RAJA/RAJA.hpp"
#if defined(RAJA_ENABLE_CUDA) && !defined(MFEM_USE_CUDA)
#error When RAJA is built with CUDA, MFEM_USE_CUDA=YES is required
#endif
#endif
namespace mfem
{
@@ -52,20 +43,20 @@ const int MAX_Q1D = 14;
#define MFEM_FORALL(i,N,...) \
ForallWrap<1>(true,N, \
[=] MFEM_DEVICE (int i) {__VA_ARGS__}, \
[&] (int i) {__VA_ARGS__})
[&] MFEM_LAMBDA (int i) {__VA_ARGS__})
// MFEM_FORALL with a 2D CUDA block
#define MFEM_FORALL_2D(i,N,X,Y,BZ,...) \
ForallWrap<2>(true,N, \
[=] MFEM_DEVICE (int i) {__VA_ARGS__}, \
[&] (int i) {__VA_ARGS__}, \
[&] MFEM_LAMBDA (int i) {__VA_ARGS__},\
X,Y,BZ)
// MFEM_FORALL with a 3D CUDA block
#define MFEM_FORALL_3D(i,N,X,Y,Z,...) \
ForallWrap<3>(true,N, \
[=] MFEM_DEVICE (int i) {__VA_ARGS__}, \
[&] (int i) {__VA_ARGS__}, \
[&] MFEM_LAMBDA (int i) {__VA_ARGS__},\
X,Y,Z)
// MFEM_FORALL that uses the basic CPU backend when use_dev is false. See for
@@ -74,7 +65,7 @@ const int MAX_Q1D = 14;
#define MFEM_FORALL_SWITCH(use_dev,i,N,...) \
ForallWrap<1>(use_dev,N, \
[=] MFEM_DEVICE (int i) {__VA_ARGS__}, \
[&] (int i) {__VA_ARGS__})
[&] MFEM_LAMBDA (int i) {__VA_ARGS__})
/// OpenMP backend
-1
View File
@@ -125,5 +125,4 @@ void SetGlobalMPI_Comm(MPI_Comm comm);
#define MFEM_DEPRECATED
#endif
#endif
+1 -1
View File
@@ -9,7 +9,7 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "hip.hpp"
#include "backends.hpp"
#include "globals.hpp"
namespace mfem
+6 -8
View File
@@ -15,16 +15,15 @@
#include "../config/config.hpp"
#include "error.hpp"
#ifdef MFEM_USE_HIP
#include <hip/hip_runtime.h>
#endif
// HIP block size used by MFEM.
#define MFEM_HIP_BLOCKS 256
#ifdef MFEM_USE_HIP
#define MFEM_DEVICE __device__
#define MFEM_LAMBDA __host__ __device__
#define MFEM_HOST_DEVICE __host__ __device__
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
// Define a HIP error check macro, MFEM_GPU_CHECK(x), where x returns/is of
// type 'hipError_t'. This macro evaluates 'x' and raises an error if the
// result is not hipSuccess.
@@ -38,17 +37,16 @@
} \
} \
while (0)
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(hipDeviceSynchronize())
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(hipStreamSynchronize(0))
#endif // MFEM_USE_HIP
// Define the MFEM inner threading macros
#if defined(MFEM_USE_HIP) && defined(__ROCM_ARCH__)
#if defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)
#define MFEM_SHARED __shared__
#define MFEM_SYNC_THREAD __syncthreads()
#define MFEM_THREAD_ID(k) hipThreadIdx_ ##k
#define MFEM_THREAD_SIZE(k) hipBlockDim_ ##k
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#define MFEM_FOREACH_THREAD(i,k,N) \
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#endif
namespace mfem
+1 -1
View File
@@ -16,8 +16,8 @@
#ifdef MFEM_USE_OCCA
#include "mem_manager.hpp"
#include "backends.hpp"
#include "device.hpp"
#include <occa.hpp>
namespace mfem
{
+1 -2
View File
@@ -13,8 +13,7 @@
#define MFEM_TEMPLATE_ASSIGN
#include "../config/tconfig.hpp"
#include "../general/cuda.hpp"
#include "../general/hip.hpp"
#include "backends.hpp"
namespace mfem
{
+58 -11
View File
@@ -74,10 +74,12 @@ void BlockOperator::Mult (const Vector & x, Vector & y) const
MFEM_ASSERT(x.Size() == width, "incorrect input Vector size");
MFEM_ASSERT(y.Size() == height, "incorrect output Vector size");
yblock.Update(y.GetData(),row_offsets);
xblock.Update(x.GetData(),col_offsets);
x.Read();
y.Write(); y = 0.0;
xblock.Update(const_cast<Vector&>(x),col_offsets);
yblock.Update(y,row_offsets);
y = 0.0;
for (int iRow=0; iRow < nRowBlocks; ++iRow)
{
tmp.SetSize(row_offsets[iRow+1] - row_offsets[iRow]);
@@ -90,6 +92,16 @@ void BlockOperator::Mult (const Vector & x, Vector & y) const
}
}
}
for (int iRow=0; iRow < nRowBlocks; ++iRow)
{
yblock.GetBlock(iRow).SyncAliasMemory(y);
}
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
for (int i=0; i < xblock.NumBlocks(); ++i) { xblock.GetBlock(i).Destroy(); }
for (int i=0; i < yblock.NumBlocks(); ++i) { yblock.GetBlock(i).Destroy(); }
}
// Action of the transpose operator
@@ -98,10 +110,11 @@ void BlockOperator::MultTranspose (const Vector & x, Vector & y) const
MFEM_ASSERT(x.Size() == height, "incorrect input Vector size");
MFEM_ASSERT(y.Size() == width, "incorrect output Vector size");
y = 0.0;
x.Read();
y.Write(); y = 0.0;
xblock.Update(x.GetData(),row_offsets);
yblock.Update(y.GetData(),col_offsets);
xblock.Update(const_cast<Vector&>(x),row_offsets);
yblock.Update(y,col_offsets);
for (int iRow=0; iRow < nColBlocks; ++iRow)
{
@@ -116,6 +129,15 @@ void BlockOperator::MultTranspose (const Vector & x, Vector & y) const
}
}
for (int iRow=0; iRow < nColBlocks; ++iRow)
{
yblock.GetBlock(iRow).SyncAliasMemory(y);
}
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
for (int i=0; i < xblock.NumBlocks(); ++i) { xblock.GetBlock(i).Destroy(); }
for (int i=0; i < yblock.NumBlocks(); ++i) { yblock.GetBlock(i).Destroy(); }
}
BlockOperator::~BlockOperator()
@@ -140,7 +162,6 @@ BlockDiagonalPreconditioner::BlockDiagonalPreconditioner(
nBlocks(offsets_.Size() - 1),
offsets(0),
op(nBlocks)
{
op = static_cast<Operator *>(NULL);
offsets.MakeRef(offsets_);
@@ -165,8 +186,11 @@ void BlockDiagonalPreconditioner::Mult (const Vector & x, Vector & y) const
MFEM_ASSERT(x.Size() == width, "incorrect input Vector size");
MFEM_ASSERT(y.Size() == height, "incorrect output Vector size");
yblock.Update(y.GetData(), offsets);
xblock.Update(x.GetData(), offsets);
x.Read();
y.Write(); y = 0.0;
xblock.Update(const_cast<Vector&>(x),offsets);
yblock.Update(y,offsets);
for (int i=0; i<nBlocks; ++i)
{
@@ -179,6 +203,16 @@ void BlockDiagonalPreconditioner::Mult (const Vector & x, Vector & y) const
yblock.GetBlock(i) = xblock.GetBlock(i);
}
}
for (int i=0; i<nBlocks; ++i)
{
yblock.GetBlock(i).SyncAliasMemory(y);
}
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
for (int i=0; i < xblock.NumBlocks(); ++i) { xblock.GetBlock(i).Destroy(); }
for (int i=0; i < yblock.NumBlocks(); ++i) { yblock.GetBlock(i).Destroy(); }
}
// Action of the transpose operator
@@ -188,8 +222,11 @@ void BlockDiagonalPreconditioner::MultTranspose (const Vector & x,
MFEM_ASSERT(x.Size() == height, "incorrect input Vector size");
MFEM_ASSERT(y.Size() == width, "incorrect output Vector size");
yblock.Update(y.GetData(), offsets);
xblock.Update(x.GetData(), offsets);
x.Read();
y.Write(); y = 0.0;
xblock.Update(const_cast<Vector&>(x),offsets);
yblock.Update(y,offsets);
for (int i=0; i<nBlocks; ++i)
{
@@ -202,6 +239,16 @@ void BlockDiagonalPreconditioner::MultTranspose (const Vector & x,
yblock.GetBlock(i) = xblock.GetBlock(i);
}
}
for (int i=0; i<nBlocks; ++i)
{
yblock.GetBlock(i).SyncAliasMemory(y);
}
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
for (int i=0; i < xblock.NumBlocks(); ++i) { xblock.GetBlock(i).Destroy(); }
for (int i=0; i < yblock.NumBlocks(); ++i) { yblock.GetBlock(i).Destroy(); }
}
BlockDiagonalPreconditioner::~BlockDiagonalPreconditioner()
+16
View File
@@ -87,6 +87,22 @@ void BlockVector::Update(double *data, const Array<int> & bOffsets)
SetBlocks();
}
void BlockVector::Update(Vector & data, const Array<int> & bOffsets)
{
blockOffsets = bOffsets.GetData();
if (numBlocks != bOffsets.Size()-1)
{
delete [] blocks;
numBlocks = bOffsets.Size()-1;
blocks = new Vector[numBlocks];
}
for (int i = 0; i < numBlocks; ++i)
{
blocks[i].MakeRef(data, blockOffsets[i], BlockSize(i));
}
}
void BlockVector::Update(const Array<int> &bOffsets)
{
Update(bOffsets, data.GetMemoryType());
+2
View File
@@ -102,6 +102,8 @@ public:
*/
void Update(double *data, const Array<int> & bOffsets);
void Update(Vector & data, const Array<int> & bOffsets);
/// Update a BlockVector with new @a bOffsets and make sure it owns its data.
/** The block-vector will be re-allocated if either:
- the offsets @a bOffsets are different from the current offsets, or
+309 -22
View File
@@ -26,10 +26,10 @@ ComplexOperator::ComplexOperator(Operator * Op_Real, Operator * Op_Imag,
, ownReal_(ownReal)
, ownImag_(ownImag)
, convention_(convention)
, x_r_(NULL, width / 2)
, x_i_(NULL, width / 2)
, y_r_(NULL, height / 2)
, y_i_(NULL, height / 2)
, x_r_()
, x_i_()
, y_r_()
, y_i_()
, u_(NULL)
, v_(NULL)
{}
@@ -68,14 +68,26 @@ const Operator & ComplexOperator::imag() const
void ComplexOperator::Mult(const Vector &x, Vector &y) const
{
double * x_data = x.GetData();
x_r_.SetData(x_data);
x_i_.SetData(&x_data[width / 2]);
x.Read();
y.UseDevice(true); y = 0.0;
y_r_.SetData(&y[0]);
y_i_.SetData(&y[height / 2]);
x_r_.MakeRef(const_cast<Vector&>(x), 0, width/2);
x_i_.MakeRef(const_cast<Vector&>(x), width/2, width/2);
y_r_.MakeRef(y, 0, height/2);
y_i_.MakeRef(y, height/2, height/2);
this->Mult(x_r_, x_i_, y_r_, y_i_);
y_r_.SyncAliasMemory(y);
y_i_.SyncAliasMemory(y);
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
x_r_.Destroy();
x_i_.Destroy();
y_r_.Destroy();
y_i_.Destroy();
}
void ComplexOperator::Mult(const Vector &x_r, const Vector &x_i,
@@ -91,31 +103,47 @@ void ComplexOperator::Mult(const Vector &x_r, const Vector &x_i,
y_r = 0.0;
y_i = 0.0;
}
if (Op_Imag_)
{
if (!v_) { v_ = new Vector(Op_Imag_->Height()); }
if (!v_) { v_ = new Vector(); }
v_->UseDevice(true);
v_->SetSize(Op_Imag_->Height());
Op_Imag_->Mult(x_i, *v_);
y_r_ -= *v_;
y_r.Add(-1.0, *v_);
Op_Imag_->Mult(x_r, *v_);
y_i_ += *v_;
y_i.Add(1.0, *v_);
}
if (convention_ == BLOCK_SYMMETRIC)
{
y_i_ *= -1.0;
y_i *= -1.0;
}
}
void ComplexOperator::MultTranspose(const Vector &x, Vector &y) const
{
double * x_data = x.GetData();
y_r_.SetData(x_data);
y_i_.SetData(&x_data[height / 2]);
x.Read();
y.UseDevice(true); y = 0.0;
x_r_.SetData(&y[0]);
x_i_.SetData(&y[width / 2]);
x_r_.MakeRef(const_cast<Vector&>(x), 0, height/2);
x_i_.MakeRef(const_cast<Vector&>(x), height/2, height/2);
this->MultTranspose(y_r_, y_i_, x_r_, x_i_);
y_r_.MakeRef(y, 0, width/2);
y_i_.MakeRef(y, width/2, width/2);
this->MultTranspose(x_r_, x_i_, y_r_, y_i_);
y_r_.SyncAliasMemory(y);
y_i_.SyncAliasMemory(y);
// Destroy alias vectors to prevent dangling aliases when the base vectors
// are deleted
x_r_.Destroy();
x_i_.Destroy();
y_r_.Destroy();
y_i_.Destroy();
}
void ComplexOperator::MultTranspose(const Vector &x_r, const Vector &x_i,
@@ -136,13 +164,17 @@ void ComplexOperator::MultTranspose(const Vector &x_r, const Vector &x_i,
y_r = 0.0;
y_i = 0.0;
}
if (Op_Imag_)
{
if (!u_) { u_ = new Vector(Op_Imag_->Width()); }
if (!u_) { u_ = new Vector(); }
u_->UseDevice(true);
u_->SetSize(Op_Imag_->Width());
Op_Imag_->MultTranspose(x_i, *u_);
y_r_.Add(convention_ == BLOCK_SYMMETRIC ? -1.0 : 1.0, *u_);
y_r.Add(convention_ == BLOCK_SYMMETRIC ? -1.0 : 1.0, *u_);
Op_Imag_->MultTranspose(x_r, *u_);
y_i_ -= *u_;
y_i.Add(-1.0, *u_);
}
}
@@ -235,6 +267,261 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
return new SparseMatrix(I, J, D, this->Height(), this->Width());
}
#ifdef MFEM_USE_SUITESPARSE
void ComplexUMFPackSolver::Init()
{
mat = NULL;
Numeric = NULL;
AI = AJ = NULL;
if (!use_long_ints)
{
umfpack_zi_defaults(Control);
}
else
{
umfpack_zl_defaults(Control);
}
}
void ComplexUMFPackSolver::SetOperator(const Operator &op)
{
int *Ap, *Ai;
void *Symbolic;
double *Ax;
double *Az;
if (Numeric)
{
if (!use_long_ints)
{
umfpack_zi_free_numeric(&Numeric);
}
else
{
umfpack_zl_free_numeric(&Numeric);
}
}
mat = const_cast<ComplexSparseMatrix *>
(dynamic_cast<const ComplexSparseMatrix *>(&op));
MFEM_VERIFY(mat, "not a ComplexSparseMatrix");
MFEM_VERIFY(mat->real().NumNonZeroElems() == mat->imag().NumNonZeroElems(),
"Real and imag Sparsity pattern mismatch: Try setting Assemble (skip_zeros = 0)");
// UMFPack requires that the column-indices in mat corresponding to each
// row be sorted.
// Generally, this will modify the ordering of the entries of mat.
mat->real().SortColumnIndices();
mat->imag().SortColumnIndices();
height = mat->real().Height();
width = mat->real().Width();
MFEM_VERIFY(width == height, "not a square matrix");
Ap = mat->real().GetI(); // assuming real and imag have the same sparsity
Ai = mat->real().GetJ();
Ax = mat->real().GetData();
Az = mat->imag().GetData();
if (!use_long_ints)
{
int status = umfpack_zi_symbolic(width,width,Ap,Ai,Ax,Az,&Symbolic,
Control,Info);
if (status < 0)
{
umfpack_zi_report_info(Control, Info);
umfpack_zi_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::SetOperator :"
" umfpack_zi_symbolic() failed!");
}
status = umfpack_zi_numeric(Ap, Ai, Ax, Az, Symbolic, &Numeric,
Control, Info);
if (status < 0)
{
umfpack_zi_report_info(Control, Info);
umfpack_zi_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::SetOperator :"
" umfpack_zi_numeric() failed!");
}
umfpack_zi_free_symbolic(&Symbolic);
}
else
{
SuiteSparse_long status;
delete [] AJ;
delete [] AI;
AI = new SuiteSparse_long[width + 1];
AJ = new SuiteSparse_long[Ap[width]];
for (int i = 0; i <= width; i++)
{
AI[i] = (SuiteSparse_long)(Ap[i]);
}
for (int i = 0; i < Ap[width]; i++)
{
AJ[i] = (SuiteSparse_long)(Ai[i]);
}
status = umfpack_zl_symbolic(width, width, AI, AJ, Ax, Az, &Symbolic,
Control, Info);
if (status < 0)
{
umfpack_zl_report_info(Control, Info);
umfpack_zl_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::SetOperator :"
" umfpack_zl_symbolic() failed!");
}
status = umfpack_zl_numeric(AI, AJ, Ax, Az, Symbolic, &Numeric,
Control, Info);
if (status < 0)
{
umfpack_zl_report_info(Control, Info);
umfpack_zl_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::SetOperator :"
" umfpack_zl_numeric() failed!");
}
umfpack_zl_free_symbolic(&Symbolic);
}
}
void ComplexUMFPackSolver::Mult(const Vector &b, Vector &x) const
{
if (mat == NULL)
mfem_error("ComplexUMFPackSolver::Mult : matrix is not set!"
" Call SetOperator first!");
int n = b.Size()/2;
double * datax = x.GetData();
double * datab = b.GetData();
// For the Block Symmetric case data the imaginary part
// has to be scaled by -1
ComplexOperator::Convention conv = mat->GetConvention();
Vector bimag;
if (conv == ComplexOperator::Convention::BLOCK_SYMMETRIC)
{
bimag.SetDataAndSize(&datab[n],n);
bimag *=-1.0;
}
// Solve the transpose, since UMFPack expects CCS instead of CRS format
if (!use_long_ints)
{
int status =
umfpack_zi_solve(UMFPACK_Aat, mat->real().GetI(), mat->real().GetJ(),
mat->real().GetData(), mat->imag().GetData(),
datax, &datax[n], datab, &datab[n], Numeric, Control, Info);
umfpack_zi_report_info(Control, Info);
if (status < 0)
{
umfpack_zi_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::Mult : umfpack_zi_solve() failed!");
}
}
else
{
SuiteSparse_long status =
umfpack_zl_solve(UMFPACK_Aat,AI,AJ,mat->real().GetData(),
mat->imag().GetData(),
datax,&datax[n],datab,&datab[n],Numeric,Control,Info);
umfpack_zl_report_info(Control, Info);
if (status < 0)
{
umfpack_zl_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::Mult : umfpack_zl_solve() failed!");
}
}
if (conv == ComplexOperator::Convention::BLOCK_SYMMETRIC)
{
bimag *=-1.0;
}
}
void ComplexUMFPackSolver::MultTranspose(const Vector &b, Vector &x) const
{
if (mat == NULL)
mfem_error("ComplexUMFPackSolver::Mult : matrix is not set!"
" Call SetOperator first!");
int n = b.Size()/2;
double * datax = x.GetData();
double * datab = b.GetData();
ComplexOperator::Convention conv = mat->GetConvention();
Vector bimag;
bimag.SetDataAndSize(&datab[n],n);
// Solve the Adjoint A^H x = b by solving
// the conjugate problem A^T \bar{x} = \bar{b}
if ((!transa && conv == ComplexOperator::HERMITIAN) ||
( transa && conv == ComplexOperator::BLOCK_SYMMETRIC))
{
bimag *=-1.0;
}
if (!use_long_ints)
{
int status =
umfpack_zi_solve(UMFPACK_A, mat->real().GetI(), mat->real().GetJ(),
mat->real().GetData(), mat->imag().GetData(),
datax, &datax[n], datab, &datab[n], Numeric, Control, Info);
umfpack_zi_report_info(Control, Info);
if (status < 0)
{
umfpack_zi_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::Mult : umfpack_zi_solve() failed!");
}
}
else
{
SuiteSparse_long status =
umfpack_zl_solve(UMFPACK_A,AI,AJ,mat->real().GetData(),
mat->imag().GetData(),
datax,&datax[n],datab,&datab[n],Numeric,Control,Info);
umfpack_zl_report_info(Control, Info);
if (status < 0)
{
umfpack_zl_report_status(Control, status);
mfem_error("ComplexUMFPackSolver::Mult : umfpack_zl_solve() failed!");
}
}
if (!transa)
{
Vector ximag;
ximag.SetDataAndSize(&datax[n],n);
ximag *=-1.0;
}
if ((!transa && conv == ComplexOperator::HERMITIAN) ||
( transa && conv == ComplexOperator::BLOCK_SYMMETRIC))
{
bimag *=-1.0;
}
}
ComplexUMFPackSolver::~ComplexUMFPackSolver()
{
delete [] AJ;
delete [] AI;
if (Numeric)
{
if (!use_long_ints)
{
umfpack_zi_free_numeric(&Numeric);
}
else
{
umfpack_zl_free_numeric(&Numeric);
}
}
}
#endif
#ifdef MFEM_USE_MPI
ComplexHypreParMatrix::ComplexHypreParMatrix(HypreParMatrix * A_Real,
+71
View File
@@ -18,6 +18,10 @@
#include "hypre.hpp"
#endif
#ifdef MFEM_USE_SUITESPARSE
#include <umfpack.h>
#endif
namespace mfem
{
@@ -109,6 +113,8 @@ public:
virtual Type GetType() const { return Complex_Operator; }
Convention GetConvention() const { return convention_; }
protected:
// Let this be hidden from the public interface since the implementation
// depends on internal members
@@ -168,6 +174,71 @@ public:
virtual Type GetType() const { return MFEM_ComplexSparseMat; }
};
#ifdef MFEM_USE_SUITESPARSE
/** @brief Interface with UMFPack solver specialized for ComplexSparseMatrix
This approach avoids forming a monolithic SparseMatrix which leads
to increased memory and flops
*/
class ComplexUMFPackSolver : public Solver
{
protected:
bool use_long_ints;
bool transa;
ComplexSparseMatrix *mat;
void *Numeric;
SuiteSparse_long *AI, *AJ;
void Init();
public:
double Control[UMFPACK_CONTROL];
mutable double Info[UMFPACK_INFO];
/** @brief For larger matrices, if the solver fails, set the parameter @a
_use_long_ints = true. */
ComplexUMFPackSolver(bool _use_long_ints = false, bool transa_ = false)
: use_long_ints(_use_long_ints), transa(transa_) { Init(); }
/** @brief Factorize the given ComplexSparseMatrix using the defaults.
For larger matrices, if the solver fails, set the parameter
@a _use_long_ints = true. */
ComplexUMFPackSolver(ComplexSparseMatrix &A, bool _use_long_ints = false,
bool transa_ = false)
: use_long_ints(_use_long_ints), transa(transa_) { Init(); SetOperator(A); }
/** @brief Factorize the given Operator @a op which must be
a ComplexSparseMatrix.
The factorization uses the parameters set in the #Control data member.
@note This method calls SparseMatrix::SortColumnIndices()
for real and imag parts of the ComplexSparseMatrix,
modifying the matrices if the column indices are not already sorted. */
virtual void SetOperator(const Operator &op);
// Set the print level field in the #Control data member.
void SetPrintLevel(int print_lvl) { Control[UMFPACK_PRL] = print_lvl; }
// This determines the action of MultTranspose (see below for details)
void SetTransposeSolve(bool transa_) { transa = transa_; }
/** @brief This is solving the system A x = b */
virtual void Mult(const Vector &b, Vector &x) const;
/** @brief
This is solving the system:
A^H x = b (when transa = false)
This is equivalent to solving the transpose block system for the
case of Convention = HERMITIAN
A^T x = b (when transa = true)
This is equivalent to solving the transpose block system for the
case of Convention = BLOCK_SYMMETRIC */
virtual void MultTranspose(const Vector &b, Vector &x) const;
virtual ~ComplexUMFPackSolver();
};
#endif
#ifdef MFEM_USE_MPI
/** @brief Specialization of the ComplexOperator built from a pair of
+1 -1
View File
@@ -12,7 +12,7 @@
#ifndef MFEM_DTENSOR
#define MFEM_DTENSOR
#include "../general/cuda.hpp"
#include "../general/backends.hpp"
namespace mfem
{
+6 -9
View File
@@ -21,6 +21,10 @@
#include <cmath>
#include <cstdlib>
#ifdef MFEM_USE_SUNDIALS
#include <nvector/nvector_parallel.h>
#endif
using namespace std;
namespace mfem
@@ -179,16 +183,9 @@ HypreParVector::~HypreParVector()
#ifdef MFEM_USE_SUNDIALS
void HypreParVector::ToNVector(N_Vector &nv)
N_Vector HypreParVector::ToNVector()
{
MFEM_ASSERT(nv && N_VGetVectorID(nv) == SUNDIALS_NVEC_PARHYP,
"invalid N_Vector");
N_VectorContent_ParHyp nv_c = (N_VectorContent_ParHyp)(nv->content);
MFEM_ASSERT(nv_c->own_parvector == SUNFALSE, "invalid N_Vector");
nv_c->local_length = x->local_vector->size;
nv_c->global_length = x->global_size;
nv_c->comm = x->comm;
nv_c->x = x;
return N_VMake_Parallel(GetComm(), Size(), GlobalSize(), GetData());
}
#endif // MFEM_USE_SUNDIALS
+2 -9
View File
@@ -34,9 +34,6 @@
#include "sparsemat.hpp"
#include "hypre_parcsr.hpp"
#ifdef MFEM_USE_SUNDIALS
#include <nvector/nvector_parhyp.h>
#endif
namespace mfem
{
@@ -163,13 +160,9 @@ public:
~HypreParVector();
#ifdef MFEM_USE_SUNDIALS
/// Return a new wrapper SUNDIALS N_Vector of type SUNDIALS_NVEC_PARHYP.
/// Return a new wrapper SUNDIALS N_Vector of type SUNDIALS_NVEC_PARALLEL.
/** The returned N_Vector must be destroyed by the caller. */
virtual N_Vector ToNVector() { return N_VMake_ParHyp(x); }
/** @brief Update an existing wrapper SUNDIALS N_Vector of type
SUNDIALS_NVEC_PARHYP to point to this Vector. */
virtual void ToNVector(N_Vector &nv);
virtual N_Vector ToNVector();
#endif
};
+1 -1
View File
@@ -18,7 +18,7 @@
#endif
#include "../config/config.hpp"
#include "../general/cuda.hpp"
#include "../general/backends.hpp"
#include "../general/globals.hpp"
#include "matrix.hpp"
-7
View File
@@ -28,13 +28,6 @@ class Matrix : public Operator
{
friend class MatrixInverse;
public:
/// Defines matrix diagonal policy upon elimination of rows and/or columns.
enum DiagonalPolicy
{
DIAG_ZERO, ///< Set the diagonal value to zero
DIAG_ONE, ///< Set the diagonal value to one
DIAG_KEEP ///< Keep the diagonal value
};
/// Creates a square matrix of size s.
explicit Matrix(int s) : Operator(s) { }
+110 -8
View File
@@ -346,7 +346,6 @@ const double RK8Solver::c[] =
AdamsBashforthSolver::AdamsBashforthSolver(int _s, const double *_a)
{
s = 0;
smax = std::min(_s,5);
a = _a;
k = new Vector[5];
@@ -365,6 +364,34 @@ AdamsBashforthSolver::AdamsBashforthSolver(int _s, const double *_a)
}
}
void AdamsBashforthSolver::GetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i >= 0) && ( i < s ),
" AdamsBashforthSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
state = k[idx[i]];
}
const Vector &AdamsBashforthSolver::GetStateVector(int i)
{
MFEM_ASSERT( (i >= 0) && ( i < s ),
" AdamsBashforthSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
return k[idx[i]];
}
void AdamsBashforthSolver::SetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i >= 0) && ( i < smax ),
" AdamsBashforthSolver::SetStateVector \n" <<
" - Tried to set non-existent state "<<i);
k[idx[i]] = state;
s = std::max(i,s);
}
void AdamsBashforthSolver::Init(TimeDependentOperator &_f)
{
ODESolver::Init(_f);
@@ -430,6 +457,31 @@ AdamsMoultonSolver::AdamsMoultonSolver(int _s, const double *_a)
}
}
const Vector &AdamsMoultonSolver::GetStateVector(int i)
{
MFEM_ASSERT( (i >= 0) && ( i < s ),
" AdamsMoultonSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
return k[idx[i+1]];
}
void AdamsMoultonSolver::GetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i >= 0) && ( i < s ),
" AdamsMoultonSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
state = k[idx[i+1]];
}
void AdamsMoultonSolver::SetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i >= 0) && ( i < smax ),
" AdamsMoultonSolver::SetStateVector \n" <<
" - Tried to set non-existent state "<<i);
k[idx[i+1]] = state;
s = std::max(i,s);
}
void AdamsMoultonSolver::Init(TimeDependentOperator &_f)
{
ODESolver::Init(_f);
@@ -641,7 +693,32 @@ void GeneralizedAlphaSolver::Init(TimeDependentOperator &_f)
y.SetSize(f->Width(), mem_type);
xdot.SetSize(f->Width(), mem_type);
xdot = 0.0;
first = true;
nstate = 0;
}
const Vector &GeneralizedAlphaSolver::GetStateVector(int i)
{
MFEM_ASSERT( (i == 0) && (nstate == 1),
"GeneralizedAlphaSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
return xdot;
}
void GeneralizedAlphaSolver::GetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i == 0) && (nstate == 1),
"GeneralizedAlphaSolver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
state = xdot;
}
void GeneralizedAlphaSolver::SetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i == 0),
"GeneralizedAlphaSolver::SetStateVector \n" <<
" - Tried to set non-existent state "<<i);
xdot = state;
nstate = 1;
}
void GeneralizedAlphaSolver::SetRhoInf(double rho_inf)
@@ -684,10 +761,10 @@ void GeneralizedAlphaSolver::PrintProperties(std::ostream &out)
// This routine assumes xdot is initialized.
void GeneralizedAlphaSolver::Step(Vector &x, double &t, double &dt)
{
if (first)
if (nstate == 0)
{
f->Mult(x,xdot);
first = false;
nstate = 1;
}
// Set y = x + alpha_f*(1.0 - (gamma/alpha_m))*dt*xdot
@@ -892,7 +969,33 @@ void GeneralizedAlpha2Solver::Init(SecondOrderTimeDependentOperator &_f)
aa.SetSize(f->Width());
d2xdt2.SetSize(f->Width());
d2xdt2 = 0.0;
first = true;
nstate = 0;
}
const Vector &GeneralizedAlpha2Solver::GetStateVector(int i)
{
MFEM_ASSERT( (i == 0) && (nstate == 1),
"GeneralizedAlpha2Solver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
return d2xdt2;
}
void GeneralizedAlpha2Solver::GetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i == 0) && (nstate == 1),
"GeneralizedAlpha2Solver::GetStateVector \n" <<
" - Tried to get non-existent state "<<i);
state = d2xdt2;
}
void GeneralizedAlpha2Solver::SetStateVector(int i, Vector &state)
{
MFEM_ASSERT( (i == 0),
"GeneralizedAlpha2Solver::SetStateVector \n" <<
" - Tried to set non-existent state "<<i);
d2xdt2 = state;
nstate = 1;
}
void GeneralizedAlpha2Solver::PrintProperties(std::ostream &out)
@@ -935,13 +1038,12 @@ void GeneralizedAlpha2Solver::Step(Vector &x, Vector &dxdt,
double fac5 = alpha_m;
// In the first pass compute d2xdt2 directy from operator.
if (first)
if (nstate == 0)
{
f->Mult(x, dxdt, d2xdt2);
first = false;
nstate = 1;
}
// Predict alpha levels
add(dxdt, fac0*dt, d2xdt2, va);
add(x, fac1*dt, va, xa);
+95 -39
View File
@@ -91,6 +91,23 @@ public:
while (t < tf) { Step(x, t, dt); }
}
/// Function for getting and setting the state vectors
virtual int GetMaxStateSize() { return 0; }
virtual int GetStateSize() { return 0; }
virtual const Vector &GetStateVector(int i)
{
mfem_error("ODESolver has no state vectors");
Vector *s = NULL; return *s; // Make some compiler happy
}
virtual void GetStateVector(int i, Vector &state)
{
mfem_error("ODESolver has no state vectors");
}
virtual void SetStateVector(int i, Vector &state)
{
mfem_error("ODESolver has no state vectors");
}
virtual ~ODESolver() { }
};
@@ -102,9 +119,9 @@ private:
Vector dxdt;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -122,9 +139,9 @@ private:
public:
RK2Solver(const double _a = 2./3.) : a(_a) { }
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -135,9 +152,9 @@ private:
Vector y, k;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -148,9 +165,9 @@ private:
Vector y, k, z;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -174,9 +191,9 @@ public:
ExplicitRKSolver(int _s, const double *_a, const double *_b,
const double *_c);
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
virtual ~ExplicitRKSolver();
};
@@ -219,9 +236,15 @@ private:
public:
AdamsBashforthSolver(int _s, const double *_a);
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
int GetMaxStateSize() override { return smax; };
int GetStateSize() override { return s; };
const Vector &GetStateVector(int i) override;
void GetStateVector(int i, Vector &state) override;
void SetStateVector(int i, Vector &state) override;
~AdamsBashforthSolver()
{
@@ -294,9 +317,15 @@ private:
public:
AdamsMoultonSolver(int _s, const double *_a);
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
int GetMaxStateSize() override { return smax-1; };
int GetStateSize() override { return s-1; };
const Vector &GetStateVector(int i) override;
void GetStateVector(int i, Vector &state) override;
void SetStateVector(int i, Vector &state) override;
~AdamsMoultonSolver()
{
@@ -363,9 +392,9 @@ protected:
Vector k;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -376,9 +405,9 @@ protected:
Vector k;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -397,9 +426,9 @@ protected:
public:
SDIRK23Solver(int gamma_opt = 1);
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -411,9 +440,9 @@ protected:
Vector k, y, z;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -425,9 +454,9 @@ protected:
Vector k, y;
public:
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
};
@@ -439,7 +468,7 @@ class GeneralizedAlphaSolver : public ODESolver
protected:
mutable Vector xdot,k,y;
double alpha_f, alpha_m, gamma;
bool first;
int nstate;
void SetRhoInf(double rho_inf);
void PrintProperties(std::ostream &out = mfem::out);
@@ -447,9 +476,15 @@ public:
GeneralizedAlphaSolver(double rho = 1.0) { SetRhoInf(rho); };
virtual void Init(TimeDependentOperator &_f);
void Init(TimeDependentOperator &_f) override;
virtual void Step(Vector &x, double &t, double &dt);
void Step(Vector &x, double &t, double &dt) override;
int GetMaxStateSize() override { return 1; };
int GetStateSize() override { return nstate; };
const Vector &GetStateVector(int i) override;
void GetStateVector(int i, Vector &state) override;
void SetStateVector(int i, Vector &state) override;
};
@@ -497,7 +532,7 @@ class SIA1Solver : public SIASolver
{
public:
SIA1Solver() {}
void Step(Vector &q, Vector &p, double &t, double &dt);
void Step(Vector &q, Vector &p, double &t, double &dt) override;
};
/// Second Order Symplectic Integration Algorithm
@@ -505,7 +540,7 @@ class SIA2Solver : public SIASolver
{
public:
SIA2Solver() {}
void Step(Vector &q, Vector &p, double &t, double &dt);
void Step(Vector &q, Vector &p, double &t, double &dt) override;
};
/// Variable order Symplectic Integration Algorithm (orders 1-4)
@@ -513,7 +548,7 @@ class SIAVSolver : public SIASolver
{
public:
SIAVSolver(int order);
void Step(Vector &q, Vector &p, double &t, double &dt);
void Step(Vector &q, Vector &p, double &t, double &dt) override;
private:
int order_;
@@ -606,11 +641,26 @@ public:
while (t < tf) { Step(x, dxdt, t, dt); }
}
/// Function for getting and setting the state vectors
virtual int GetMaxStateSize() { return 0; };
virtual int GetStateSize() { return 0; }
virtual const Vector &GetStateVector(int i)
{
mfem_error("ODESolver has no state vectors");
Vector *s = NULL; return *s; // Make some compiler happy
}
virtual void GetStateVector(int i, Vector &state)
{
mfem_error("ODESolver has no state vectors");
}
virtual void SetStateVector(int i, Vector &state)
{
mfem_error("ODESolver has no state vectors");
}
virtual ~SecondOrderODESolver() { }
};
/// The classical newmark method.
/// Newmark, N. M. (1959) A method of computation for structural dynamics.
/// Journal of Engineering Mechanics, ASCE, 85 (EM3) 67-94.
@@ -625,11 +675,11 @@ private:
public:
NewmarkSolver(double beta_ = 0.25, double gamma_ = 0.5) { beta = beta_; gamma = gamma_; };
virtual void PrintProperties(std::ostream &out = mfem::out);
void PrintProperties(std::ostream &out = mfem::out);
virtual void Init(SecondOrderTimeDependentOperator &_f);
void Init(SecondOrderTimeDependentOperator &_f) override;
virtual void Step(Vector &x, Vector &dxdt, double &t, double &dt);
void Step(Vector &x, Vector &dxdt, double &t, double &dt) override;
};
class LinearAccelerationSolver : public NewmarkSolver
@@ -661,7 +711,7 @@ class GeneralizedAlpha2Solver : public SecondOrderODESolver
protected:
Vector xa,va,aa,d2xdt2;
double alpha_f, alpha_m, beta, gamma;
bool first;
int nstate;
public:
GeneralizedAlpha2Solver(double rho_inf = 1.0)
@@ -675,11 +725,17 @@ public:
gamma = 0.5 + alpha_m - alpha_f;
};
virtual void PrintProperties(std::ostream &out = mfem::out);
void PrintProperties(std::ostream &out = mfem::out);
virtual void Init(SecondOrderTimeDependentOperator &_f);
void Init(SecondOrderTimeDependentOperator &_f) override;
virtual void Step(Vector &x, Vector &dxdt, double &t, double &dt);
void Step(Vector &x, Vector &dxdt, double &t, double &dt) override;
int GetMaxStateSize() override { return 1; };
int GetStateSize() override { return nstate; };
const Vector &GetStateVector(int i) override;
void GetStateVector(int i, Vector &state) override;
void SetStateVector(int i, Vector &state) override;
};
/// The classical midpoint method.
+27 -6
View File
@@ -403,8 +403,10 @@ TripleProductOperator::~TripleProductOperator()
ConstrainedOperator::ConstrainedOperator(Operator *A, const Array<int> &list,
bool _own_A)
: Operator(A->Height(), A->Width()), A(A), own_A(_own_A)
bool _own_A,
DiagonalPolicy _diag_policy)
: Operator(A->Height(), A->Width()), A(A), own_A(_own_A),
diag_policy(_diag_policy)
{
// 'mem_class' should work with A->Mult() and MFEM_FORALL():
mem_class = A->GetMemoryClass()*Device::GetDeviceMemoryClass();
@@ -464,11 +466,30 @@ void ConstrainedOperator::Mult(const Vector &x, Vector &y) const
auto d_x = x.Read();
// Use read+write access - we are modifying sub-vector of y
auto d_y = y.ReadWrite();
MFEM_FORALL(i, csz,
switch (diag_policy)
{
const int id = idx[i];
d_y[id] = d_x[id];
});
case DIAG_ONE:
MFEM_FORALL(i, csz,
{
const int id = idx[i];
d_y[id] = d_x[id];
});
break;
case DIAG_ZERO:
MFEM_FORALL(i, csz,
{
const int id = idx[i];
d_y[id] = 0.0;
});
break;
case DIAG_KEEP:
// Needs action of the operator diagonal on vector
mfem_error("ConstrainedOperator::Mult #1");
break;
default:
mfem_error("ConstrainedOperator::Mult #2");
break;
}
}
RectangularConstrainedOperator::RectangularConstrainedOperator(
+144 -3
View File
@@ -41,6 +41,14 @@ protected:
Operator *SetupRAP(const Operator *Pi, const Operator *Po);
public:
/// Defines operator diagonal policy upon elimination of rows and/or columns.
enum DiagonalPolicy
{
DIAG_ZERO, ///< Set the diagonal value to zero
DIAG_ONE, ///< Set the diagonal value to one
DIAG_KEEP ///< Keep the diagonal value
};
/// Initializes memory for true vectors of linear system
void InitTVectors(const Operator *Po, const Operator *Ri, const Operator *Pi,
Vector &x, Vector &b,
@@ -442,6 +450,132 @@ public:
virtual ~TimeDependentOperator() { }
};
/** TimeDependentAdjointOperator is a TimeDependentOperator with Adjoint rate
equations to be used with CVODESSolver. */
class TimeDependentAdjointOperator : public TimeDependentOperator
{
public:
/**
\brief The TimedependentAdjointOperator extends the TimeDependentOperator
class to use features in SUNDIALS CVODESSolver for computing quadratures
and solving adjoint problems.
To solve adjoint problems one needs to implement the AdjointRateMult
method to tell CVODES what the adjoint rate equation is.
QuadratureIntegration (optional) can be used to compute values over the
forward problem
QuadratureSensitivityMult (optional) can be used to find the sensitivity
of the quadrature using the adjoint solution in part.
SUNImplicitSetupB (optional) can be used to setup custom solvers for the
newton solve for the adjoint problem.
SUNImplicitSolveB (optional) actually uses the solvers from
SUNImplicitSetupB to solve the adjoint problem.
See SUNDIALS user manuals for specifics.
\param[in] dim Dimension of the forward operator
\param[in] adjdim Dimension of the adjoint operator. Typically it is the
same size as dim. However, SUNDIALS allows users to specify the size if
one wants to perform custom operations.
\param[in] t Starting time to set
\param[in] type The TimeDependentOperator type
*/
TimeDependentAdjointOperator(int dim, int adjdim, double t = 0.,
Type type = EXPLICIT) :
TimeDependentOperator(dim, t, type),
adjoint_height(adjdim)
{}
/// Destructor
virtual ~TimeDependentAdjointOperator() {};
/**
\brief Provide the operator integration of a quadrature equation
\param[in] y The current value at time t
\param[out] qdot The current quadrature rate value at t
*/
virtual void QuadratureIntegration(const Vector &y, Vector &qdot) const {};
/** @brief Perform the action of the operator:
@a yBdot = k = f(@a y,@2 yB, t), where
@param[in] y The primal solution at time t
@param[in] yB The adjoint solution at time t
@param[out] yBdot the rate at time t
*/
virtual void AdjointRateMult(const Vector &y, Vector & yB,
Vector &yBdot) const = 0;
/**
\brief Provides the sensitivity of the quadrature w.r.t to primal and
adjoint solutions
\param[in] y the value of the primal solution at time t
\param[in] yB the value of the adjoint solution at time t
\param[out] qBdot the value of the sensitivity of the quadrature rate at
time t
*/
virtual void QuadratureSensitivityMult(const Vector &y, const Vector &yB,
Vector &qBdot) const {}
/** @brief Setup the ODE linear system \f$ A(x,t) = (I - gamma J) \f$ or
\f$ A = (M - gamma J) \f$, where \f$ J(x,t) = \frac{df}{dt(x,t)} \f$.
@param[in] t The current time
@param[in] x The state at which \f$A(x,xB,t)\f$ should be evaluated.
@param[in] xB The state at which \f$A(x,xB,t)\f$ should be evaluated.
@param[in] fxB The current value of the ODE rhs function, \f$f(x,t)\f$.
@param[in] jokB Flag indicating if the Jacobian should be updated.
@param[out] jcurB Flag to signal if the Jacobian was updated.
@param[in] gammaB The scaled time step value.
If not re-implemented, this method simply generates an error.
Presently, this method is used by SUNDIALS ODE solvers, for more details,
see the SUNDIALS User Guides.
*/
virtual int SUNImplicitSetupB(const double t, const Vector &x,
const Vector &xB, const Vector &fxB,
int jokB, int *jcurB, double gammaB)
{
mfem_error("TimeDependentAdjointOperator::SUNImplicitSetupB() is not "
"overridden!");
return (-1);
}
/** @brief Solve the ODE linear system \f$ A(x,xB,t) xB = b \f$ as setup by
the method SUNImplicitSetup().
@param[in] b The linear system right-hand side.
@param[in,out] x On input, the initial guess. On output, the solution.
@param[in] tol Linear solve tolerance.
If not re-implemented, this method simply generates an error.
Presently, this method is used by SUNDIALS ODE solvers, for more details,
see the SUNDIALS User Guides. */
virtual int SUNImplicitSolveB(Vector &x, const Vector &b, double tol)
{
mfem_error("TimeDependentAdjointOperator::SUNImplicitSolveB() is not "
"overridden!");
return (-1);
}
/// Returns the size of the adjoint problem state space
int GetAdjointHeight() {return adjoint_height;}
protected:
int adjoint_height; /// Size of the adjoint problem
};
/// Base abstract class for second order time dependent operators.
/** Operator of the form: (x,dxdt,t) -> f(x,dxdt,t), where k = f(x,dxdt,t)
generally solves the algebraic equation F(x,dxdt,k,t) = G(x,dxdt,t).
@@ -671,20 +805,27 @@ protected:
bool own_A; ///< Ownership flag for A.
mutable Vector z, w; ///< Auxiliary vectors.
MemoryClass mem_class;
DiagonalPolicy diag_policy; ///< Diagonal policy for constrained dofs
public:
/** @brief Constructor from a general Operator and a list of essential
indices/dofs.
Specify the unconstrained operator @a *A and a @a list of indices to
constrain, i.e. each entry @a list[i] represents an essential-dof. If the
constrain, i.e. each entry @a list[i] represents an essential dof. If the
ownership flag @a own_A is true, the operator @a *A will be destroyed
when this object is destroyed. */
ConstrainedOperator(Operator *A, const Array<int> &list, bool own_A = false);
when this object is destroyed. The @a diag_policy determines how the
operator sets entries corresponding to essential dofs. */
ConstrainedOperator(Operator *A, const Array<int> &list, bool own_A = false,
DiagonalPolicy diag_policy = DIAG_ONE);
/// Returns the type of memory in which the solution and temporaries are stored.
virtual MemoryClass GetMemoryClass() const { return mem_class; }
/// Set the diagonal policy for the constrained operator.
void SetDiagonalPolicy(const DiagonalPolicy _diag_policy)
{ diag_policy = _diag_policy; }
/** @brief Eliminate "essential boundary condition" values specified in @a x
from the given right-hand side @a b.
+18 -4
View File
@@ -134,7 +134,7 @@ OperatorJacobiSmoother::OperatorJacobiSmoother(const BilinearForm &a,
OperatorJacobiSmoother::OperatorJacobiSmoother(const Vector &d,
const Array<int> &ess_tdofs,
const double dmpng)
const double dmpng, const bool inverse)
:
Solver(d.Size()),
N(d.Size()),
@@ -143,16 +143,30 @@ OperatorJacobiSmoother::OperatorJacobiSmoother(const Vector &d,
ess_tdof_list(ess_tdofs),
residual(N)
{
Setup(d);
Setup(d, inverse);
}
void OperatorJacobiSmoother::Setup(const Vector &diag)
void OperatorJacobiSmoother::Setup(const Vector &diag, const bool inverse)
{
residual.UseDevice(true);
const double delta = damping;
auto D = diag.Read();
auto DI = dinv.Write();
MFEM_FORALL(i, N, DI[i] = delta / D[i]; );
if (inverse)
{
if (delta > 0.0)
{
MFEM_FORALL(i, N, DI[i] = delta * D[i]; );
}
else
{
MFEM_FORALL(i, N, DI[i] = D[i]; );
}
}
else
{
MFEM_FORALL(i, N, DI[i] = delta / D[i]; );
}
auto I = ess_tdof_list.Read();
MFEM_FORALL(i, ess_tdof_list.Size(), DI[I[i]] = delta; );
}
+8 -4
View File
@@ -114,23 +114,25 @@ public:
/** Setup a Jacobi smoother with the diagonal of @a a obtained by calling
a.AssembleDiagonal(). It is assumed that the underlying operator acts as
the identity on entries in ess_tdof_list, corresponding to (assembled)
DIAG_ONE policy or ConstratinedOperator in the matrix-free setting. */
DIAG_ONE policy or ConstrainedOperator in the matrix-free setting. */
OperatorJacobiSmoother(const BilinearForm &a,
const Array<int> &ess_tdof_list,
const double damping=1.0);
/** Application is by the *inverse* of the given vector. It is assumed that
the underlying operator acts as the identity on entries in ess_tdof_list,
corresponding to (assembled) DIAG_ONE policy or ConstratinedOperator in
corresponding to (assembled) DIAG_ONE policy or ConstrainedOperator in
the matrix-free setting. */
OperatorJacobiSmoother(const Vector &d,
const Array<int> &ess_tdof_list,
const double damping=1.0);
const double damping=1.0,
const bool inverse=false);
~OperatorJacobiSmoother() {}
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const { Mult(x, y); }
void SetOperator(const Operator &op) { oper = &op; }
void Setup(const Vector &diag);
void Setup(const Vector &diag, const bool inverse=false);
private:
const int N;
@@ -181,6 +183,8 @@ public:
void Mult(const Vector&x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const { Mult(x, y); }
void SetOperator(const Operator &op_)
{
oper = &op_;
+44 -20
View File
@@ -494,24 +494,29 @@ void SparseMatrix::GetDiag(Vector & d) const
d.SetSize(height);
int j, end;
for (int i = 0; i < height; i++)
{
auto I = this->ReadI();
auto J = this->ReadJ();
auto A = this->ReadData();
auto dd = d.Write();
end = I[i+1];
for (j = I[i]; j < end; j++)
MFEM_FORALL(i, height,
{
const int begin = I[i];
const int end = I[i+1];
int j;
for (j = begin; j < end; j++)
{
if (J[j] == i)
{
d[i] = A[j];
dd[i] = A[j];
break;
}
}
if (j == end)
{
d[i] = 0.;
dd[i] = 0.;
}
}
});
}
/// Produces a DenseMatrix from a SparseMatrix
@@ -2145,31 +2150,46 @@ void SparseMatrix::DiagScale(const Vector &b, Vector &x, double sc) const
{
MFEM_VERIFY(Finalized(), "Matrix must be finalized.");
const int nnz = J.Capacity();
const bool use_dev = b.UseDevice() || x.UseDevice();
auto bp = b.Read(use_dev);
auto xp = x.Write(use_dev);
auto Ap = Read(A, nnz);
auto Ip = Read(I, height+1);
auto Jp = Read(J, nnz);
bool scale = (sc != 1.0);
for (int i = 0, j = 0; i < height; i++)
MFEM_FORALL(i, height,
{
int end = I[i+1];
for ( ; true; j++)
int end = Ip[i+1];
for (int j = Ip[i]; true; j++)
{
MFEM_VERIFY(j != end, "Couldn't find diagonal in row. i = " << i
<< ", j = " << j
<< ", I[i+1] = " << end );
if (J[j] == i)
if (j == end)
{
MFEM_VERIFY(std::abs(A[j]) > 0.0, "Diagonal " << j << " must be nonzero");
MFEM_ABORT_KERNEL("Diagonal not found in SparseMatrix::DiagScale");
}
if (Jp[j] == i)
{
if (!(std::abs(Ap[j]) > 0.0))
{
MFEM_ABORT_KERNEL("Zero diagonal in SparseMatrix::DiagScale");
}
if (scale)
{
x(i) = sc * b(i) / A[j];
xp[i] = sc * bp[i] / Ap[j];
}
else
{
x(i) = b(i) / A[j];
xp[i] = bp[i] / Ap[j];
}
break;
}
}
j = end;
}
});
return;
}
@@ -2749,6 +2769,10 @@ void SparseMatrix::Print(std::ostream & out, int _width) const
return;
}
// HostRead forces synchronization
HostReadI();
HostReadJ();
HostReadData();
for (i = 0; i < height; i++)
{
out << "[row " << i << "]\n";
+6
View File
@@ -578,6 +578,12 @@ public:
Type GetType() const { return MFEM_SPARSEMAT; }
};
inline std::ostream& operator<<(std::ostream& os, SparseMatrix const& mat)
{
mat.Print(os);
return os;
}
/// Applies f() to each element of the matrix (after it is finalized).
void SparseMatrixFunction(SparseMatrix &S, double (*f)(double));

Some files were not shown because too many files have changed in this diff Show More