Compare commits

...
541 Commits
Author SHA1 Message Date
dylan-copeland 0cf995cfc7 Fix single precision build. 2025-12-12 18:29:30 -08:00
dylan-copeland c9535ee6fc Templates for some reducers. 2025-12-12 17:09:03 -08:00
dylan-copeland 03766e06be More templates. 2025-12-12 13:57:12 -08:00
dylan-copeland 9a96ed3cb1 Adding some missing templates. 2025-12-12 13:25:10 -08:00
Dylan Copeland 16afb15588 Merge branch 'master' of github.com:mfem/mfem into mp 2025-12-12 12:35:54 -08:00
Dylan Copeland e480d543eb Merge branch 'master' of github.com:mfem/mfem into mp 2025-12-12 12:34:20 -08:00
Tzanio Kolev fd9c5fbfc1 Merge pull request #5152 from mfem/new-dev-version-4.9.1
Update version numbers to 4.9.1 -- a new development version
2025-12-12 11:13:49 -08:00
Veselin Dobrev b5c712dc80 Update version numbers to 4.9.1 -- a new development version 2025-12-12 10:03:11 -08:00
Tzanio Kolev d9d6526cc1 Merge pull request #5057 from mfem/mfem-4.9-dev
Final changes for mfem-4.9
2025-12-11 10:59:53 -08:00
Tzanio Kolev c7515017c8 minor 2025-12-11 10:43:08 -08:00
Tzanio Kolev d0e4d9a242 minor 2025-12-11 10:40:28 -08:00
Tzanio Kolev eb7a9abc46 Update doxygen for mfem-4.9 2025-12-11 10:39:02 -08:00
Veselin Dobrev 3f7e5a3aa4 Remove a FIXME that was addressed 2025-12-11 10:21:41 -08:00
camierjs c328a0f635 CHANGELOG minor typo 2025-12-11 09:57:25 -08:00
Tzanio Kolev f5632ae914 minor 2025-12-11 09:54:03 -08:00
Tzanio Kolev c46fff561b Final CHANGELOG update before the mfem-4.9 release 2025-12-11 09:36:07 -08:00
Will Pazner 9cc9765e15 Add missing Doxygen comment 2025-12-10 21:23:05 -08:00
camierjs ae72ee8016 ParticleSet Fallback auto return type 2025-12-10 10:55:54 -08:00
Veselin Dobrev 091789d078 In ex41/ex41p, update the default time step to be the same as in ex9/ex9p
and remove a sample run that is now the same as the default.

Fix a -Wextra warning in mesh_readers.cpp.
2025-12-09 18:48:15 -08:00
camierjs b534c77ce3 - Adding -tf 1.0 to ex41p* test command
- Fix std::pow to int cast warning
2025-12-09 17:44:01 -08:00
camierjs 87ca8b3560 Reduce ex41 t_final downto 1.0 2025-12-09 17:10:04 -08:00
camierjs 387005e37a with style 2025-12-09 15:25:57 -08:00
camierjs 1d81f19c37 minor argument type safety casts 2025-12-09 15:16:11 -08:00
Jan Nikl bb26cc73d8 Minor fix of signedness. 2025-12-09 14:30:06 -08:00
camierjs dd11be5d6b Rename UTF8 MTOP variables to ASCII ones 2025-12-09 14:14:23 -08:00
Veselin Dobrev ecc2490ef9 Fix in DenseTensor::MemoryUsage() 2025-12-09 11:12:03 -08:00
camierjs b02842265b Silence some 'declared but never referenced' TMOP specialization warnings 2025-12-09 10:51:56 -08:00
Tzanio Kolev 9ad4e1258f Merge branch 'master' into mfem-4.9-dev 2025-12-08 15:02:14 -08:00
Tzanio Kolev 733f8f231b Merge pull request #4977 from mfem/imex-conv-diff-dg
IMEX Implementation with DG Convection-Diffusion example [imex-conv-diff-dg]
2025-12-08 14:59:41 -08:00
Jan Nikl 7f71efd74a Update CHANGELOG for #5097 2025-12-08 10:32:14 -08:00
blaz 822d6dcf01 remove Bcast to avoid confusions 2025-12-07 13:59:08 -08:00
blaz 0d70d354a5 style 2025-12-06 19:41:07 -08:00
blaz 9e35901815 MPI_BOOL 2025-12-06 19:39:13 -08:00
blaz 14b4c7b4d7 Bcast 2025-12-05 23:54:46 -08:00
blaz 8be02504cb adding message for changing the time step 2025-12-05 22:41:27 -08:00
blaz 9025fdec27 avoids different behaviour between processes 2025-12-05 22:36:05 -08:00
blaz 21d1bc4a8b modified limits 2025-12-05 20:23:28 -08:00
blaz 26a832b81f style 2025-12-05 20:18:03 -08:00
Tzanio Kolev f9f83882ce Merge pull request #4986 from mfem/particleset-particle-dev
Particle Tracking (`ParticleSet`)
2025-12-05 20:17:08 -08:00
blaz e1efc3b437 replaced direct comparison of real numbers 2025-12-05 20:16:02 -08:00
Veselin Dobrev 238523b52b Some additional edits mostly related to moving the 'navier' directory
to the 'fluids' sub-directory.
2025-12-05 15:06:54 -08:00
Veselin Dobrev a70350df55 Merge branch 'master' into particleset-particle-dev 2025-12-05 14:36:34 -08:00
Tzanio Kolev 903e11aee4 Merge pull request #5097 from mfem/task/2025_10_dc_quad_points
Add quadrature func support to VisIt and Conduit data collections
2025-12-05 13:58:12 -08:00
Tzanio Kolev 1c952e3c1f Merge pull request #5059 from mfem/isf
Incompressible Schrödinger Flow
2025-12-05 12:05:35 -08:00
John Camier b2d8ef9e2a Merge branch 'master' into isf 2025-12-05 09:58:29 -08:00
Veselin Dobrev 536a3c50be In miniapps/navier/makefile, fix the shared build 2025-12-05 09:31:55 -08:00
Will Pazner b3eaff280f Merge pull request #5138 from mfem/amgf-mixedint
fix mixed-int bug
2025-12-05 09:27:11 -08:00
John Camier 9ead7ddcc7 Merge branch 'master' into isf 2025-12-05 07:53:17 -08:00
Veselin Dobrev fca40e0662 Merge pull request #5083 from mfem/dg-diffusion-pa-matrix-coeff
Add support for PA DG diffusion with matrix coefficients
2025-12-04 23:21:23 -08:00
camierjs c3e6837706 Merge branch 'master' into isf 2025-12-04 21:39:04 -08:00
blaz b20015c70d fix for problem 3 2025-12-04 21:03:27 -08:00
Ketan Mittal 6a376e1cfd Merge branch 'master' into particleset-particle-dev 2025-12-04 19:36:51 -08:00
Joseph Signorelli 4691c4d751 Merge branch 'particleset-particle-dev' of github.com:mfem/mfem into particleset-particle-dev 2025-12-04 21:36:32 -06:00
Joseph Signorelli 01a5f979a0 fix mistake 2025-12-04 21:36:26 -06:00
Mittal, Ketan 39226e2520 remove some leftover comment 2025-12-04 19:36:22 -08:00
Joseph Signorelli d514d0578c Update CHANGELOG 2025-12-04 21:31:15 -06:00
Socratis Petrides ebe07e7ddb fix mixed-int bug 2025-12-04 18:30:21 -08:00
Tzanio Kolev b6cd380106 Merge pull request #5115 from mfem/dfem-pa-caching
dFEM PA caching
2025-12-04 17:26:39 -08:00
camierjs 3803c3305e - Remove unnecessary general/forall.hpp include in miniapps and tests
- Fix cast warnings in mesh (nc)nurbs files
2025-12-04 16:49:09 -08:00
Veselin Dobrev 88f9dc96db Changes to support single precision MFEM build in ConduitDataCollection 2025-12-04 16:36:01 -08:00
Joseph Signorelli 08b3639eca fix typo 2025-12-04 18:17:19 -06:00
Joseph Signorelli 306702b4ab characteristic --> property 2025-12-04 18:15:55 -06:00
Mittal, Ketan 93ee0e31b3 add * to the sample run 2025-12-04 15:51:59 -08:00
Mittal, Ketan 9037765299 minor changes to miniapp description 2025-12-04 15:50:25 -08:00
Joseph Signorelli d0a53c205c Add note stating GSLIB requirement for particle inclusion 2025-12-04 17:38:54 -06:00
Joseph Signorelli 1eb69ebe21 Add description for Navier Bifurcation 2025-12-04 17:29:39 -06:00
Tzanio KolevandChris Vogl c96f0d3fe6 Update CHANGELOG
Co-authored-by: Chris Vogl <vogl2@llnl.gov>
2025-12-04 15:21:51 -08:00
Veselin Dobrev 0a4cce7bd5 Fix some compiler warnings from -Wshadow and -Wextra 2025-12-04 14:53:50 -08:00
Joseph Signorelli dc6e90914f minor 2025-12-04 15:34:21 -06:00
Joseph Signorelli abd828804e minor doc fix 2025-12-04 15:34:15 -06:00
Joseph Signorelli 89d5b3a95d address reviewer comments 2025-12-04 11:52:07 -06:00
Joseph SignorelliandJan Nikl 6f7f169b7e Update fem/particleset.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-12-04 07:45:59 -06:00
Veselin Dobrev e2f5504b6c Remove ex41p from the lists of examples with device support 2025-12-03 17:44:14 -08:00
camierjs 48de40c399 Added general/forall header in top MFEM one 2025-12-03 17:18:36 -08:00
Julian Andrej 93856df95f index bug 2025-12-03 14:35:11 -08:00
Joseph Signorelli 0bf6682aa0 minor 2025-12-03 15:52:22 -06:00
Joseph Signorelli 4dd85be8e3 add warning for invalidation of Particle ref 2025-12-03 15:49:25 -06:00
Julian Andrej c299adb3ff fix derivative indexing 2025-12-03 13:41:08 -08:00
Julian Andrej 2008738c1f more docs 2025-12-03 13:21:26 -08:00
Julian Andrej 798141da18 add more docs 2025-12-03 13:13:33 -08:00
Julian Andrej aa03f968ef switch to _THREAD_DIRECT 2025-12-03 12:59:38 -08:00
Julian Andrej a7de138602 Merge branch 'master' into dfem-pa-caching 2025-12-03 12:58:51 -08:00
Joseph Signorelli edd6f10eb4 style 2025-12-03 14:46:50 -06:00
Joseph Signorelli e484a01593 Address reviewer comments
- IDs --> global IDs
- AddNamedField docs update
- Verify particle in SetParticle
- Space in Particle_2 fields/tags added
2025-12-03 14:45:16 -06:00
Mittal, Ketan 7b7cda633d minor doc fix 2025-12-03 11:02:39 -08:00
Mittal, Ketan f0c9e15cf3 update doxygen documentation for sample particle data 2025-12-03 10:59:28 -08:00
Mittal, Ketan 111569288f move example particle documentation inside doxygen 2025-12-02 18:47:26 -08:00
Mittal, Ketan bbe243ad54 minor 2025-12-02 18:25:17 -08:00
camierjs 87b701bfd1 Merge branch 'master' into isf 2025-12-02 17:46:20 -08:00
camierjs 674c982be6 Serial out of source builds still needs relative general/forall.hpp 2025-12-02 17:46:08 -08:00
Ketan Mittal 66d2bbe977 Merge branch 'master' into particleset-particle-dev 2025-12-02 17:06:47 -08:00
Ketan MittalandJan Nikl 0274f2b6fa Apply suggestions from code review
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-12-02 17:06:30 -08:00
Mittal, Ketan c31f303ed2 documentation 2025-12-02 17:05:29 -08:00
Tzanio Kolev 9708cc522e Merge pull request #5135 from mfem/gitlab-ci-extra-time
Gitlab CI: allocate more time for the dane-baseline pipeline
2025-12-02 12:49:54 -08:00
bslazarov ebdd44d5bd comment modification 2025-12-02 11:46:00 -08:00
bslazarov 48976d3856 changelog 2025-12-02 11:42:16 -08:00
camierjs 8ec27e5adf Fix cmake list ordering 2025-12-02 11:39:21 -08:00
camierjs e5dea3f531 Resolve conflicts 2025-12-02 11:38:12 -08:00
Tzanio Kolev a546e320fd Merge pull request #4996 from mfem/contact-miniapp
Contact miniapp
2025-12-02 11:28:18 -08:00
camierjs 02d3a24007 Remove uneeded forall header file 2025-12-02 11:11:57 -08:00
camierjs 95f530ef55 Merge branch 'master' into isf 2025-12-02 10:48:07 -08:00
camierjs 27c54f58fa - Added the 'miniapps/fluids' in CHANGELOG
- Fix NVCC error (enclosing parent function for an extended __host__ __device__ lambda must allow its address to be taken)
- Fix relative path to 'general/forall.hpp'
2025-12-02 10:47:53 -08:00
Jan Nikl 60dc887101 Updated CHANGELOG for #4659 2025-12-02 10:07:21 -08:00
Veselin Dobrev fbcd9cf4d3 In Gitlab CI, change the allocation time for 'dane-baseline' to 1 hour
The previous commit had the change in the wrong place.
2025-12-02 09:56:59 -08:00
Tzanio Kolev 939f8f5522 Updated CHANGELOG 2025-12-02 09:39:28 -08:00
Mittal, Ketan 35b8ae1ce4 documentation and move UpdateId 2025-12-02 09:38:12 -08:00
Tzanio Kolev f284bdf21a Merge branch 'master' into mfem-4.9-dev 2025-12-02 08:31:12 -08:00
Veselin Dobrev f60799af0c In Gitlab CI, increase the time allocation for dane-baseline to 1 hour 2025-12-02 08:04:47 -08:00
Joseph Signorelli 2a9202cf1a Add comment clarifying tag storage in Particle 2025-12-02 08:09:35 -06:00
Will Pazner bd6b580c23 Add symmetric matrix coefficient case to PA DG Diffusion unit test 2025-12-01 23:06:42 -08:00
Will Pazner 65813867e4 Merge remote-tracking branch 'origin/master' into dg-diffusion-pa-matrix-coeff 2025-12-01 23:03:27 -08:00
Tzanio Kolev 74ff518840 Merge pull request #4840 from mfem/qf-project-gf-fallback
Add fallback for QuadratureFunction::ProjectGridFunction
2025-12-01 20:03:00 -08:00
Tzanio Kolev 4492be9ebb Merge pull request #4028 from mfem/table-array
Use Array<int> instead of Memory<int> in Table
2025-12-01 20:02:00 -08:00
Tzanio Kolev b62e215620 Merge pull request #5121 from tomstitt/claude/review-integ-kernels-01LSxNDE62DLqsm9LRtXfbcJ
Fix array shape in `SmemPADiffusionDiagonal2D`
2025-12-01 19:59:41 -08:00
Tzanio Kolev 8497c5129b Merge pull request #5127 from mfem/fix-clang-cuda
Fix CLANG + CUDA builds
2025-12-01 19:59:19 -08:00
Tzanio Kolev 12cda5ec86 Merge pull request #5113 from mfem/mfem-bounds-const
Fix const-ness for bounding related methods
2025-12-01 19:58:52 -08:00
Cyrus Harrison 41fa7cc6ae Remove comment 2025-12-01 08:38:58 -08:00
Cyrus HarrisonandJan Nikl 3787d95b62 Update fem/datacollection.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-12-01 08:34:20 -08:00
blaz b6ef7a9d5d modifications 2025-11-30 23:34:45 -08:00
Boyan Lazarov 487c665968 Merge branch 'master' into imex-conv-diff-dg 2025-11-29 01:11:46 -08:00
blaz af2439c80b style 2025-11-28 21:21:02 -08:00
blaz a81fd590ac modification of the time integrators 2025-11-28 21:20:14 -08:00
Boyan LazarovandJan Nikl 10ef282ee6 Update fem/lor/lor.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-28 20:18:16 -08:00
John Camier 25e1562c41 Merge branch 'master' into dfem-pa-caching 2025-11-28 10:04:21 -08:00
camierjs a5595f0887 Fix makefile miniapps/fluids/schrodinger-flow dir 2025-11-28 09:52:56 -08:00
camierjs 0fce2c9fb4 Rename to miniapps/fluids/schrodinger-flow 2025-11-28 09:26:25 -08:00
Veselin Dobrev d90a23f195 Fix the build with PUMI: use 'std::swap' instead of just 'swap'.
In class DenseSymmetricMatrix:
* Fix a warning from -Wextra about implicit copy-ctor.
* Explicitly use defaulted copy/move ctor/assignment.
* A small fix in the SetSize method.
* Rewrap doxygen comments to 80 chars
2025-11-26 15:32:23 -08:00
Tzanio Kolev 0c5fa5c148 Merge branch 'master' into contact-miniapp 2025-11-25 19:50:00 -08:00
Mittal, Ketan c907c4403f force all default constructors and destructor 2025-11-25 15:40:23 -08:00
Mittal, Ketan 38634ea24a force copy constructor 2025-11-25 15:10:11 -08:00
Will Pazner 838dd47259 TMOP W caching in new files 2025-11-25 14:00:31 -08:00
Will Pazner c78863a437 Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	fem/tmop/tmop_pa_da3.cpp
#	fem/tmop/tmop_pa_tc2.cpp
#	fem/tmop/tmop_pa_tc3.cpp
#	general/array.hpp
#	linalg/densemat.hpp
2025-11-25 14:00:20 -08:00
Mittal, Ketan 7a4d71d40b merge and resolve conflicts 2025-11-25 13:48:09 -08:00
Ketan Mittal 2985487736 Merge branch 'master' into mfem-bounds-const 2025-11-25 13:25:11 -08:00
Mittal, Ketan e3377b35d7 force particle move constructor 2025-11-25 13:20:06 -08:00
Tzanio Kolev 811cc65e5c Merge branch 'master' into isf 2025-11-25 12:18:57 -08:00
Tzanio Kolev 2caa75e35a Merge pull request #5072 from mfem/mtop-gpu
[MTOP] solver with GPU support
2025-11-25 12:16:20 -08:00
Tzanio Kolev adf110b3a7 Merge pull request #4981 from mfem/multivector-dev
`ParticleVector`
2025-11-25 12:15:53 -08:00
Tzanio Kolev 46fd4bdb93 Merge pull request #5117 from mfem/mexp-nurbs-m
Fix mesh-explorer material view for 3D NURBS meshes
2025-11-25 12:10:55 -08:00
Tzanio Kolev 97adb8d71e Merge branch 'master' into task/2025_10_dc_quad_points 2025-11-25 11:42:28 -08:00
Tzanio KolevandJan Nikl b6caf24a13 Update fem/datacollection.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-25 11:42:11 -08:00
Mittal, Ketan b199703b61 changes to address signed/unsigned casts 2025-11-25 11:18:32 -08:00
John Camier ab8080f413 Merge branch 'master' into claude/review-integ-kernels-01LSxNDE62DLqsm9LRtXfbcJ 2025-11-25 10:38:47 -06:00
camierjs 7f27e4605e Use MFEM_ABORT_KERNEL instead 2025-11-25 08:21:08 -08:00
Mittal, Ketan a2eb5daca4 navier bifurcation - particle vis optional 2025-11-24 20:24:06 -08:00
Mittal, Ketan 58cf6a430d minor 2025-11-24 17:40:47 -08:00
camierjs dbab49b725 Avoid taking address of pure virtual function 2025-11-24 17:22:51 -08:00
Mittal, Ketan 137a91cd29 rename AddField to AddNamedField and move data for rotating time history 2025-11-24 16:13:02 -08:00
Mittal, Ketan 8229290c6b initialization warning 2025-11-24 14:36:29 -08:00
Mittal, Ketan 15f82dd8a0 minor 2025-11-24 11:30:53 -08:00
Mittal, Ketan a860c5d18a merge and resolve conflicts 2025-11-24 10:21:40 -08:00
Mittal, Ketan 275c4762a9 fix vis functions 2025-11-24 10:18:54 -08:00
Mittal, Ketan 916e23ddc4 formatting 2025-11-24 09:23:33 -08:00
Mittal, Ketan d45bdbcea2 update CHANGELOG 2025-11-24 09:16:57 -08:00
Mittal, Ketan 9d9c7387a7 Merge branch 'master' of https://github.com/mfem/mfem into multivector-dev 2025-11-24 09:10:14 -08:00
Julian AndrejandJohn Camier fea85d855b Update fem/dfem/doperator.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-11-24 08:28:24 -08:00
Tzanio Kolev 266ff0cdf6 Merge branch 'master' into contact-miniapp 2025-11-23 17:52:07 -08:00
Mittal, Ketan 4817b4df6e Merge branch 'particleset-particle-dev' of https://github.com/mfem/mfem into particleset-particle-dev 2025-11-22 17:43:19 -08:00
Mittal, Ketan 13ad1094ba reviewer comments 2025-11-22 17:43:06 -08:00
John Camier 8c88e76283 Merge branch 'master' into isf 2025-11-21 21:38:19 -06:00
Boyan LazarovandJan Nikl 1a863dd5e1 Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 18:33:41 -08:00
blaz 800ae10652 style 2025-11-21 18:32:18 -08:00
blaz ebdbae1a32 changes addressing review comments 2025-11-21 18:31:38 -08:00
Boyan LazarovandJan Nikl 02570c2034 Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 18:16:43 -08:00
Boyan LazarovandJan Nikl 7f1bf4c48c Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 18:16:02 -08:00
blaz bd65769bac modifications 2025-11-21 18:15:30 -08:00
blaz 70bc8f51e7 modifications 2025-11-21 12:41:29 -08:00
Boyan LazarovandJan Nikl 98bc7fde72 Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:23:57 -08:00
Boyan LazarovandJan Nikl 08a1454bad Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:20:53 -08:00
Boyan LazarovandJan Nikl 4a8e18de90 Update examples/ex41.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:20:31 -08:00
Boyan LazarovandJan Nikl de71d2ec28 Update linalg/ode.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:20:12 -08:00
Boyan LazarovandJan Nikl c968008b62 Update linalg/ode.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:19:48 -08:00
Boyan LazarovandJan Nikl a188730d9c Update linalg/ode.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:18:39 -08:00
Boyan LazarovandJan Nikl 4f11387929 Update linalg/ode.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:18:17 -08:00
Boyan LazarovandJan Nikl e84092a4a1 Update linalg/ode.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:17:54 -08:00
Boyan LazarovandJan Nikl 712e806b77 Update linalg/ode.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:17:19 -08:00
Boyan LazarovandJan Nikl 3f76fe3282 Update linalg/ode.hpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-21 12:15:16 -08:00
blaz ecd6becbf9 removing WIP 2025-11-21 12:14:26 -08:00
Tzanio Kolev e44df538a5 Merge branch 'master' into dfem-pa-caching 2025-11-21 10:39:56 -08:00
Tzanio Kolev fa8c861b75 Merge pull request #5022 from mfem/dfem-assemble-matrix
dFEM assemble implementations
2025-11-21 10:39:17 -08:00
Socratis Petrides 5453c63c91 cmakelist fix 2025-11-21 09:52:16 -08:00
camierjs 81497ae1ed Sample runs file to fluids/schrodinger_flow 2025-11-20 18:12:42 -06:00
Julian Andrej cc41b690d2 Merge branch 'dfem-assemble-matrix' into dfem-pa-caching
# Conflicts:
#	fem/dfem/util.hpp
2025-11-20 09:03:12 -08:00
Joseph Signorelli b476b769c9 minor 2025-11-20 06:20:36 -06:00
Claude 5db9a79022 Fix smem array dimension bug in SmemPADiffusionDiagonal2D
Fixed array dimension mismatch similar to the bugs fixed in PRs #5108
and #4931. The QD array was declared as [MD1][MQ1] but accessed as
[Q][D], causing incorrect memory layout when MQ1 > MD1.

Changed:
  MFEM_SHARED real_t QD[3][NBZ][MD1][MQ1];
To:
  MFEM_SHARED real_t QD[3][NBZ][MQ1][MD1];

This matches the access pattern QD0[qx][dy] where qx < Q1D and dy < D1D,
and the pointer cast real_t (*)[MD1] which expects the second dimension
to be MD1.
2025-11-19 23:00:05 +00:00
Julian Andrej b9f1a89470 adapt mesh paths 2025-11-19 14:53:57 -08:00
Mittal, Ketan f844193a23 modify particle injection scheme to randomize initial location in navier example 2025-11-19 14:20:27 -08:00
Julian Andrej 36fc1b9705 move miniapps 2025-11-19 14:10:58 -08:00
Mittal, Ketan e3eb2516fa minor 2025-11-19 13:13:15 -08:00
Cyrus 86b200f138 style 2025-11-19 12:55:22 -08:00
Cyrus Harrison aa0dbb384f apply review suggestions 2025-11-19 12:50:06 -08:00
Mittal, Ketan 11ff698c77 update removeparticles logic 2025-11-19 11:32:05 -08:00
Mittal, Ketan 5291bd59fc change particle id type to unsigned long long 2025-11-19 10:52:11 -08:00
camierjs e1edcb321d Fix nvcc access errors and VectorCoefficient partially overridden warnings 2025-11-19 12:14:42 -06:00
John Camier c0d9732d1b Merge branch 'master' into mtop-gpu 2025-11-19 09:45:55 -08:00
Mittal, Ketan 237f566334 merge remote changes and resolve conflicts 2025-11-18 14:01:48 -08:00
Mittal, Ketan f9cb7e0a06 more reviewer comments 2025-11-18 13:18:16 -08:00
Julian Andrej 354859c91d size_t -> int for map key 2025-11-18 08:29:48 -08:00
Socratis Petrides 6bc2b2b5aa another doc fix 2025-11-17 19:53:55 -08:00
Ketan MittalandJan Nikl f970b2cf8f Apply suggestions from code review
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-17 19:36:14 -08:00
Mittal, Ketan 96b1ba0e4f address some reviewer comments 2025-11-17 19:28:59 -08:00
Socratis Petrides dd2e531218 fix doxygen comment 2025-11-17 19:12:31 -08:00
Socratis Petrides 769c396613 fixing comments in ip 2025-11-17 18:59:11 -08:00
Socratis Petrides b7a5940f00 forgotten makefile changes 2025-11-17 18:55:03 -08:00
Socratis Petrides cab5e5f99c added the new direct solver interface to contact driver 2025-11-17 18:53:13 -08:00
Socratis Petrides 0226a21a76 added generic direct solver interf 2025-11-17 18:52:32 -08:00
John Camier 86867e2378 Merge branch 'master' into isf 2025-11-17 18:22:30 -08:00
Socratis Petrides 9302c4229a remove leftover comment 2025-11-17 15:12:16 -08:00
Dylan Copeland 096605f532 Revert empty line. 2025-11-17 15:03:29 -08:00
Dylan Copeland 69b4689048 Fix a bug so that "m" option works for 3D NURBS meshes. 2025-11-17 14:58:21 -08:00
Julian Andrej c3ee799e24 use fold expression 2025-11-17 13:57:57 -08:00
Ketan MittalandJan Nikl 2a504d30a7 Apply suggestions from code review
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-11-17 13:56:18 -08:00
Julian Andrej 8056ba5610 try again 2025-11-17 13:49:22 -08:00
Julian Andrej 7dd05fe015 guard against MSVC N=0 case 2025-11-17 12:58:08 -08:00
Socratis Petrides 861cb9cb6b Merge branch 'master' into contact-miniapp 2025-11-17 11:59:29 -08:00
Socratis Petrides 458254a1f6 set relax_type depending on Hypre version 2025-11-17 11:59:00 -08:00
Socratis Petrides 4defb5c401 removing no longer needed c++ standard spec 2025-11-17 11:58:09 -08:00
Mittal, Ketan aecd9648af merge with master and resolve conflicts 2025-11-17 11:37:43 -08:00
Mittal, Ketan 1e592a7442 change reordering logic 2025-11-17 11:35:47 -08:00
Julian Andrej 78608e4fcd remove prints 2025-11-17 11:09:15 -08:00
Julian Andrej 78a90bbee8 remove prints 2025-11-17 11:08:57 -08:00
Julian Andrej d0f4f9baaf deduction simplification 2025-11-17 11:07:04 -08:00
Julian Andrej 060e253132 simplification for deduction 2025-11-17 11:06:15 -08:00
Joseph Signorelli bc7ea43317 Move ordering to linalg 2025-11-17 12:17:53 -06:00
Tzanio Kolev 9cc28d8772 Merge branch 'master' into mfem-bounds-const 2025-11-17 05:25:06 -08:00
Vladimir Z Tomov cef3517d60 updated contact/README with more details for the build. 2025-11-14 18:00:22 -08:00
Julian Andrej 72b7390aca pa caching 2025-11-14 15:04:06 -08:00
camierjs e6c9f73fc6 Use more informative variable names
Update README with equations and solver steps
2025-11-14 11:36:34 -08:00
Socratis Petrides 555d5771a6 Merge branch 'amgf' into contact-miniapp 2025-11-14 09:55:00 -08:00
Socratis Petrides 147204ee70 readme edits 2025-11-13 17:28:15 -08:00
Socratis Petrides 4b8832190d cmake fixes 2025-11-13 17:23:55 -08:00
Socratis Petrides 69159b13a6 Merge branch 'amgf' into contact-miniapp 2025-11-13 16:54:05 -08:00
Ketan Mittal edbdf0628d Merge branch 'master' into mfem-bounds-const 2025-11-13 14:16:53 -08:00
Tzanio Kolev 7a9253d13c Merge branch 'master' into dfem-assemble-matrix 2025-11-13 13:18:20 -08:00
John Camier 20dcc1f991 Merge branch 'master' into mtop-gpu 2025-11-13 11:03:48 -08:00
camierjs 806ed1da96 Merge branch 'master' into isf 2025-11-13 10:58:38 -08:00
camierjs 1c486dbb39 Add sample runs, propagate new miniapps/fluids changes 2025-11-13 10:54:45 -08:00
camierjs 6f121c8abf Move ISF miniapp to fluids subdirectory 2025-11-13 10:53:52 -08:00
Julian Andrej 53a5c2f687 remove unused variables 2025-11-13 10:40:27 -08:00
Julian Andrej 6b279d6e32 initialize qp only once each thread 2025-11-13 10:22:21 -08:00
Mittal, Ketan 05d5294709 fix const-ness for bounding related methods 2025-11-12 23:09:44 -08:00
Julian Andrej 5ed263e9b4 try threadblocks 2025-11-12 16:56:36 -08:00
Ketan Mittal 7b85c04c72 Merge branch 'master' into particleset-particle-dev 2025-11-12 14:51:39 -08:00
camierjs 5450915b0b CHANGELOG mtop miniapp entry 2025-11-12 10:31:08 -08:00
camierjs 06d6bdbad0 Merge branch 'mtop-gpu' of github.com:mfem/mfem into mtop-gpu 2025-11-12 09:49:38 -08:00
camierjs c2955a8d52 Rename to GetSolutionVector & GetAdjointSolutionVector 2025-11-12 09:49:13 -08:00
John Camier 5b1de08790 Merge branch 'master' into mtop-gpu 2025-11-11 20:44:47 -08:00
John Camier 6d4492bad1 Merge branch 'master' into isf 2025-11-11 20:44:23 -08:00
John Camier 081124511d Merge branch 'master' into isf 2025-11-11 12:18:22 -08:00
Will Pazner 7e98c14f9b Fix copy and move semantics in DenseSymmetricMatrix 2025-11-11 11:28:20 -08:00
Julian Andrej 4528a608ea Merge branch 'master' into dfem-assemble-matrix 2025-11-11 11:24:58 -08:00
Joseph Signorelli 3be205bd4a minor doc fixes 2025-11-11 07:58:05 -06:00
blaz 1f20f8a232 removed device setup as it is not supported 2025-11-10 21:07:41 -08:00
Ketan Mittal 71699fe790 Merge branch 'master' into multivector-dev 2025-11-10 18:09:47 -08:00
Mittal, Ketan 269db5255b Merge branch 'master' of https://github.com/mfem/mfem into particleset-particle-dev 2025-11-10 18:09:17 -08:00
Socratis Petrides 25bf3e0235 style 2025-11-10 16:32:08 -08:00
Socratis Petrides 763ea10fd1 remove deprecated warning 2025-11-10 16:31:45 -08:00
Socratis Petrides 4439841f79 adjusting the contact driverr to the amgf changes 2025-11-10 16:28:53 -08:00
Socratis Petrides d834fd8cb3 Merge branch 'amgf' into contact-miniapp 2025-11-10 16:11:15 -08:00
Socratis Petrides 1b2ad14923 Merge branch 'amgf' into contact-miniapp 2025-11-10 14:45:40 -08:00
Socratis Petrides 95fc0b414f fix readme line length 2025-11-10 14:44:24 -08:00
Socratis Petrides 4f0b12acd6 Merge branch 'amgf' into contact-miniapp 2025-11-10 14:32:02 -08:00
Socratis Petrides 04453a12c7 merge with amgf 2025-11-10 13:27:51 -08:00
camierjs b497159d50 Remove used private field 2025-11-10 13:00:02 -08:00
camierjs 60e0bef7d3 Merge branch 'master' into mtop-gpu 2025-11-10 12:05:06 -08:00
camierjs f6d6148108 Fix comments 2025-11-10 12:04:58 -08:00
camierjs 7cda588460 Remove unused declarations 2025-11-10 12:03:52 -08:00
Mittal, Ketan 5272af9702 skip documentation for kernel specialization 2025-11-10 11:49:29 -08:00
Mittal, Ketan 2898b4e900 fix greek symbols 2025-11-10 11:08:11 -08:00
camierjs c8d2c38b57 Merge branch 'master' into mtop-gpu 2025-11-10 08:31:17 -08:00
camierjs 2a24edf180 Use '-1' for 'all' BCs, add verifications
Add sample runs in mtop example
Remove unused code in mtop example
2025-11-10 08:31:01 -08:00
Julian Andrej 555ce60a73 style 2025-11-10 08:00:57 -08:00
Julian Andrej 76de06bf32 remove vectordivergence from gpu tests 2025-11-10 07:57:57 -08:00
Joseph Signorelli 5955c67c0e fix type check in MFEM_ASSERT 2025-11-09 18:33:43 -06:00
Joseph Signorelli c6d43016b1 style 2025-11-09 18:27:16 -06:00
Joseph Signorelli 9d9ce86383 Remove AddSpecialization since it is not functional 2025-11-09 18:25:18 -06:00
Joseph Signorelli 32cb62627a Cleanup + enhancements to new transfer capability 2025-11-09 18:25:01 -06:00
John Camier a68111620a Merge branch 'master' into isf 2025-11-09 07:37:04 -08:00
John Camier a5fd83b1fb Merge branch 'master' into mtop-gpu 2025-11-09 07:36:56 -08:00
blaz 8a02f2ab4a style 2025-11-09 00:15:52 -08:00
blaz 6ab2ede252 implementation of requested changes 2025-11-09 00:13:50 -08:00
Will Pazner 4d042959e8 Add convenience function Mesh::IsMixedMesh 2025-11-08 14:03:37 -08:00
Joseph Signorelli 7ae91efec2 Only cast to size_t for int type comparison 2025-11-08 09:55:06 -06:00
Mittal, Ketan 48da9bad4f fix int comparison with size_t 2025-11-07 16:55:44 -08:00
Mittal, Ketan 4f04240372 cmake fix for navier-bifurcation 2025-11-07 16:51:12 -08:00
Mittal, Ketan caf484d910 modify navier bifurcation so that it can also be compiled without particles 2025-11-07 16:28:36 -08:00
Tzanio Kolev 19e4c53acd Merge branch 'master' into qf-project-gf-fallback 2025-11-07 13:34:56 -08:00
Cyrus 34b41058f9 style 2025-11-07 13:27:37 -08:00
Cyrus Harrison 9e48d1b618 Merge branch 'master' into task/2025_10_dc_quad_points 2025-11-07 13:08:08 -08:00
Joseph Signorelli a1c1e48404 Merge branch 'particleset-particle-dev' of github.com:mfem/mfem into particleset-particle-dev 2025-11-07 15:04:09 -06:00
Joseph Signorelli b94d258523 minor doc updates 2025-11-07 15:04:00 -06:00
Cyrus Harrison a4ffea2f8a add more entries to the mfem_root file for gf and qf 2025-11-07 13:03:00 -08:00
Mittal, Ketan bb8ff078fb restore transfer function 2025-11-07 12:59:51 -08:00
Mittal, Ketan d5098c8381 clean up and redistribute only particles not on same rank 2025-11-07 12:38:03 -08:00
Mittal, Ketan c007fb4ad4 cosmetic 2025-11-07 10:33:29 -08:00
Mittal, Ketan 5526a18358 cleanup and new transfer function 2025-11-07 09:16:31 -08:00
Mittal, Ketan 618c0233ea make style and some formatting changes 2025-11-05 13:47:42 -08:00
Mittal, Ketan 66b30e5492 merge particlevector branch 2025-11-05 13:17:29 -08:00
Mittal, Ketan 9f07a13589 fix makefile and add some ifdef guards 2025-11-05 12:41:53 -08:00
Tzanio Kolev 78cd3d3838 Merge branch 'master' into multivector-dev 2025-11-05 09:53:34 -08:00
Joseph Signorelli 7da0f4c226 TestResize --> TestSetNumParticles for clarity 2025-11-05 08:24:08 -06:00
Joseph Signorelli 3681b6f6a2 Merge branch 'multivector-dev' of github.com:mfem/mfem into multivector-dev 2025-11-05 08:19:08 -06:00
Joseph Signorelli 021bf157a7 Documentation updates 2025-11-05 08:16:16 -06:00
Joseph Signorelli 0463749eb9 update_data --> keep_data 2025-11-05 08:16:07 -06:00
Mittal, Ketan b8443bf867 formatting fixes 2025-11-04 17:09:54 -08:00
Mittal, Ketan 793fd42653 minor 2025-11-04 16:59:11 -08:00
Joseph Signorelli 7e6034c4e2 potential fix to mem leak? 2025-11-04 18:10:00 -06:00
Mittal, Ketan db4f95d252 Merge branch 'multivector-dev' of https://github.com/mfem/mfem into multivector-dev 2025-11-04 14:45:21 -08:00
Mittal, Ketan 40bed3db8f make style 2025-11-04 14:45:06 -08:00
Ketan Mittal 2f6c8b720c Merge branch 'master' into multivector-dev 2025-11-04 14:43:25 -08:00
Mittal, Ketan 44b8778af1 add update data to SetNumParticles and GrowSize 2025-11-04 14:42:43 -08:00
Mittal, Ketan 4311f1ab04 make style 2025-11-04 14:07:01 -08:00
Mittal, Ketan cc1a5774e1 add class documentation 2025-11-04 14:01:50 -08:00
Mittal, Ketan c1b7a529d6 resolve some documentation mix-up due to earlier merge 2025-11-04 13:43:11 -08:00
Mittal, Ketan 6dca8a1c54 merge upstream changes 2025-11-04 13:38:34 -08:00
Mittal, Ketan b0b21633b7 fix line wraps and add some checks 2025-11-04 13:36:54 -08:00
Joseph Signorelli 71aa4a366d Add update_data flag for SetOrdering + SetVDim 2025-11-04 15:21:29 -06:00
Joseph Signorelli 218880f7e1 add wrong_comp_count check to unit test 2025-11-04 15:14:23 -06:00
Joseph Signorelli 0c61b567a5 Merge branch 'array-vector-improvements-dev' into multivector-dev 2025-11-04 14:41:08 -06:00
Joseph Signorelli 55fb86f566 Merge branch 'master' into multivector-dev 2025-11-04 14:39:22 -06:00
Joseph Signorelli 94f2670dc0 style 2025-11-04 14:36:29 -06:00
Joseph Signorelli 749a04e7b1 MultiVector --> ParticleVector 2025-11-04 14:35:46 -06:00
Joseph Signorelli c6744e3f39 multivector --> particlevector file names 2025-11-04 14:21:22 -06:00
Joseph Signorelli 8a35a7843a style 2025-11-04 14:11:53 -06:00
Joseph Signorelli 967e811442 Keep existing data for SetVDim, with unit test 2025-11-04 14:11:24 -06:00
Joseph Signorelli 260f7c345b Add Get/SetComponents w/ unit test 2025-11-04 13:55:02 -06:00
Joseph Signorelli 1bfe259d2a add test_ordering to cmakelists 2025-11-04 13:47:56 -06:00
Joseph Signorelli a4b5eb8bda Add unit test for getting + setting vector values 2025-11-04 13:47:27 -06:00
Joseph Signorelli 0fce771a08 rm file header duplication 2025-11-04 13:46:48 -06:00
Joseph Signorelli 331151d4f6 Create new file for test ordering 2025-11-04 13:46:22 -06:00
Julian Andrej 78b4b00359 remove unnecessary header 2025-11-04 11:31:09 -08:00
Julian Andrej 9c110c6acb Merge branch 'master' into dfem-assemble-matrix
# Conflicts:
#	tests/unit/dfem/test_mass.cpp
2025-11-04 08:36:38 -08:00
Tzanio Kolev c523e620d9 make style 2025-11-04 07:32:43 -08:00
Joseph Signorelli eabce59758 Default vdim to 1 2025-11-04 08:09:02 -06:00
Joseph Signorelli 779a4e6c0b Move ordering to general in its own file 2025-11-04 08:07:26 -06:00
Cyrus Harrison af36461372 add quad func support to visit and conduit data collections 2025-11-03 15:18:02 -08:00
Joseph Signorelli 43231cb143 Revert "Move Ordering to multivector.hpp/cpp" and move Ordering::Reorder to fespace.cpp/hpp
This reverts commit 769d2914c8.
2025-11-03 11:19:34 -06:00
John Camier 89a4775928 Merge branch 'master' into isf 2025-11-02 08:50:03 -08:00
John Camier d235e23b6a Merge branch 'master' into mtop-gpu 2025-11-02 08:49:58 -08:00
John Camier 5f088873f3 Merge branch 'master' into dg-diffusion-pa-matrix-coeff 2025-11-02 08:49:46 -08:00
Will Pazner 7757c47fcf Add symmetric matrix test case to "PA DG Diffusion" 2025-11-01 14:06:21 -07:00
Will Pazner cd2e84fa1c Fix copy and move semantics in DenseSymmetricMatrix 2025-11-01 14:06:01 -07:00
Will Pazner 9fb62e141c Add non-symmetric test to "PA DG Diffusion" 2025-10-31 12:32:45 -07:00
Julian Andrej fdd909b132 typos and missing params 2025-10-30 08:34:43 -07:00
Julian Andrej 9222808f7c documentation 2025-10-30 08:23:20 -07:00
Julian Andrej 792c5b7617 revert to random input vector 2025-10-30 07:42:18 -07:00
John Camier 2a23163240 Merge branch 'master' into isf 2025-10-29 19:36:46 -07:00
John Camier 09e4b7296e Merge branch 'master' into mtop-gpu 2025-10-29 19:36:36 -07:00
Julian Andrej 8001d0b75a make style 2025-10-28 16:49:45 -07:00
Julian Andrej 2bebd0e4a4 relax norms properly 2025-10-28 16:47:16 -07:00
Julian Andrej 3154767cc5 relax tolerance to 2*eps 2025-10-28 13:10:35 -07:00
Julian Andrej f4f4889a4c fix memory leak 2025-10-28 11:42:28 -07:00
Julian Andrej 7850d9d0a9 integration rule 2025-10-28 10:57:20 -07:00
John Camier 1d4029bcd8 Merge branch 'master' into isf 2025-10-27 18:19:05 -07:00
John Camier decc2f695d Merge branch 'master' into mtop-gpu 2025-10-27 18:18:56 -07:00
Julian Andrej 83330e2639 revert debug device 2025-10-27 16:17:01 -07:00
Julian Andrej 5889c2155e use existing matrix comparison test - gpu still failing 2025-10-27 15:38:10 -07:00
Julian Andrej 16694dd1c7 Merge branch 'dfem-assemble-matrix' of github.com:mfem/mfem into dfem-assemble-matrix 2025-10-27 08:04:01 -07:00
Julian Andrej e8c3dc1e8f explicit casts 2025-10-27 08:03:51 -07:00
Julian Andrej 8a6ac41d69 slight norm comparison discrepancy remaining on rt-2d-q3 2025-10-27 08:03:43 -07:00
camierjs 2bad979cac Add GPU tags for dFEM diffusion, divergence and lvector parallel unit tests 2025-10-24 18:05:58 -07:00
John Camier e9f0e86d61 Merge branch 'master' into isf 2025-10-24 08:57:04 -07:00
John Camier 9e058cbfe9 Merge branch 'master' into mtop-gpu 2025-10-24 08:56:55 -07:00
Boyan Lazarov 8afc72d0f5 Merge branch 'master' into imex-conv-diff-dg 2025-10-23 22:49:20 -07:00
blaz b598b81496 using only the interface of TimeDependentOperator 2025-10-23 22:46:57 -07:00
Will Pazner 579f88a06d Add unit test for PA DG diffusion with matrix coefficients 2025-10-21 15:52:42 -07:00
Will Pazner 69650e350b Add support for PA DG diffusion with matrix coefficients 2025-10-21 15:52:42 -07:00
Will Pazner 9604a64178 Add unit test for PA with MatrixConstantCoefficient 2025-10-21 15:52:42 -07:00
Will Pazner 4733235d42 Fix bug in CoefficientVector::Project(MatrixCoefficient&, bool) when tranpose is true 2025-10-21 15:52:42 -07:00
Will Pazner c6d455699d Add support for constant matrix coefficients in PA diffusion integrator 2025-10-21 15:52:42 -07:00
camierjs 2658a312dd Fix documentation typo 2025-10-21 12:12:12 -07:00
camierjs 52dffe5f8a Add visualization keys 2025-10-21 11:15:02 -07:00
John Camier 8fbd815ad9 Merge branch 'master' into dfem-assemble-matrix 2025-10-21 10:09:46 -07:00
John Camier 86c103ad23 Merge branch 'master' into mfem-4.9-dev 2025-10-21 09:03:15 -07:00
camierjs 90379a4792 mtop_test_iso_elasticity mesh file directory and output numbers 2025-10-19 10:24:33 -07:00
camierjs 5f33d92ca8 Move mtop data meshes to miniapp folder 2025-10-19 09:30:40 -07:00
John Camier d911f7986a Merge branch 'master' into isf 2025-10-19 09:27:44 -07:00
John Camier c04e52bc6a Merge branch 'master' into mtop-gpu 2025-10-19 09:27:35 -07:00
Socratis Petrides f492c5c70d add README 2025-10-18 19:15:39 -07:00
Socratis Petrides 474d9811dd minor edits in the driver 2025-10-17 21:19:39 -07:00
Socratis Petrides 9a2ef1263f contact-miniapp: squash all changes since branching from master 2025-10-17 18:39:55 -07:00
Will Pazner 6602df4fd2 In QuadratureInterpolator::SupportsFESpace, return false for mixed meshes or variable orders
In QuadratureFunction::ProjectGridFunction unit test, test (element) QuadratureSpace also.
2025-10-17 14:58:46 -07:00
Will Pazner 63d983ed6a Add unit test for QuadratureFunction::ProjectGridFunction 2025-10-15 14:54:25 -07:00
Will Pazner 59a95cc8e6 In QuadratureFunction::ProjectGridFunction, fall back earlier
GetElementRestriction or GetFaceRestriction may fail in unsupported cases.

In the case of FaceQuadratureSpace, fall back on non-tensor meshes since
ElementDofOrdering::NATIVE is not supported in the restriction operator.
2025-10-15 14:52:36 -07:00
Jan Nikl 440d552e85 Merge branch 'master' into qf-project-gf-fallback 2025-10-15 10:04:34 -07:00
John Camier 91e56806bd Merge branch 'master' into mtop-gpu 2025-10-15 07:46:42 -07:00
John Camier b61771199b Merge branch 'master' into isf 2025-10-14 16:29:07 -07:00
camierjs 3bfa5f0e37 [MTOP] solver with GPU support 2025-10-14 15:37:14 -07:00
John Camier 2747db62a0 Merge branch 'master' into isf 2025-10-13 08:12:49 -07:00
Tzanio Kolev 1dd816d11f Merge branch 'master' into mfem-4.9-dev 2025-10-12 16:57:16 -07:00
camierjs 6cb1f3a11d Restrict identifiers characters 2025-10-08 15:21:49 -07:00
camierjs 8e8f7a5933 Avoid visualization in ISF tests 2025-10-08 14:40:40 -07:00
camierjs 94e5520fc4 Fix CMake pschrodinger_flow tests 2025-10-08 14:09:49 -07:00
camierjs 9e2abb4907 Add Incompressible Schrödinger Flow miniapp make tests 2025-10-08 13:43:45 -07:00
camierjs b011eea5ac Single precision device sincos fix 2025-10-08 13:23:16 -07:00
camierjs 82a69bb15b Allow single build 2025-10-08 13:10:38 -07:00
Boyan Lazarov 78034273ea Merge branch 'master' into imex-conv-diff-dg 2025-10-08 12:31:14 -07:00
blaz fd6fb0c7be style 2025-10-08 12:30:34 -07:00
camierjs 718fe7a898 Update miniapps/solvers README 2025-10-08 12:29:56 -07:00
camierjs 0a4930e714 Merge branch 'master' into isf 2025-10-08 12:02:15 -07:00
camierjs 05eb28b889 Simplify complex functions 2025-10-08 12:01:53 -07:00
blaz e1b547ef97 small modification 2025-10-06 23:00:38 -07:00
Tzanio Kolev 0a55df91f1 Initial changes for mfem-4.9 2025-10-06 07:51:32 -07:00
Anthony 435e1dfe3c Merge branch 'master' into imex-conv-diff-dg 2025-10-02 18:51:19 +00:00
Tzanio Kolev 792af80eac Merge branch 'master' into dfem-assemble-matrix 2025-09-24 08:12:52 -07:00
camierjs 737d3d6f2b Options & sample runs 2025-09-23 11:38:11 -07:00
camierjs c5f0b5dee1 Merge branch 'master' into isf 2025-09-23 09:56:30 -07:00
Julian Andrej 593622ae88 host/device bug 2025-09-23 09:48:57 -07:00
Julian Andrej 256821fb6c maybe unused annotations 2025-09-23 09:48:47 -07:00
Julian Andrej cffcd4ea89 hypre gpu fix 2025-09-23 09:48:39 -07:00
Julian Andrej 66dfc9352f nvcc fixes 2025-09-22 11:58:55 -07:00
Julian Andrej 4cd6c1fbfb add matrix assemble and amg option to minsurface miniapp 2025-09-22 10:06:04 -07:00
John Camier 78afa5d313 Merge branch 'master' into table-array 2025-09-21 10:31:08 -07:00
Julian Andrej 47ab067ead Merge branch 'master' into dfem-assemble-matrix 2025-09-19 10:17:29 -07:00
Toni-ko 5872250edc Trying to fix some failing checks 2025-09-19 09:39:04 -07:00
Julian Andrej 72682e3f46 assemble methods and unit tests 2025-09-19 09:16:26 -07:00
Toni-ko b41d2681ac fix for some failing checks 2025-09-18 17:26:33 -07:00
Toni-ko 5840edb3f5 Added option to use continuous elements 2025-09-18 17:09:03 -07:00
Toni-ko 668156f7c4 style update 2025-09-18 16:37:54 -07:00
Toni-ko fad7b3c0fd set problem type using template, along with other small modifications 2025-09-18 16:37:35 -07:00
blaz f2918eda9e small mod 2 2025-09-18 11:09:23 -07:00
blaz 6989bc3899 small modifications 2025-09-18 11:01:23 -07:00
camierjs 4849db33bf Update gitignore 2025-09-09 11:29:50 -07:00
Toni-ko 9bb4bbb87a update gitignore 2025-08-25 17:02:51 -07:00
Toni-ko b2be14cfe4 fixed option parser issue 2025-08-25 15:41:33 -07:00
Toni-ko 6505a97745 added an override mark to mult1 2025-08-25 15:14:51 -07:00
Toni-ko 1ef3beec81 fixes for tests 2025-08-25 14:54:22 -07:00
Toni-ko d66004b3b3 code style 2025-08-25 14:42:15 -07:00
Toni-ko 41985bcbce fixes for tests 2025-08-25 14:42:01 -07:00
Toni-ko c9b6d0c9d2 Fixed some unused variable issues 2025-08-25 12:36:19 -07:00
Toni-ko e016ca926a code style 2025-08-25 11:44:54 -07:00
Toni-ko 6a1ef643d3 Fixed warning with hidden function 2025-08-25 11:40:24 -07:00
Will Pazner d9d4b5bb72 Rename member variable to fix shadow warning 2025-08-23 15:13:05 -07:00
Will Pazner 5c04ba55f6 Fix failing TMOP tests
Keep a cached copy of GeomToPerfGeomJac for use on device.

Be sure to call HostRead before DenseMatrix::Det.
2025-08-23 15:06:03 -07:00
Joseph Signorelli 820e14875b Merge branch 'master' into particleset-particle-dev 2025-08-19 16:42:49 -07:00
Joseph Signorelli f43c6771c6 Merge branch 'master' into multivector-dev 2025-08-19 16:42:31 -07:00
Joseph Signorelli 2ffbdf1a18 add newlines to end of all files 2025-08-19 13:30:09 -07:00
Joseph Signorelli 065bbcb9a5 Use Array<int> of size 1 for tags in Particle, add set tag ref, and include unit test for GetParticleRef 2025-08-19 13:24:16 -07:00
Joseph Signorelli 09327924ee Merge branch 'master' into multivector-dev 2025-08-19 12:15:01 -07:00
Joseph Signorelli e659e823e9 Add newlines to end of files 2025-08-19 12:06:43 -07:00
Joseph Signorelli a660b5fc07 Potential fix to std::iota not found for windows build 2025-08-18 16:14:12 -07:00
Joseph Signorelli 502e422d4b Potential fix to Particle::tags memory leak 2025-08-18 15:47:32 -07:00
Joseph Signorelli a97a13a342 Minor documentation improvements 2025-08-18 15:28:26 -07:00
Joseph Signorelli 65a95551ea Fix another int comparison w/ std::size_t 2025-08-18 14:56:04 -07:00
Joseph Signorelli 9fb85d2d0b Fix remaining -Wall 2025-08-18 14:42:53 -07:00
Joseph Signorelli 99cdb577fb Fix unused const variable (for when MFEM_USE_GSLIB not defined) 2025-08-18 14:23:35 -07:00
Joseph Signorelli 4c36ae0f47 Fix initialize of std::string w/ nullptr 2025-08-18 13:42:53 -07:00
Joseph Signorelli ad839667c0 Single-precision 2025-08-18 12:46:47 -07:00
Joseph Signorelli 8e058595b4 fix reorder-ctor error 2025-08-18 11:28:43 -07:00
Joseph Signorelli e06f1a4267 use std::size_t for loops over std .size() types 2025-08-18 11:21:28 -07:00
Joseph Signorelli eef84c7a10 Do not build navier_particles + navier_bifurcation if not MFEM_USE_GSLIB, in makefile 2025-08-18 11:07:12 -07:00
Joseph Signorelli 916d14d2b7 Fix use of string after lifetime ends 2025-08-18 11:06:56 -07:00
Joseph Signorelli 00dc6d2780 fix test errors 2025-08-15 15:56:04 -07:00
Joseph Signorelli dfc786fbbf Fix docs 2025-08-15 15:31:48 -07:00
Joseph Signorelli 1f8dbc7bfe Fix doc 2025-08-15 15:20:23 -07:00
Joseph Signorelli 194f3005bb Add channel2.mesh 2025-08-15 15:16:46 -07:00
Joseph Signorelli 05103d26a9 Add clean to makefile for bifurcation 2025-08-15 14:42:27 -07:00
Joseph Signorelli 6a6f5e4d23 Add navier bifurcation (+ output) to gitignore 2025-08-15 14:40:59 -07:00
Joseph Signorelli 5dd665f37b minor 2025-08-15 14:38:00 -07:00
Joseph Signorelli f5d33d661a style NavierParticles 2025-08-15 14:37:18 -07:00
Joseph Signorelli 217b3d2d2d Add Navier_Bifurcation 2025-08-15 14:36:58 -07:00
Joseph Signorelli 5974bfbafb Add GetCurrentVorticity to Navier 2025-08-15 14:36:27 -07:00
Joseph Signorelli 05da808856 Add NavierParticles class 2025-08-15 14:13:22 -07:00
Joseph Signorelli 7d159da97c Add particles_redist miniapp 2025-08-15 14:00:55 -07:00
Joseph Signorelli b6d6473d36 Formatting + style 2025-08-15 14:00:43 -07:00
Joseph Signorelli 9ff5d24102 serial compile bug fixes 2025-08-15 13:57:47 -07:00
Joseph Signorelli 3dd9427c7e Fix bug when compiling w/o GSLIB 2025-08-15 13:57:47 -07:00
Joseph Signorelli 0d0c02b715 Add miniapp common particle functions + ParticleTrajectories class 2025-08-15 13:57:47 -07:00
Joseph Signorelli 355e434575 Add particles_extras.cpp/hpp to miniapps/common 2025-08-15 13:57:47 -07:00
Veselin Dobrev 1bf4c4f7fe Merge branch 'master' into table-array 2025-08-15 12:23:56 -07:00
Joseph Signorelli f8fa4854bf Add particle/particleset unit test. 2025-08-15 11:01:00 -07:00
Joseph Signorelli bbed72b3c6 Add ParticleSet class 2025-08-15 10:47:42 -07:00
Joseph Signorelli f377c63ea7 Add Particle class 2025-08-15 10:40:56 -07:00
Joseph Signorelli ed80737a9a Create particleset.cpp/hpp 2025-08-15 10:32:22 -07:00
Joseph Signorelli 6aba0652f1 Merge branch 'multivector-dev' into particleset-particle-dev 2025-08-13 16:48:37 -07:00
Joseph Signorelli f85ee8391d Merge branch 'fdpts-improve-dev' into particleset-particle-dev 2025-08-13 16:48:09 -07:00
Joseph Signorelli 715ab0a328 Fix typo causing doc fail 2025-08-13 16:35:47 -07:00
Joseph Signorelli 160100e0b3 style 2025-08-13 16:27:19 -07:00
Joseph Signorelli 997942b44e Add MultiVector w/ unit tests 2025-08-13 16:22:23 -07:00
Joseph Signorelli a1bb9cfe9f style 2025-08-13 16:22:23 -07:00
Joseph Signorelli 2489dce8a0 Add Vector::Reserve 2025-08-13 16:22:23 -07:00
Joseph Signorelli 4ea338fe53 Add Vector::DeleteAt w/ unit test 2025-08-13 16:22:23 -07:00
Joseph Signorelli eb5a0eb132 Add Array::DeleteAt w/ unit test. 2025-08-13 16:22:23 -07:00
Joseph Signorelli 236ba45fb9 Add Ordering::Reorder w/ unit test 2025-08-13 16:11:52 -07:00
Joseph Signorelli e23975a2d8 Add test_multivector.cpp 2025-08-13 16:10:41 -07:00
Joseph Signorelli 769d2914c8 Move Ordering to multivector.hpp/cpp 2025-08-13 16:02:37 -07:00
Joseph Signorelli 5c87a6c665 Create new files multivector.cpp/hpp 2025-08-13 15:47:17 -07:00
Veselin Dobrev 32efc9be3c Merge pull request #4978 from mfem/table-array-additions
Proposed addtions and changes to PR #4028 (branch table-array)
2025-08-13 12:16:20 -07:00
Veselin Dobrev 2675f36788 Update formatting 2025-08-13 10:57:22 -07:00
Veselin Dobrev 3372b2d1a8 Proposed addtions and changes to PR #4028 (branch table-array)
* Add mfem::swap for the classes Memory, Array, Array2D, and Vector.
* These mfem::swap functions are for use by the standard library.
* Re-define mfem::Swap to use mfem::swap, or, if is not defined, std::swap
* Remove some calls to Memory::Reset after Memory::Delete since the latter
  calls the former
* Update/improve the definitions of the Vector move-constructor and move-
  assignment; in the copy-assignment, skip the virtual calls to
  v.UseDevice(bool) when they are not needed.
* In DenseMatrix::SetSize, update the size of the data Array to match
  height x width.
* Propagate the parameter use_dev in Table::{ReadI,ReadJ} to the respective
  Array::Read calls.
* In Table::Size_of_connections, return J.Size() instead of I[size].
* Some small Doxygen tweaks.
2025-08-12 14:51:58 -07:00
Veselin Dobrev 8a278c70d6 Merge branch 'master' into table-array 2025-08-12 14:44:38 -07:00
blaz e5d4917c65 fix memory leaks 2025-08-11 20:23:32 -07:00
blaz 258d8c7dec Merge branch 'master' into imex-conv-diff-dg 2025-08-11 19:13:04 -07:00
Toni-ko e3ce7f2de4 resolving an error 2025-08-11 18:06:20 -07:00
Toni-ko a579e7f6c1 attempting to resolve failing check 2025-08-11 18:02:07 -07:00
Toni-ko e65eae9a59 typo 2025-08-11 17:52:12 -07:00
Toni-ko 93c45b205b code style update 2025-08-11 17:11:35 -07:00
Toni-ko 8805e6cfcf Added Sample Runs. Updated comments. Code cleanup. 2025-08-11 16:22:38 -07:00
Will Pazner 7fea470192 Modify Array::Copy to match semantics of Vector
Also revert Array copy assignment to use Array::Copy
2025-08-11 15:24:25 -07:00
Will Pazner 6bab6a283f Use same copy assignment semantics in Array as in Vector 2025-08-09 10:25:57 -07:00
Will PaznerandVeselin Dobrev 5a7edc9548 Fix Array move constructor and assignment operator
Co-authored-by: Veselin Dobrev <dobrev@llnl.gov>
2025-08-09 08:28:40 -07:00
Toni-ko 305dd2e17c example 41 using lor preconditioner 2025-08-08 13:42:17 -07:00
Toni-ko 4bc378f84f Merge remote-tracking branch 'origin/lor_dg_preconditioner' into imex-conv-diff-dg 2025-08-07 17:55:21 -07:00
Toni-ko 2ccd455706 Included preconditioner 2025-08-07 15:41:11 -07:00
Veselin Dobrev bdb627a5ae Merge branch 'master' into table-array
Resolved conflicts:
   general/array.hpp
2025-08-07 11:24:27 -07:00
Toni-ko f319c14ecc Parallel Version of example 41 works 2025-08-06 09:41:41 -07:00
Will Pazner 30d975fb9b Don't need specializations of mfem::Swap for Array or Array2D 2025-08-05 15:39:29 -07:00
Will Pazner dd68bbe9fd Use explicitly defaulted copy and move operations in Array2D 2025-08-05 15:38:43 -07:00
Will Pazner d1efdf93be Use default move constructor and assignment in Array<T> 2025-08-05 15:38:14 -07:00
Will Pazner e95d59cdf1 Implement mfem::Swap in terms of std::swap 2025-08-05 15:37:49 -07:00
Will Pazner b8c1b5c2fa Use explicitly defaulted move and copy member functions in DenseMatrix 2025-08-05 15:37:13 -07:00
Will Pazner a30af2ed43 Variable name consistency in Memory
Documentation was 'other', code was 'orig'. Changed to 'other'.
2025-08-05 15:36:39 -07:00
Toni-ko bb467f97da First draft of the parallel version of ex41. Still some debugging to do. 2025-08-01 16:24:44 -07:00
Toni-ko 479433c871 Added a higher order IMEX Scheme - fixed some typos 2025-07-31 14:25:40 -07:00
Toni-ko 4269b06149 Changed SplitODESolver so that it inherits from ODESolver. Added an RK2 IMEX Method. 2025-07-30 15:27:47 -07:00
Will Pazner f5da02ce45 Fix bug in DenseTensor::NewMemoryAndSize 2025-07-29 13:11:13 -07:00
Toni-ko b8660af425 Implements SplitTimeDependentOperator class along with supporting methods. Implements an IMEX scheme for ex41. Solutions to ex41 look reasonable. 2025-07-28 16:55:23 -07:00
Will Pazner dc13e67e87 Add placeholder for example 41, DG convection-diffusion
Currently just a cleaned-up version of ex9. Will add diffusion and
IMEX time integration.
2025-07-25 11:24:16 -07:00
camierjs 39262f0376 Incompressible Schrödinger Flow Miniapp 2025-07-21 16:34:19 -07:00
Tzanio Kolev 1206e9c575 Merge branch 'master' into qf-project-gf-fallback 2025-05-03 13:27:56 -07:00
Will Pazner 86b16c8989 Add fallback for QuadratureFunction::ProjectGridFunction
The (slower) fallback will be used when QuadratureInterpolator is not supported
for the finite element space.
2025-04-29 10:35:44 -07:00
Dylan Copeland cc762f8f39 More CI build fixes. 2025-01-31 19:41:13 -08:00
dylan-copeland e9980453cc Fixes for single precision build. 2025-01-31 17:04:49 -08:00
dylan-copeland 816ac475e5 Fixes for -Wshadow. 2025-01-31 16:15:28 -08:00
Dylan Copeland b4fb21d3fa Fixes for clang. 2025-01-31 16:03:07 -08:00
Dylan Copeland 7c7dc1469e Fixes for another compiler. 2025-01-31 14:14:40 -08:00
dylan-copeland b7903bfc6b Fixed mac build. 2025-01-31 10:54:16 -08:00
Dylan Copeland ad2483d36e Templates for data types in serial matrices. 2025-01-31 10:31:38 -08:00
Dylan Copeland 738100e50a Template for vector data type. 2025-01-30 12:21:47 -08:00
Will Pazner d0a994fa02 Add Array<T>::NewMemoryAndSize
Analogous to Vector::NewMemoryAndSize.

Implement DenseTensor::NewMemoryAndSize in terms of Array<T>::NewMemoryAndSize.
2024-11-22 10:50:12 -08:00
Will Pazner 6e4f6d11be Merge remote-tracking branch 'origin/master' into table-array 2024-11-22 10:41:03 -08:00
Will Pazner 8c8b79fdd9 Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	general/table.cpp
#	general/table.hpp
2024-06-24 11:08:08 -07:00
Will Pazner 326acce08c Replace double with real_t in BatchLUFactor 2024-06-02 14:03:46 -07:00
Will Pazner 3d78f073ae Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	general/table.hpp
#	linalg/densemat.cpp
#	linalg/densemat.hpp
2024-06-02 13:25:08 -07:00
Will Pazner 9f1c1b811a Merge branch 'densemat-array' into table-array 2024-01-30 12:24:41 -08:00
Will Pazner 843365752f Fix Array copy/move bug in DeviceConformingProlongationOperator 2024-01-02 13:29:39 -08:00
Will Pazner cc21c6b859 Add move assignment operator to Array 2024-01-02 11:40:37 -08:00
Will Pazner d788f190ea Use Array<double> in DenseMatrix and DenseTensor 2023-12-13 21:39:36 -08:00
Will Pazner d0bb1a86ed Delete specialization of mfem::Swap for Table 2023-12-13 21:11:08 -08:00
Will Pazner fafec06958 Use std::move in mfem::Swap 2023-12-13 21:10:49 -08:00
Will Pazner e52a2b4dcd Use Array<int> instead of Memory<int> in Table
Also make some small Doxygen edits
2023-12-13 14:13:38 -08:00
214 changed files with 25505 additions and 5919 deletions
+26 -14
View File
@@ -92,6 +92,10 @@ examples/ex9.mesh
examples/ex9-mesh.*
examples/ex9-init.*
examples/ex9-final.*
examples/ex41.mesh
examples/ex41-mesh.*
examples/ex41-init.*
examples/ex41-final.*
examples/deformed.*
examples/velocity.*
examples/elastic_energy.*
@@ -223,6 +227,9 @@ miniapps/electromagnetics/Joule_[0-9]*
miniapps/electromagnetics/Lorentz_[0-9]*
miniapps/electromagnetics/Lorentz.dat
miniapps/fluids/schrodinger-flow/schrodinger_flow
miniapps/fluids/schrodinger-flow/pschrodinger_flow
miniapps/gslib/field-diff
miniapps/gslib/field-interp
miniapps/gslib/findpts
@@ -230,6 +237,7 @@ miniapps/gslib/pfindpts
miniapps/gslib/schwarz_ex1
miniapps/gslib/schwarz_ex1p
miniapps/gslib/interpolated.gf
miniapps/gslib/particles_redist
miniapps/meshing/mobius-strip
miniapps/meshing/klein-bottle
@@ -276,10 +284,8 @@ miniapps/meshing/refined.mesh
miniapps/meshing/bounding-box*
miniapps/meshing/jacobian-determinant*
miniapps/mtop/parheat
miniapps/mtop/ParHeat/*
miniapps/mtop/seqheat
miniapps/mtop/SeqHeat/*
miniapps/mtop/ParaView/
miniapps/mtop/mtop_test_iso_elasticity
miniapps/autodiff/paradiff
miniapps/autodiff/seqadiff
@@ -289,16 +295,19 @@ miniapps/autodiff/seq_example
miniapps/autodiff/seq_test
miniapps/autodiff/Example/*
miniapps/navier/navier_mms
miniapps/navier/navier_kovasznay
miniapps/navier/navier_kovasznay_vs
miniapps/navier/navier_tgv
miniapps/navier/navier_shear
miniapps/navier/navier_3dfoc
miniapps/navier/navier_turbchan
miniapps/navier/navier_cht
miniapps/navier/tgv_out*.txt
miniapps/navier/*_output
miniapps/fluids/navier/navier_mms
miniapps/fluids/navier/navier_kovasznay
miniapps/fluids/navier/navier_kovasznay_vs
miniapps/fluids/navier/navier_tgv
miniapps/fluids/navier/navier_shear
miniapps/fluids/navier/navier_3dfoc
miniapps/fluids/navier/navier_turbchan
miniapps/fluids/navier/navier_cht
miniapps/fluids/navier/navier_bifurcation
miniapps/fluids/navier/Navier_Bifurcation_[0-9]*
miniapps/fluids/navier/ParaView
miniapps/fluids/navier/tgv_out*.txt
miniapps/fluids/navier/*_output
miniapps/nurbs/nurbs_ex1
miniapps/nurbs/nurbs_ex1p
@@ -421,6 +430,9 @@ miniapps/tribol/contact-patch-test
miniapps/diag-smoothers/abs-l1-jacobi
miniapps/diag-smoothers/mg-abs-l1-jacobi
miniapps/contact/contact
miniapps/contact/ParaView
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
-2
View File
@@ -8,8 +8,6 @@
https://mfem.org
FIXME: this file needs to be updated
This directory contains most of the GitLab CI configuration. MFEM runs both PR
and nightly testing on GitLab.
+1 -1
View File
@@ -32,7 +32,7 @@ mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
# run
if [[ "${MACHINE_NAME}" == "dane" ]]; then
salloc --nodes=1 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
salloc --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
else
+128 -70
View File
@@ -8,9 +8,13 @@
https://mfem.org
Version 4.8.1 (development)
Version 4.9.1 (development)
===========================
Version 4.9, released on Dec 11, 2025
=====================================
Starting with this version, MFEM requires a C++17 compiler.
Discretization improvements
@@ -19,86 +23,132 @@ Discretization improvements
nonlinear finite element operators, based on Enzyme or dual numbers AD at
quadrature points. These features are part of the new mfem::future namespace
and some of the API can change in the future. See the new dFEM minimal surface
miniapp in the miniapps/dfem/ directory for illustration of dFEM's use.
miniapp in the miniapps/dfem/ directory for illustration of dFEM's use. Using
Enzyme for AD in MFEM is tested with clang v19 and requires clang/LLVM built
with plugin support. See INSTALL for more details.
- Using Enzyme for AD in MFEM is tested with clang v19 and requires clang/LLVM
built with plugin support. See INSTALL for more details.
- Introduced initial support for particle methods in MFEM with new classes
Particle, ParticleSet and ParticleVector.
* Particle is a convenient interface for individual particle data.
* ParticleSet manages and stores particle data in a struct-of-arrays form,
carrying particle coordinates and IDs along with an arbitrary number of
Vector and integer data for each particle.
* ParticleVector is a Vector-derived container that stores vector data for an
arbitrary number of particles contiguously based on specified vdim/ordering.
See the new particle miniapps in miniapps/gslib/ and miniapps/fluids/navier/.
- Added a new miniapp and specialized AMG solver (AMGF) for optimization-based
contact mechanics. The miniapp solves large-scale frictionless contact using a
self-contained Interior Point (IP) solver, mortar-based contact constraints
provided by Tribol. The resulting linear systems are solved with the new AMGF
solver (see below). Benchmark examples include the two-block, ironing, and
beam-sphere problems. See the miniapps/contact/ directory.
- Added support for boundary integration to the hyperbolic framework. Two new
classes BdrHyperbolicDirichletIntegrator and BoundaryHyperbolicFlowIntegrator
have been introduced for implementation of weak Dirichlet boundary conditions
with a general flux or for the linear case respectively.
- Added a method to compute piecewise linear bounds on high-order functions on
tensor-product elements.
- Added support for interior face integration enabling DG methods in
ParMixedBilinearForm, ParNonlinearForm and ParBlockNonlinearForm.
- In the ParMoonolith integration, added support for variational resampling of
H1 vector fields.
- Added support for boundary integration to the hyperbolic framework. In this
regard, new classes `BdrHyperbolicDirichletIntegrator` and
`BoundaryHyperbolicFlowIntegrator` have been introduced for implementation
of weak Dirichlet boundary conditions with a general flux or for the linear
case respectively.
- Added method to compute piecewise linear bounds on high-order functions on
tensor-product elements.
- Parallel anisotropic refinement of hexahedral meshes is now supported,
provided that neighboring hexahedra are not refined in conflicting directions.
A new ParMesh method is added to check for such conflicts, before refinement.
- Introduced IMEX ODE solvers based on a split-operator framework. Added
examples ex41 and ex41p demonstrating IMEX DG/CG discretizations of the
convectiondiffusion equation, with ex41p using DG LOR preconditioning.
Meshing improvements
--------------------
- The TMOP kernel hierarchy has been restructured to reduce compilation time.
Most large kernels have been split into smaller, specific ones, with kernels
for each metric. The directory structure has been updated with assemble,
metrics, mult and tools subdirectories. The new kernel dispatch and
specialization system has also been integrated.
Unit tests have been revised to ensure --all tests pass.
Most large kernels have been split into smaller specific kernels for each
metric. The directory structure has been updated with assemble, metrics, mult
and tools subdirectories. New kernel dispatch and specialization system has
also been integrated. Unit tests have been revised to ensure --all tests pass.
- Introduced NC-patch NURBS meshes, which are conforming element-wise but allow
for nonconforming patch topology. This new mesh format supports element
spacing formulas for refinement, as well as local refinement factors for a
subset of knot vectors.
- Added support for higher order meshes in Mesh::MakeSimplicial and
ParMesh::MakeSimplicial.
- Added a new miniapp for interpolating a surface grid of points in 3D using a
smooth NURBS surface, that can then be sampled at arbitrary resolution while
staying close to the original geometry. See miniapps/nurbs/nurbs_surface.
- Parallel anisotropic refinement of hexahedral meshes is now supported,
provided that neighboring hexahedra are not refined in conflicting directions.
A new ParMesh method is added to check for such conflicts, before refinement.
- Added support for higher order meshes in (Par)Mesh::MakeSimplicial.
Linear and nonlinear solvers
----------------------------
- Added FilteredSolver: a base class for solvers with filtering. It handles
cases where a solver performs well except in small subspaces, by adding a
filtering step formulated as a subspace correction.
- Added AMGFSolver: a derived class of FilteredSolver, specialized for AMG with
Filtering (AMGF), providing robust preconditioning for linear systems arising
in constrained optimization problems such as frictionless contact.
GPU computing
-------------
- The function Vector::SetSubVector(const Array<int> &, const real_t) now
executes on device if either the vector or the array have the device flag
set. This is most often used for setting constant essential boundary
conditions. A new function Vector::SetSubVectorHost has been added in cases
where host execution is always needed (e.g. when the DOFs array is small).
- Added the 'gpu', 'raja-gpu', and 'ceed-gpu' backend aliases/shortcuts which
automatically select between CUDA or HIP.
- Added the option to enable GPU-aware MPI in MFEM using the environment
variable 'MFEM_GPU_AWARE_MPI' set to any value. Setting this environment
variable is an alternative to calling 'Device::SetGPUAwareMPI(true)'.
- Implemented a GPU-accelerated matrix-free AMR derefinement GridFunction update
operator. This supports mixed geometry meshes and variable order spaces, and
is the default derefinement operator constructed by FiniteElementSpace::Update
and ParFiniteElementSpace::Update. The operator requires the finite element
space to be nonconforming.
- Introduced MFEM_FOREACH_THREAD_DIRECT, which directly maps loop tasks to GPU
threads, assigning one task per thread.
- Implemented a GPU-accelerated matrix-free AMR derefinement `GridFunction`
update operator. This supports mixed geometry meshes and variable order
spaces, and is the default derefinement operator constructed by
`FiniteElementSpace::Update` and `ParFiniteElementSpace::Update`.
The operator requires `FiniteElementSpace::Nonconforming() == true`.
- The function Vector::SetSubVector(const Array<int> &, const real_t) now
executes on device if either the vector or the array have the device flag
set. This is most often used for setting constant essential BCs. A new method,
SetSubVectorHost, has been added for cases where host execution is always
needed (e.g. when the DOFs array is small).
- Added GPU support in GradientGridFunction and InnerProduct Coefficient classes
by implementing their Project methods.
- Added new method: GridFunction::GetGradients, with GPU support, for computing
the gradients of a GridFunction on all elements.
- Added GPU support in GradientGridFunctionCoefficient and
InnerProductCoefficient by implementing their Project methods.
Linear and nonlinear solvers
----------------------------
- Added `FilteredSolver`: a base class for solvers with filtering. It handles cases
where a solver performs well except in small subspaces, by adding a filtering step
formulated as a subspace correction.
- Added `AMGFSolver`: a derived class of `FilteredSolver`, specialized for
AMG with Filtering (AMGF), providing robust preconditioning for linear systems
arising in constrained optimization problems such as frictionless contact.
- The CUDA-specific names used by some of the unit tests like 'cunit_tests' and
'pcunit_tests' were replaced by names using 'gpu' instead of 'c' (short for
CUDA) or 'cuda'. These tests automatically run the CUDA/HIP tests based on the
MFEM build configuration.
New and updated examples and miniapps
-------------------------------------
- Added miniapps to demonstrate an implementation of the absolute-value
L(1)-Jacobi preconditioners in partially assembled operators. This includes
Multigrid wrapper to demonstrate the effectiveness of these Jacobi-type
operators as smoothers.
These miniapps can be found in `miniapps/diag-smoothers`.
- Added the miniapps/fluids directory and moved the previous Navier and the new
incompressible Schrödinger flow miniapps into it.
- Introduced the new Incompressible Schrödinger Flow (ISF) miniapp, which models
inviscid fluid dynamics by solving the linear Schrödinger equation, leveraging
the hydrodynamical analogy to quantum mechanics.
- New particle-related miniapps:
* New transient Navier-Stokes fluid-particles solver NavierParticles in
miniapps/fluids/navier/navier_particles, for modeling tracer particles in
fluid flow, demonstrating use of the new ParticleSet class.
* New Navier miniapp, miniapps/fluids/navier/navier_bifurcation, showing the
use of NavierParticles in a 2D bifurcating channel flow.
* New FindPointsGSLIB miniapp, miniapps/gslib/particles_redist, showing
parallel-redistribution of particle data between MPI ranks.
* Particle visualization features in common/particles_extras for viewing
particle locations and trajectories (ParticleTrajectories) using GLVis.
- Added a new miniapp (meshing/mesh-bounding-boxes) that computes the bounding
boxes for each element of a given mesh, and the bounds on the determinant of
@@ -111,36 +161,44 @@ New and updated examples and miniapps
of a charged particle, subject to Lorentz forces, in electrostatic and/or
magnetostatic fields as computed by the volta or tesla miniapps.
API changes:
-----------
- mfem::internal::tensor and mfem::internal::dual have been moved to
mfem::future::tensor and mfem::future::dual.
- Added miniapps to demonstrate an implementation of the absolute-value
l1-Jacobi preconditioners in partially assembled operators. This includes
Multigrid wrapper to demonstrate the effectiveness of these Jacobi-type
operators as smoothers. See the miniapps/diag-smoothers/ directory.
- API addition: in class `Operator`, added virtual functions: `AbsMult`, and
`AbsMultTranspose`; in class `Vector`, added `Abs` and `Pow`.
- Updated the mtop miniapp with a GPU enabled forward and adjoint solver for
isotropic linear elasticity.
Miscellaneous
-------------
- Added the "gpu", "raja-gpu", and "ceed-gpu" backend aliases/shortcuts which
automatically select between CUDA or HIP.
- Introduced MFEM_FETCH_TPLS CMake option to enable downloading, configuring,
and building of TPLs alongside MFEM (currently supported TPLs are hypre,
METIS, and GSLIB).
- The CUDA-specific names used by some of the unit tests like 'cunit_tests' and
'pcunit_tests' were replaced by names using 'gpu' instead of 'c' (short for
CUDA) or 'cuda'. These tests automatically run the CUDA/HIP tests based on the
MFEM build configuration.
- Added quadrature function support to the VisIt and Conduit data collections.
- Added the option to enable GPU-aware MPI in MFEM using the environment
variable 'MFEM_GPU_AWARE_MPI' set to any value. Setting this environment
variable is an alternative to calling 'Device::SetGPUAwareMPI(true)'.
- Added access to the internal parallel matrix in Par(Mixed)BilinearForm and
related utility methods for elimination of BCs.
- FindPointsGSLIB has a new constructor that accepts the mesh object and
internally calls the Setup() method so users do not have to. The FreeData()
method has also been moved to the destructor so users do not need to manually
free-up the memory if the destructor is called before MPI_Finalize().
- Added parallel Address Sanitizer, serial and parallel Undefined Behavior
Sanitizer and serial Memory Sanitizer GitHub actions tests on Ubuntu.
- FindPointsGSLIB has a new constructor that accepts the mesh object and
internally calls the Setup() method so that the user does not have to.
The FreeData() method has also been moved to the destructor so the user does
not need to manually free-up the memory if the destructor is called before
MPI_Finalize().
API changes
-----------
- mfem::internal::tensor and mfem::internal::dual have been moved to
mfem::future::tensor and mfem::future::dual.
- API addition: in class Operator, added virtual functions: AbsMult, and
AbsMultTranspose; in class Vector, added Abs and Pow.
- ParBilinearForm::EliminateEssentialVDofsInRhs() has been deprecated in favor
of ParallelEliminateEssentialTDofsInRhs().
Version 4.8, released on Apr 9, 2025
====================================
+1 -1
View File
@@ -59,7 +59,7 @@ project(mfem NONE)
# Current version of MFEM, see also `makefile`.
# mfem_VERSION = (string)
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
set(${PROJECT_NAME}_VERSION 4.8.1)
set(${PROJECT_NAME}_VERSION 4.9.1)
# Prohibit in-source build
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
+8 -1
View File
@@ -129,6 +129,10 @@ The MFEM source code has the following structure:
│ ├── moonolith
│ ├── qinterp
│ └── tmop
│ | ├── assemble
│ | ├── metrics
│ | ├── mult
│ | └── tools
├── general
├── linalg
│ ├── batched
@@ -139,16 +143,19 @@ The MFEM source code has the following structure:
│ ├── adjoint
│ ├── autodiff
│ ├── common
│ ├── contact
│ ├── dfem
│ ├── dpg
│ ├── electromagnetics
│ ├── fluids
│ │ ├── navier
│ │ └── schrodinger-flow
│ ├── gslib
│ ├── hdiv-linear-solver
│ ├── hooke
│ ├── meshing
│ ├── mtop
│ ├── multidomain
│ ├── navier
│ ├── nurbs
│ ├── parelag
│ ├── performance
+1 -1
View File
@@ -101,7 +101,7 @@ $ cd ../miniapps
$ ls
CMakeLists.txt common meshing nurbs shifted toys
adjoint electromagnetics mtop parelag solvers
autodiff gslib navier performance tools
autodiff gslib fluids performance tools
```
And an example in "toys"
+14 -2
View File
@@ -85,6 +85,10 @@ groups_serial=(
"DPG miniapps:"
"miniapps/dpg"
"{acoustics,convection-diffusion,diffusion,maxwell}.cpp"'
'"isf"
"Schrodinger flow miniapps:"
"miniapps/fluids/schrodinger-flow"
"schrodinger_flow.cpp"'
'"gslib"
"GSLIB miniapps:"
"miniapps/gslib"
@@ -166,6 +170,10 @@ groups_parallel=(
"miniapps/electromagnetics"
"joule.cpp"'
# "{joule,maxwell,tesla,volta}.cpp"' # todo: multiline sample runs
'"isf"
"Schrodinger flow miniapps:"
"miniapps/fluids/schrodinger-flow"
"pschrodinger_flow.cpp"'
'"adjoint"
"Adjoint miniapps:"
"miniapps/adjoint"
@@ -191,7 +199,7 @@ groups_parallel=(
# todo: miniapps/multidomain
'"navier"
"Navier miniapps:"
"miniapps/navier"
"miniapps/fluids/navier"
"navier_cht.cpp"'
# todo: add other navier miniapps
'"nurbs"
@@ -281,6 +289,10 @@ groups_all=(
"miniapps/electromagnetics"
"joule.cpp"'
# "{joule,maxwell,tesla,volta}.cpp"' # todo: multiline sample runs
'"isf"
"Schrodinger flow miniapps:"
"miniapps/fluids/schrodinger-flow"
"{,p}schrodinger_flow.cpp"'
'"adjoint"
"Adjoint miniapps:"
"miniapps/adjoint"
@@ -308,7 +320,7 @@ groups_all=(
# todo: miniapps/multidomain
'"navier"
"Navier miniapps:"
"miniapps/navier"
"miniapps/fluids/navier"
"navier_cht.cpp"'
# todo: add other navier miniapps
'"nurbs"
+156
View File
@@ -0,0 +1,156 @@
MFEM mesh v1.0
#
# MFEM Geometry Types (see fem/geom.hpp):
#
# POINT = 0
# SEGMENT = 1
# TRIANGLE = 2
# SQUARE = 3
# TETRAHEDRON = 4
# CUBE = 5
# PRISM = 6
# PYRAMID = 7
#
dimension
2
elements
25
3 3 0 1 2 3
3 3 1 4 5 2
3 3 4 6 7 5
3 3 6 8 9 7
3 3 8 10 11 9
3 3 10 12 13 11
3 3 12 14 15 13
3 3 14 16 17 15
3 3 16 18 19 17
3 3 18 20 21 19
3 3 20 22 23 21
3 3 22 24 25 23
3 3 24 26 27 25
3 3 26 28 29 27
3 3 28 30 31 29
3 3 30 32 33 31
3 3 32 34 35 33
3 3 17 19 36 37
3 3 37 36 38 39
3 3 39 38 40 41
3 3 41 40 42 43
3 3 43 42 44 45
3 3 45 44 46 47
3 3 47 46 48 49
3 3 49 48 50 51
boundary
52
2 1 0 1
2 1 2 3
1 1 3 0
2 1 1 4
2 1 5 2
2 1 4 6
2 1 7 5
2 1 6 8
2 1 9 7
2 1 8 10
2 1 11 9
2 1 10 12
2 1 13 11
2 1 12 14
2 1 15 13
2 1 14 16
2 1 17 15
2 1 16 18
2 1 18 20
2 1 21 19
2 1 20 22
2 1 23 21
2 1 22 24
2 1 25 23
2 1 24 26
2 1 27 25
2 1 26 28
2 1 29 27
2 1 28 30
2 1 31 29
2 1 30 32
2 1 33 31
2 1 32 34
3 1 34 35
2 1 35 33
2 1 19 36
2 1 37 17
2 1 36 38
2 1 39 37
2 1 38 40
2 1 41 39
2 1 40 42
2 1 43 41
2 1 42 44
2 1 45 43
2 1 44 46
2 1 47 45
2 1 46 48
2 1 49 47
2 1 48 50
4 1 50 51
2 1 51 49
vertices
52
2
0 0
1 0
1 1
0 1
2 0
2 1
3 0
3 1
4 0
4 1
5 0
5 1
6 0
6 1
7 0
7 1
8 0
8 1
9 0
9 1
10 0
10 1
11 0
11 1
12 0
12 1
13 0
13 1
14 0
14 1
15 0
15 1
16 0
16 1
17 0
17 1
9 2
8 2
9 3
8 3
9 4
8 4
9 5
8 5
9 6
8 6
9 7
8 7
9 8
8 8
9 9
8 9
+4 -2
View File
@@ -48,7 +48,7 @@ PROJECT_NAME = MFEM
# could be handy for archiving the generated documentation or if some version
# control system is used.
PROJECT_NUMBER = v4.8.1
PROJECT_NUMBER = v4.9.1
# Using the PROJECT_BRIEF tag one can provide an optional one line description
# for a project that appears at the top of each page and should give viewer a
@@ -973,10 +973,13 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/miniapps/adjoint \
@MFEM_SOURCE_DIR@/miniapps/autodiff \
@MFEM_SOURCE_DIR@/miniapps/common \
@MFEM_SOURCE_DIR@/miniapps/contact \
@MFEM_SOURCE_DIR@/miniapps/dfem \
@MFEM_SOURCE_DIR@/miniapps/dpg \
@MFEM_SOURCE_DIR@/miniapps/dpg/util \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/fluids/navier \
@MFEM_SOURCE_DIR@/miniapps/fluids/schrodinger-flow \
@MFEM_SOURCE_DIR@/miniapps/gslib \
@MFEM_SOURCE_DIR@/miniapps/hdiv-linear-solver \
@MFEM_SOURCE_DIR@/miniapps/hooke \
@@ -987,7 +990,6 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/miniapps/meshing \
@MFEM_SOURCE_DIR@/miniapps/mtop \
@MFEM_SOURCE_DIR@/miniapps/multidomain \
@MFEM_SOURCE_DIR@/miniapps/navier \
@MFEM_SOURCE_DIR@/miniapps/nurbs \
@MFEM_SOURCE_DIR@/miniapps/parelag \
@MFEM_SOURCE_DIR@/miniapps/performance \
+4
View File
@@ -46,6 +46,7 @@ list(APPEND ALL_EXE_SRCS
ex38.cpp
ex39.cpp
ex40.cpp
ex41.cpp
)
if (MFEM_USE_MPI)
@@ -89,6 +90,7 @@ if (MFEM_USE_MPI)
ex37p.cpp
ex39p.cpp
ex40p.cpp
ex41p.cpp
)
endif()
@@ -131,6 +133,8 @@ if (MFEM_ENABLE_TESTING)
list(APPEND THIS_TEST_OPTIONS "-dg")
elseif(${TEST_NAME} MATCHES "ex37p*")
list(APPEND THIS_TEST_OPTIONS "-mi" "3")
elseif(${TEST_NAME} MATCHES "ex41p*")
list(APPEND THIS_TEST_OPTIONS "-tf" "1.0")
endif()
if (NOT (${TEST_NAME} MATCHES ".*p$"))
+1 -1
View File
@@ -412,7 +412,7 @@ void ReducedSystemOperator::Mult(const Vector &k, Vector &y) const
Operator &ReducedSystemOperator::GetGradient(const Vector &k) const
{
delete Jacobian;
Jacobian = Add(1.0, M->SpMat(), dt, S->SpMat());
Jacobian = Add((real_t)1.0, M->SpMat(), dt, S->SpMat());
add(*v, dt, k, w);
add(*x, dt, w, z);
SparseMatrix *grad_H = dynamic_cast<SparseMatrix *>(&H->GetGradient(z));
+1 -1
View File
@@ -476,7 +476,7 @@ void ReducedSystemOperator::Mult(const Vector &k, Vector &y) const
Operator &ReducedSystemOperator::GetGradient(const Vector &k) const
{
delete Jacobian;
SparseMatrix *localJ = Add(1.0, M->SpMat(), dt, S->SpMat());
SparseMatrix *localJ = Add((real_t)1.0, M->SpMat(), dt, S->SpMat());
add(*v, dt, k, w);
add(*x, dt, w, z);
localJ->Add(dt*dt, H->GetLocalGradient(z));
+1 -1
View File
@@ -323,7 +323,7 @@ void ConductionOperator::ImplicitSolve(const real_t dt,
// for du_dt, where K is linearized by using u from the previous timestep
if (!T)
{
T = Add(1.0, Mmat, dt, Kmat);
T = Add((real_t)1.0, Mmat, dt, Kmat);
current_dt = dt;
T_solver.SetOperator(*T);
}
+1 -1
View File
@@ -414,7 +414,7 @@ void ConductionOperator::ImplicitSolve(const real_t dt,
// for du_dt, where K is linearized by using u from the previous timestep
if (!T)
{
T = Add(1.0, Mmat, dt, Kmat);
T = Add((real_t)1.0, Mmat, dt, Kmat);
current_dt = dt;
T_solver.SetOperator(*T);
}
+1 -1
View File
@@ -139,7 +139,7 @@ void WaveOperator::ImplicitSolve(const real_t fac0, const real_t fac1,
// for d2udt2
if (!T)
{
T = Add(1.0, Mmat, fac0, Kmat);
T = Add((real_t)1.0, Mmat, fac0, Kmat);
T_solver.SetOperator(*T);
}
K->FullMult(u, z);
+52 -3
View File
@@ -56,6 +56,51 @@ void f_exact(const Vector &, Vector &);
real_t freq = 1.0, kappa;
int dim;
void SolveSingle(SparseMatrix &A, const Vector &B, Vector &X)
{
VectorMP<float> Bs, Xs;
const real_t *data = A.GetData();
const int n = A.GetI()[A.NumRows()];
int *Icopy = new int[A.NumRows() + 1];
int *Jcopy = new int[n];
float *sdata = new float[n];
for (int i=0; i<n; ++i)
{
sdata[i] = data[i];
Jcopy[i] = A.GetJ()[i];
}
for (int i=0; i<A.NumRows() + 1; ++i)
{
Icopy[i] = A.GetI()[i];
}
SparseMatrixMP<float> As(Icopy, Jcopy, sdata, A.NumRows(), A.NumCols());
Bs.SetSize(B.Size());
Xs.SetSize(X.Size());
for (int i=0; i<B.Size(); ++i)
{
Bs[i] = B[i];
}
for (int i=0; i<X.Size(); ++i)
{
Xs[i] = X[i];
}
GSSmootherMP<float> Ms(As);
PCG<float>(As, Ms, Bs, Xs, 1, 500, 1e-12, 0.0);
for (int i=0; i<X.Size(); ++i)
{
X[i] = Xs[i];
}
}
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
@@ -185,6 +230,7 @@ int main(int argc, char *argv[])
cout << "Size of linear system: " << A->Height() << endl;
/*
// 11. Solve the linear system A X = B.
if (pa) // Jacobi preconditioning in partial assembly mode
{
@@ -193,20 +239,23 @@ int main(int argc, char *argv[])
}
else
{
#ifndef MFEM_USE_SUITESPARSE
#ifndef MFEM_USE_SUITESPARSE
// 11. Define a simple symmetric Gauss-Seidel preconditioner and use it to
// solve the system Ax=b with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 500, 1e-12, 0.0);
#else
#else
// 11. If MFEM was compiled with SuiteSparse, use UMFPACK to solve the
// system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
#endif
}
*/
SolveSingle((SparseMatrix&)(*A), B, X);
// 12. Recover the solution as a finite element grid function.
a->RecoverFEMSolution(X, *b, x);
+12
View File
@@ -297,6 +297,18 @@ int main(int argc, char *argv[])
sol_sock << "solution\n" << *pmesh << x << flush;
}
VectorMP<float> xf(x.Size());
for (int i=0; i<x.Size(); ++i)
{
xf[i] = x[i];
}
if (myid == 0)
{
cout << "Norm of x " << x.Norml2() << endl;
cout << "Norm of xf " << xf.Norml2() << endl;
}
// 18. Free the used memory.
delete a;
delete sigma;
+589
View File
@@ -0,0 +1,589 @@
// MFEM Example 41
//
// Compile with: make ex41
//
// Sample runs:
// ex41
// ex41 -cg
// ex41 -m ../data/periodic-hexagon.mesh -p 0 -r 2 -dt 0.005 -tf 10
// ex41 -m ../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex41 -m ../data/periodic-hexagon.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex41 -m ../data/amr-quad.mesh -p 1 -r 2 -dt 0.002 -tf 9
// ex41 -m ../data/star-q3.mesh -p 1 -r 2 -dt 0.001 -tf 9
// ex41 -m ../data/star-mixed.mesh -p 1 -r 2 -dt 0.005 -tf 9
// ex41 -m ../data/disc-nurbs.mesh -p 1 -r 3 -dt 0.005 -tf 9
// ex41 -m ../data/disc-nurbs.mesh -p 2 -r 3 -dt 0.005 -tf 9
// ex41 -m ../data/periodic-square.mesh -p 3 -r 4 -dt 0.0025 -tf 9 -vs 20
// ex41 -m ../data/periodic-cube.mesh -p 0 -r 2 -o 2 -dt 0.01 -tf 8
//
// Device sample runs:
//
// Description: This example code solves the time-dependent advection-diffusion
// equation du/dt + v.grad(u) - a div(grad(u)) = 0, where v is a
// given fluid velocity, a is the diffusion coefficient, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), and the use of IMEX
// ODE time integrators.
//
// The option to use continuous finite elements is available too.
#include "mfem.hpp"
using namespace std;
using namespace mfem;
// Mesh bounding box
Vector bb_min, bb_max;
// Velocity coefficient
template<int problem=0>
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
real_t center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const real_t w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const real_t w = M_PI/2;
real_t d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
template<int problem=0>
real_t u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
real_t center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return exp(-40.*pow(X(0)-0.5,2));
case 2:
case 3:
{
real_t rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const real_t s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( std::erfc(w*(X(0)-cx-rx))*std::erfc(-w*(X(0)-cx+rx)) *
std::erfc(w*(X(1)-cy-ry))*std::erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
real_t x_ = X(0), y_ = X(1), rho, phi;
rho = std::hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const real_t f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
/// Solver for the implicit part of the ODE (the diffusion term).
/// Solves systems of the form: (M + dt*S) k = rhs.
class Implicit_Solver : public Solver
{
private:
SparseMatrix &M, &S, A;
CGSolver linear_solver;
BlockILU prec;
real_t dt;
public:
Implicit_Solver(SparseMatrix &M_, SparseMatrix &S_,
const FiniteElementSpace &fes)
: M(M_),
S(S_),
prec(fes.GetTypicalFE()->GetDof(),
BlockILU::Reordering::MINIMUM_DISCARDED_FILL),
dt(1.0)
{
linear_solver.iterative_mode = false;
linear_solver.SetRelTol(1e-9);
linear_solver.SetAbsTol(0.0);
linear_solver.SetMaxIter(100);
linear_solver.SetPrintLevel(0);
linear_solver.SetPreconditioner(prec);
}
void SetTimeStep(real_t dt_)
{
real_t ddt = dt-dt_;
real_t epsilon;
epsilon = std::numeric_limits<real_t>::epsilon();
epsilon*=10;
if (std::abs(ddt) > epsilon)
{
dt = dt_;
// Form operator A = M + dt*S
A = S;
A *= dt;
A += M;
// this will also call SetOperator on the preconditioner
linear_solver.SetOperator(A);
}
}
void SetOperator(const Operator &op) override
{
linear_solver.SetOperator(op);
}
void Mult(const Vector &x, Vector &y) const override
{
linear_solver.Mult(x, y);
}
};
/** A time-dependent operator for the right-hand side of the ODE. The weak
form of the advection-diffusion equation is M du/dt = K u - S u + b,
where M is the mass matrix, K and S are the advection and diffusion
matrices, and b describes the flow on the boundary. In the case of IMEX
evolution, the diffusion term is treated implicitly, and the advection
term is treated explicitly. */
class IMEX_Evolution : public TimeDependentOperator
{
private:
BilinearForm &M, &K, &S;
const Vector &b;
unique_ptr<Solver> M_prec;
CGSolver M_solver;
unique_ptr<Implicit_Solver> implicit_solver;
mutable Vector z;
public:
IMEX_Evolution(BilinearForm &M_, BilinearForm &K_, BilinearForm &S_,
const Vector &b_);
/// Evaluate k1=M^{-1}*G1(u,t); -> k1 = M^{-1}*(K*u + b)
void Mult1(const Vector &x, Vector &y) const;
/// Evaluate k2: M*k2 = G2(u+k2*dt,t); -> (M+S*dt)*k2=-S*u
void ImplicitSolve2(const real_t dt, const Vector &x, Vector &k);
void Mult(const Vector &x, Vector &y) const override
{
if (TimeDependentOperator::EvalMode::ADDITIVE_TERM_1 == GetEvalMode())
{
Mult1(x,y);
}
else
{
mfem_error("TimeDependentOperator::Mult() is not overridden!");
}
}
void ImplicitSolve(const real_t dt, const Vector &x, Vector &k) override
{
if (TimeDependentOperator::EvalMode::ADDITIVE_TERM_2 == GetEvalMode())
{
ImplicitSolve2(dt,x,k);
}
else
{
mfem_error("TimeDependentOperator::ImplicitSolve() is not overridden!");
}
}
};
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
int problem = 0;
const char *mesh_file = "../data/periodic-square.mesh";
int ref_levels = 2;
int order = 3;
int ode_solver_type = 64; //IMEXRK3(3,4,3)
real_t t_final = 10.0;
real_t dt = 0.01;
bool paraview = false;
bool cg = false;
int vis_steps = 50;
real_t diffusion_term = 0.01;
real_t kappa = -1.0;
real_t sigma = -1.0;
bool visualization = true;
bool visit = false;
bool binary = false;
int precision = 8;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ref_levels, "-r", "--refine",
"Number of times to refine the mesh uniformly.");
args.AddOption(&order, "-o", "--order", "Order of the finite elements.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
ODESolver::IMEXTypes.c_str());
args.AddOption(&t_final, "-tf", "--t-final", "Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step", "Time step.");
args.AddOption(&diffusion_term, "-dc", "--diffusion-coeff",
"Diffusion coefficient in the PDE.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&cg, "-cg", "--continuous-galerkin", "-dg",
"--discontinuous-galerkin",
"Use Continuous-Galerkin Finite elements (Default is DG)");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
if (kappa < 0)
{
kappa = (order+1)*(order+1);
}
args.PrintOptions(cout);
// 2. Read the mesh from the given mesh file. We can handle geometrically
// periodic meshes in this code.
Mesh mesh(mesh_file);
const int dim = mesh.Dimension();
// 3. Define the IMEX (Split) ODE solver used for time integration. The IMEX
// solvers currently available are: 61 - Forward Backward Euler,
// 62 - IMEXRK2(2,2,2), 63 - IMEXRK2(2,3,2), and 64 - IMEX_DIRK_RK3.
unique_ptr<ODESolver> ode_solver = ODESolver::SelectIMEX(ode_solver_type);
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement, where 'ref_levels' is a
// command-line parameter.
for (int lev = 0; lev < ref_levels; lev++) {mesh.UniformRefinement();}
if (mesh.NURBSext) {mesh.SetCurvature(max(order, 1));}
mesh.GetBoundingBox(bb_min, bb_max, max(order, 1));
// 5. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
FiniteElementCollection *fec = NULL;
if (cg)
{
fec = new H1_FECollection(order, dim);
}
else
{
fec = new DG_FECollection(order, dim, BasisType::GaussLobatto);
}
FiniteElementSpace fes(&mesh, fec);
cout << "Number of unknowns: " << fes.GetVSize() << endl;
// 6. Set up and assemble the bilinear and linear forms corresponding to the
// DG discretization. The DGTraceIntegrator involves integrals over mesh
// interior faces.
std::unique_ptr<VectorFunctionCoefficient> velocity;
if (0==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<0>));
}
else if (1==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<1>));
}
else if (2==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<2>));
}
else if (3==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<3>));
}
ConstantCoefficient diff_coeff(diffusion_term);
BilinearForm m(&fes);
BilinearForm k(&fes);
BilinearForm s(&fes);
Vector b(fes.GetTrueVSize());
b = 0.0; //The inflow on the boundaries is set to zero.
m.AddDomainIntegrator(new MassIntegrator);
constexpr real_t alpha = -1.0;
k.AddDomainIntegrator(new ConvectionIntegrator(*velocity, alpha));
s.AddDomainIntegrator(new DiffusionIntegrator(diff_coeff));
if (!cg)
{
k.AddInteriorFaceIntegrator(new NonconservativeDGTraceIntegrator(*velocity,
alpha));
k.AddBdrFaceIntegrator(new NonconservativeDGTraceIntegrator(*velocity, alpha));
s.AddInteriorFaceIntegrator(new DGDiffusionIntegrator(diff_coeff, sigma,
kappa));
s.AddBdrFaceIntegrator(new DGDiffusionIntegrator(diff_coeff, sigma, kappa));
}
int skip_zeros = 0;
m.Assemble(skip_zeros);
k.Assemble(skip_zeros);
s.Assemble(skip_zeros);
m.Finalize(skip_zeros);
k.Finalize(skip_zeros);
s.Finalize(skip_zeros);
// 7. Define the initial conditions.
std::unique_ptr<FunctionCoefficient> u0;
if (0==problem)
{
u0.reset(new FunctionCoefficient(u0_function<0>));
}
else if (1==problem)
{
u0.reset(new FunctionCoefficient(u0_function<1>));
}
else if (2==problem)
{
u0.reset(new FunctionCoefficient(u0_function<2>));
}
else if (3==problem)
{
u0.reset(new FunctionCoefficient(u0_function<3>));
}
GridFunction u(&fes);
u.ProjectCoefficient(*u0);
// Create data collection for solution output: either VisItDataCollection for
// ascii data files, or SidreDataCollection for binary data files.
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example41", &mesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example41", &mesh);
dc->SetPrecision(precision);
}
dc->RegisterField("solution", &u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
// 8. Set up paraview visualization, if desired.
unique_ptr<ParaViewDataCollection> pv;
if (paraview)
{
pv = make_unique<ParaViewDataCollection>("Example41", &mesh);
pv->SetPrefixPath("ParaView");
pv->RegisterField("solution", &u);
pv->SetLevelsOfDetail(order);
pv->SetDataFormat(VTKFormat::BINARY);
pv->SetHighOrderOutput(true);
pv->SetCycle(0);
pv->SetTime(0.0);
pv->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
visualization = false;
cout << "GLVis visualization disabled.\n";
}
else
{
sout.precision(precision);
sout << "solution\n" << mesh << u;
sout << "pause\n";
sout << flush;
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
// 9. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
IMEX_Evolution adv(m, k, s, b);
real_t t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
bool done = false;
for (int ti = 0; !done; )
{
real_t dt_real = min(dt, t_final - t);
ode_solver->Step(u, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
cout << "time step: " << ti << ", time: " << t << endl;
if (paraview)
{
pv->SetCycle(ti);
pv->SetTime(t);
pv->Save();
}
if (visualization)
{
sout << "solution\n" << mesh << u << flush;
}
if (visit)
{
dc->SetCycle(ti);
dc->SetTime(t);
dc->Save();
}
}
}
delete fec;
return 0;
}
// Implementation of class IMEX_Evolution
IMEX_Evolution::IMEX_Evolution(BilinearForm &M_, BilinearForm &K_,
BilinearForm &S_, const Vector &b_)
: TimeDependentOperator(M_.FESpace()->GetTrueVSize()),
M(M_), K(K_), S(S_), b(b_), z(height)
{
Array<int> ess_tdof_list;
if (M.GetAssemblyLevel() == AssemblyLevel::LEGACY)
{
M_prec = make_unique<DSmoother>(M.SpMat());
M_solver.SetOperator(M.SpMat());
implicit_solver = make_unique<Implicit_Solver>(M.SpMat(), S.SpMat(),
*M.FESpace());
}
else
{
MFEM_ABORT("Implicit time integration is not supported with partial assembly");
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void IMEX_Evolution::Mult1(const Vector &x, Vector &y) const
{
// Perform the explicit step
// y = M^{-1} (K x + b)
K.Mult(x, z);
z += b;
M_solver.Mult(z, y);
}
void IMEX_Evolution::ImplicitSolve2(const real_t dt, const Vector &x, Vector &k)
{
// Perform the implicit step
// solve for k, k = -(M+dt S)^{-1} S x
MFEM_VERIFY(implicit_solver != NULL,
"Implicit time integration is not supported with partial assembly");
S.Mult(x, z);
z.Neg();
implicit_solver->SetTimeStep(dt);
implicit_solver->Mult(z, k);
}
+737
View File
@@ -0,0 +1,737 @@
// MFEM Example 41 - Parallel Version
//
// Compile with: make ex41p
//
// Sample runs:
// mpirun -np 4 ex41p
// mpirun -np 4 ex41p -cg
// mpirun -np 4 ex41p -m ../data/periodic-hexagon.mesh -p 0 -dt 0.005 -tf 10
// mpirun -np 4 ex41p -m ../data/periodic-square.mesh -p 1 -dt 0.005 -tf 9
// mpirun -np 4 ex41p -m ../data/periodic-hexagon.mesh -p 1 -dt 0.005 -tf 9
// mpirun -np 4 ex41p -m ../data/star-q3.mesh -p 1 -rp 1 -dt 0.001 -tf 9
// mpirun -np 4 ex41p -m ../data/disc-nurbs.mesh -p 1 -rp 1 -dt 0.005 -tf 9
// mpirun -np 4 ex41p -m ../data/disc-nurbs.mesh -p 2 -rp 1 -dt 0.005 -tf 9
// mpirun -np 4 ex41p -m ../data/periodic-square.mesh -rp 2 -dt 0.0025 -tf 9 -vs 20
// mpirun -np 4 ex41p -m ../data/periodic-cube.mesh -p 0 -rs 2 -o 2 -dt 0.01 -tf 8
//
// Device sample runs:
//
// Description: This example code solves the time-dependent advection-diffusion
// equation du/dt + v.grad(u) - a div(grad(u)) = 0, where v is a
// given fluid velocity, a is the diffusion coefficient, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), DG-LOR Preconditioning
// and the use of IMEX ODE time integrators.
//
// The Option to use Continuous Finite Elements is available too.
#include "mfem.hpp"
using namespace std;
using namespace mfem;
// Mesh bounding box
Vector bb_min, bb_max;
// Velocity coefficient
template<int problem=0>
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
real_t center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const real_t w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const real_t w = M_PI/2;
real_t d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
template<int problem=0>
real_t u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
real_t center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return exp(-40.*pow(X(0)-0.5,2));
case 2:
case 3:
{
real_t rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const real_t s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( std::erfc(w*(X(0)-cx-rx))*std::erfc(-w*(X(0)-cx+rx)) *
std::erfc(w*(X(1)-cy-ry))*std::erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
real_t x_ = X(0), y_ = X(1), rho, phi;
rho = std::hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const real_t f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
class Implicit_Solver : public Solver
{
private:
HypreParMatrix &M, &S;
HypreParMatrix *A;
CGSolver linear_solver;
real_t dt;
SparseMatrix M_diag;
public:
Implicit_Solver(HypreParMatrix &M_, HypreParMatrix &S_,
const FiniteElementSpace &fes)
: M(M_),
S(S_),
A(nullptr),
linear_solver(M.GetComm()),
dt(1.0)
{
linear_solver.iterative_mode = false;
linear_solver.SetRelTol(1e-9);
linear_solver.SetAbsTol(0.0);
linear_solver.SetMaxIter(100);
linear_solver.SetPrintLevel(0);
M.GetDiag(M_diag);
}
void SetTimeStep(real_t dt_)
{
real_t ddt = dt-dt_;
// syncronize ddt across all processes
MPI_Comm comm = M.GetComm();
int myrank;
MPI_Comm_rank(comm, &myrank);
MPI_Bcast(&ddt, 1, MPI_DOUBLE, 0, comm);
real_t epsilon;
epsilon = std::numeric_limits<real_t>::epsilon();
// allow for some tolerance in the time stepping process
epsilon*=10;
if (fabs(ddt) > epsilon)
{
if (0==myrank)
{
cout << "Updating Implicit_Solver time step from " << dt
<< " to " << dt_ << endl;
}
delete A;
dt = dt_;
// Form operator A = M + dt*S
A = Add(dt, S, 1.0, M);
linear_solver.SetOperator(*A);
}
}
void SetOperator(const Operator &op) override
{
linear_solver.SetOperator(op);
}
void Mult(const Vector &x, Vector &y) const override
{
linear_solver.Mult(x, y);
}
void SetPreconditioner(Solver &precond)
{
linear_solver.SetPreconditioner(precond);
}
~Implicit_Solver() override
{
delete A;
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of the advection-diffusion equation is (M + dt S) du/dt = Su - K u + b
, where M and K are the mass and advection matrices, and b describes the
flow on the boundary. In the case of IMEX evolution, the diffusion term is
treated implicitly, and the advection term is treated explicitly. */
class IMEX_Evolution : public TimeDependentOperator
{
private:
OperatorHandle M, K, S, A;
const Vector &b;
Solver *M_prec;
CGSolver M_solver;
Implicit_Solver *implicit_solver;
LORSolver<HypreBoomerAMG>* lor_solver;
mutable Vector z;
mutable Vector w;
public:
IMEX_Evolution(ParBilinearForm &M_, ParBilinearForm &K_, ParBilinearForm &S_,
const Vector &b_, ParBilinearForm &A_);
virtual
~IMEX_Evolution()
{
delete implicit_solver;
delete lor_solver;
delete M_prec;
}
void Mult1(const Vector &x, Vector &y) const;
void ImplicitSolve2(const real_t dt, const Vector &x, Vector &k);
void Mult(const Vector &x, Vector &y) const override
{
if (TimeDependentOperator::EvalMode::ADDITIVE_TERM_1 == GetEvalMode())
{
Mult1(x,y);
}
else
{
mfem_error("TimeDependentOperator::Mult() is not overridden!");
}
}
void ImplicitSolve(const real_t dt, const Vector &x, Vector &k) override
{
if (TimeDependentOperator::EvalMode::ADDITIVE_TERM_2 == GetEvalMode())
{
ImplicitSolve2(dt,x,k);
}
else
{
mfem_error("TimeDependentOperator::ImplicitSolve() is not overridden!");
}
}
};
int main(int argc, char *argv[])
{
// 1. Initialize MPI and HYPRE.
Mpi::Init();
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
// 2. Parse command-line options.
int problem = 0;
const char *mesh_file = "../data/periodic-square.mesh";
int ser_ref_levels = 2;
int par_ref_levels = 0;
int order = 3;
int ode_solver_type = 64; // 61 - Forward Backward Euler
// 62 - IMEXRK2(2,2,2)
// 63 - IMEXRK2(2,3,2)
// 64 - IMEXRK3(3,4,3)
real_t t_final = 10.0;
real_t dt = 0.01;
bool paraview = false;
bool cg = false;
int vis_steps = 50;
bool adios2 = false;
bool binary = false;
real_t diffusion_term = 0.01;
real_t kappa = -1.0;
real_t sigma = -1.0;
bool visualization = true;
bool visit = false;
int precision = 16;
cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
"Number of times to refine the mesh uniformly in serial.");
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
"Number of times to refine the mesh uniformly in parallel.");
args.AddOption(&order, "-o", "--order",
"Order (degree) of the finite elements.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
ODESolver::IMEXTypes.c_str());
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
"Time step.");
args.AddOption(&diffusion_term, "-dc", "--diffusion-coeff",
"Diffusion coefficient in the PDE.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&adios2, "-adios2", "--adios2-streams", "-no-adios2",
"--no-adios2-streams",
"Save data using adios2 streams.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&cg, "-cg", "--continuous-galerkin", "-dg",
"--discontinuous-galerkin",
"Use Continuous-Galerkin Finite elements (Default is DG)");
args.Parse();
if (!args.Good())
{
if (Mpi::Root())
{
args.PrintUsage(cout);
}
return 1;
}
if (Mpi::Root())
{
args.PrintOptions(cout);
}
if (kappa < 0)
{
kappa = (order+1)*(order+1);
}
// 3. Read the mesh from the given mesh file. We can handle geometrically
// periodic meshes in this code.
Mesh *mesh = new Mesh(mesh_file);
const int dim = mesh->Dimension();
// 4. Define the IMEX (Split) ODE solver used for time integration. The IMEX
// solvers currently available are: 55 - Forward Backward Euler,
// 56 - IMEXRK2(2,2,2), 57 - IMEXRK2(2,3,2), and
unique_ptr<ODESolver> ode_solver = ODESolver::SelectIMEX(ode_solver_type);
// 5. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement, where 'ref_levels' is a
// command-line parameter.
for (int lev = 0; lev < ser_ref_levels; lev++) { mesh->UniformRefinement(); }
if (mesh->NURBSext)
{
mesh->SetCurvature(max(order, 1));
}
mesh->GetBoundingBox(bb_min, bb_max, max(order, 1));
// 6. Define the parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
for (int lev = 0; lev < par_ref_levels; lev++)
{
pmesh->UniformRefinement();
}
// 7. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
FiniteElementCollection *fec = NULL;
if (cg)
{
fec = new H1_FECollection(order, dim);
}
else
{
fec = new DG_FECollection(order, dim, BasisType::GaussLobatto);
}
ParFiniteElementSpace *fes = new ParFiniteElementSpace(pmesh, fec);
HYPRE_BigInt global_vSize = fes->GlobalTrueVSize();
if (Mpi::Root())
{
cout << "Number of unknowns: " << global_vSize << endl;
}
// 8. Set up and assemble the bilinear and linear forms corresponding to the
// DG discretization. The DGTraceIntegrator involves integrals over mesh
// interior faces.
std::unique_ptr<VectorFunctionCoefficient> velocity;
if (0==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<0>));
}
else if (1==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<1>));
}
else if (2==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<2>));
}
else if (3==problem)
{
velocity.reset(new VectorFunctionCoefficient(dim, velocity_function<3>));
}
ConstantCoefficient diff_coeff(diffusion_term);
ConstantCoefficient dt_diff_coeff(dt*diffusion_term);
ParBilinearForm *m = new ParBilinearForm(fes);
ParBilinearForm *k = new ParBilinearForm(fes);
ParBilinearForm *s = new ParBilinearForm(fes);
m->AddDomainIntegrator(new MassIntegrator());
constexpr real_t alpha = -1.0;
k->AddDomainIntegrator(new ConvectionIntegrator(*velocity, alpha));
s->AddDomainIntegrator(new DiffusionIntegrator(diff_coeff));
// For the preconditioner - create billinear form corresponding to
// operator (M + dt S)
ParBilinearForm *a = new ParBilinearForm(fes);
a->AddDomainIntegrator(new MassIntegrator);
a->AddDomainIntegrator(new DiffusionIntegrator(dt_diff_coeff));
if (!cg)
{
k->AddInteriorFaceIntegrator(new NonconservativeDGTraceIntegrator(*velocity,
alpha));
k->AddBdrFaceIntegrator(new NonconservativeDGTraceIntegrator(*velocity, alpha));
s->AddInteriorFaceIntegrator(new DGDiffusionIntegrator(diff_coeff, sigma,
kappa));
s->AddBdrFaceIntegrator(new DGDiffusionIntegrator(diff_coeff, sigma, kappa));
a->AddInteriorFaceIntegrator(new DGDiffusionIntegrator(dt_diff_coeff, sigma,
kappa));
a->AddBdrFaceIntegrator(new DGDiffusionIntegrator(dt_diff_coeff, sigma, kappa));
}
int skip_zeros = 0;
m->Assemble(skip_zeros);
k->Assemble(skip_zeros);
s->Assemble(skip_zeros);
a->Assemble();
m->Finalize(skip_zeros);
k->Finalize(skip_zeros);
s->Finalize(skip_zeros);
a->Finalize(skip_zeros);
HypreParVector b(fes);
b = 0.0;
// 9. Define the initial conditions. Set up visualization (if desired).
std::unique_ptr<FunctionCoefficient> u0;
if (0==problem)
{
u0.reset(new FunctionCoefficient(u0_function<0>));
}
else if (1==problem)
{
u0.reset(new FunctionCoefficient(u0_function<1>));
}
else if (2==problem)
{
u0.reset(new FunctionCoefficient(u0_function<2>));
}
else if (3==problem)
{
u0.reset(new FunctionCoefficient(u0_function<3>));
}
ParGridFunction *u = new ParGridFunction(fes);
u->ProjectCoefficient(*u0);
HypreParVector *U = u->GetTrueDofs();
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example41-Parallel", pmesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example41-Parallel", pmesh);
dc->SetPrecision(precision);
// To save the mesh using MFEM's parallel mesh format:
// dc->SetFormat(DataCollection::PARALLEL_FORMAT);
}
dc->RegisterField("solution", u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
ParaViewDataCollection *pd = NULL;
if (paraview)
{
pd = new ParaViewDataCollection("Example41P", pmesh);
pd->SetPrefixPath("ParaView");
pd->RegisterField("solution", u);
pd->SetLevelsOfDetail(order);
pd->SetDataFormat(VTKFormat::BINARY);
pd->SetHighOrderOutput(true);
pd->SetCycle(0);
pd->SetTime(0.0);
pd->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
if (Mpi::Root())
{
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
}
visualization = false;
if (Mpi::Root())
{
cout << "GLVis visualization disabled.\n";
}
}
else
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout.precision(precision);
sout << "solution\n" << *pmesh << *u;
sout << "pause\n";
sout << flush;
if (Mpi::Root())
{
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
}
#ifdef MFEM_USE_ADIOS2
ADIOS2DataCollection *adios2_dc = NULL;
if (adios2)
{
std::string postfix(mesh_file);
postfix.erase(0, std::string("../data/").size() );
postfix += "_o" + std::to_string(order);
const std::string collection_name = "ex41-p-" + postfix + ".bp";
adios2_dc = new ADIOS2DataCollection(MPI_COMM_WORLD, collection_name, pmesh);
// output data substreams are half the number of mpi processes
adios2_dc->SetParameter("SubStreams", std::to_string(num_procs/2) );
// adios2_dc->SetLevelsOfDetail(2);
adios2_dc->RegisterField("solution", u);
adios2_dc->SetCycle(0);
adios2_dc->SetTime(0.0);
adios2_dc->Save();
}
#endif
// 10. Define the time-dependent evolution operator describing the
// ODE right-hand side, and perform time-integration (looping
// over the time iterations, ti, with a time-step dt).
IMEX_Evolution adv(*m, *k, *s, b, *a);
real_t t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
bool done = false;
for (int ti = 0; !done; )
{
real_t dt_real = min(dt, t_final - t);
ode_solver->Step(*U, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
if (Mpi::Root())
{
cout << "time step: " << ti << ", time: " << t << endl;
}
*u = *U;
if (visualization)
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout << "solution\n" << *pmesh << *u << flush;
}
if (paraview)
{
pd->SetCycle(ti);
pd->SetTime(t);
pd->Save();
}
#ifdef MFEM_USE_ADIOS2
// transient solutions can be visualized with ParaView
if (adios2)
{
adios2_dc->SetCycle(ti);
adios2_dc->SetTime(t);
adios2_dc->Save();
}
#endif
}
}
// 11. Free the used memory.
delete pd;
delete U;
delete u;
delete a;
delete s;
delete k;
delete m;
delete fes;
delete pmesh;
delete dc;
delete fec;
return 0;
}
// Implementation of class IMEX_Evolution
IMEX_Evolution::IMEX_Evolution(ParBilinearForm &M_, ParBilinearForm &K_,
ParBilinearForm &S_, const Vector &b_, ParBilinearForm &A_)
: TimeDependentOperator(M_.ParFESpace()->GetTrueVSize()), b(b_),
M_solver(M_.ParFESpace()->GetComm()), z(height), w(height)
{
if (M_.GetAssemblyLevel()==AssemblyLevel::LEGACY)
{
M.Reset(M_.ParallelAssemble(), true);
K.Reset(K_.ParallelAssemble(), true);
S.Reset(S_.ParallelAssemble(), true);
}
else
{
M.Reset(&M_, false);
K.Reset(&K_, false);
S.Reset(&S_, false);
}
M_solver.SetOperator(*M);
Array<int> ess_tdof_list;
if (M_.GetAssemblyLevel() == AssemblyLevel::LEGACY)
{
A.Reset(A_.ParallelAssemble(), true);
HypreParMatrix &M_mat = *M.As<HypreParMatrix>();
HypreParMatrix &S_mat = *S.As<HypreParMatrix>();
HypreSmoother *hypre_prec = new HypreSmoother(M_mat, HypreSmoother::Jacobi);
M_prec = hypre_prec;
implicit_solver = new Implicit_Solver(M_mat, S_mat, *M_.FESpace());
lor_solver = new LORSolver<HypreBoomerAMG>(A_, ess_tdof_list);
lor_solver->GetSolver().SetSystemsOptions(A_.ParFESpace()->GetVDim(), true);
implicit_solver -> SetPreconditioner(*lor_solver);
}
else
{
MFEM_ABORT("Implicit time integration is not supported with partial assembly");
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void IMEX_Evolution::Mult1(const Vector &x, Vector &y) const
{
// Perform the explicit step
// y = M^{-1} (K x + b)
K->Mult(x, z);
z += b;
M_solver.Mult(z, y);
}
void IMEX_Evolution::ImplicitSolve2(const real_t dt, const Vector &x, Vector &k)
{
// Perform the implicit step
// solve for k, k = -(M+dt S)^{-1} S x
MFEM_VERIFY(implicit_solver != NULL,
"Implicit time integration is not supported with partial assembly");
S->Mult(x, z);
z*= -1.0;
implicit_solver->SetTimeStep(dt);
implicit_solver->Mult(z, k);
}
+6 -2
View File
@@ -22,11 +22,11 @@ MFEM_LIB_FILE = mfem_is_not_built
SEQ_EXAMPLES = ex0 ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 \
ex17 ex18 ex19 ex20 ex21 ex22 ex23 ex24 ex25 ex26 ex27 ex28 ex29 ex30 \
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40
ex31 ex33 ex34 ex36 ex37 ex38 ex39 ex40 ex41
PAR_EXAMPLES = ex0p ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p \
ex12p ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p ex24p \
ex25p ex26p ex27p ex28p ex29p ex30p ex31p ex32p ex33p ex34p ex35p ex36p \
ex37p ex39p ex40p
ex37p ex39p ex40p ex41p
SEQ_DEVICE_EXAMPLES = ex1 ex3 ex4 ex5 ex6 ex9 ex14 ex22 ex24 ex25 ex26 ex34
PAR_DEVICE_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex14p \
ex22p ex24p ex25p ex26p ex34p ex35p
@@ -157,6 +157,10 @@ ex37-test-seq: ex37
@$(call mfem-test,$<,, Serial example,-mi 3)
ex37p-test-par: ex37p
@$(call mfem-test,$<, $(RUN_MPI), Parallel example,-mi 3)
ex41-test-seq: ex41
@$(call mfem-test,$<,, Serial example,-tf 1.0)
ex41p-test-par: ex41p
@$(call mfem-test,$<, $(RUN_MPI), Parallel example,-tf 1.0)
# Testing: optional tests
ifeq ($(MFEM_USE_STRUMPACK),YES)
ex11p-test-strumpack: ex11p
+2
View File
@@ -179,6 +179,7 @@ set(SRCS
hyperbolic.cpp
integrator.cpp
bounds.cpp
particleset.cpp
)
set(HDRS
@@ -308,6 +309,7 @@ set(HDRS
hyperbolic.hpp
integrator.hpp
bounds.hpp
particleset.hpp
)
if (MFEM_USE_SIDRE)
+9 -5
View File
@@ -207,7 +207,8 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
}
}
PLBound::PLBound(FiniteElementSpace *fes, int ncp_i, int cp_type_i)
PLBound::PLBound(const FiniteElementSpace *fes, const int ncp_i,
const int cp_type_i)
{
MFEM_VERIFY(!fes->IsVariableOrder(),
"Variable order meshes not yet supported.");
@@ -264,7 +265,8 @@ PLBound::PLBound(FiniteElementSpace *fes, int ncp_i, int cp_type_i)
Setup(nb, ncp, b_type, cp_type, tol);
}
void PLBound::Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get1DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
real_t x,w;
intmin.SetSize(ncp);
@@ -346,7 +348,8 @@ void PLBound::Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get2DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
intmin.SetSize(ncp*ncp);
intmax.SetSize(ncp*ncp);
@@ -482,7 +485,8 @@ void PLBound::Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get3DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
int nb2 = nb*nb,
ncp2 = ncp*ncp,
@@ -624,7 +628,7 @@ void PLBound::Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::GetNDBounds(int rdim, Vector &coeff,
void PLBound::GetNDBounds(const int rdim, const Vector &coeff,
Vector &intmin, Vector &intmax) const
{
if (rdim == 1)
+9 -8
View File
@@ -9,8 +9,8 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_BOUND
#define MFEM_BOUND
#ifndef MFEM_BOUNDS
#define MFEM_BOUNDS
#include "../config/config.hpp"
#include "fespace.hpp"
@@ -89,7 +89,8 @@ public:
}
// Constructor
PLBound(FiniteElementSpace *fes, int ncp_i = -1, int cp_type_i = 0);
PLBound(const FiniteElementSpace *fes,
const int ncp_i = -1, const int cp_type_i = 0);
// Get minimum number of control points needed to bound the given bases
int GetMinimumPointsForGivenBases(int nb_i, int b_type_i,
@@ -105,7 +106,7 @@ public:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D/2D/3D.
void GetNDBounds(int rdim, Vector &coeff,
void GetNDBounds(const int rdim, const Vector &coeff,
Vector &intmin, Vector &intmax) const;
/// Get number of control points used to compute the bounds.
@@ -113,15 +114,15 @@ public:
private:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D.
void Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get1DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 2D.
void Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get2DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 3D.
void Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get3DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Setup matrix used to compute values at given 1D locations in [0,1]
/// for Bernstein bases.
@@ -133,4 +134,4 @@ private:
} // namespace mfem
#endif // MFEM_BOUND
#endif // MFEM_BOUNDS
-1
View File
@@ -33,7 +33,6 @@ class FiniteElement;
class FiniteElementSpace;
class ElementTransformation;
class IntegrationRule;
class Vector;
/** @brief Function that determines if a CEED kernel should be used, based on
the current mfem::Device configuration. */
+4 -3
View File
@@ -2027,7 +2027,7 @@ void CoefficientVector::Project(MatrixCoefficient &coeff, bool transpose)
{
if (auto *const_coeff = dynamic_cast<MatrixConstantCoefficient*>(&coeff))
{
SetConstant(const_coeff->GetMatrix());
SetConstant(const_coeff->GetMatrix(), transpose);
}
else if (auto *const_sym_coeff =
dynamic_cast<SymmetricMatrixConstantCoefficient*>(&coeff))
@@ -2088,7 +2088,7 @@ void CoefficientVector::SetConstant(const Vector &constant)
}
}
void CoefficientVector::SetConstant(const DenseMatrix &constant)
void CoefficientVector::SetConstant(const DenseMatrix &constant, bool transpose)
{
const int nq = (storage & CoefficientStorage::CONSTANTS) ? 1 : qs.GetSize();
const int width = constant.Width();
@@ -2101,7 +2101,8 @@ void CoefficientVector::SetConstant(const DenseMatrix &constant)
{
for (int i = 0; i < height; ++i)
{
(*this)[i + j*height + iq*vdim] = constant(i, j);
const real_t val = transpose ? constant(j,i) : constant(i,j);
(*this)[i + j*height + iq*vdim] = val;
}
}
}
+1 -1
View File
@@ -2520,7 +2520,7 @@ public:
void SetConstant(const Vector &constant);
/// Set this vector to the given constant matrix.
void SetConstant(const DenseMatrix &constant);
void SetConstant(const DenseMatrix &constant, bool transpose=false);
/// Set this vector to the given constant symmetric matrix.
void SetConstant(const DenseSymmetricMatrix &constant);
+269 -47
View File
@@ -70,8 +70,8 @@ ConduitDataCollection::~ConduitDataCollection()
void ConduitDataCollection::Save()
{
std::string dir_name = MeshDirectoryName();
int err = create_directory(dir_name, mesh, myid);
if (err)
int err_ = create_directory(dir_name, mesh, myid);
if (err_)
{
MFEM_ABORT("Error creating directory: " << dir_name);
}
@@ -88,6 +88,7 @@ void ConduitDataCollection::Save()
<< verify_info.to_json());
}
// wrap all grid functions
FieldMapConstIterator itr;
for ( itr = field_map.begin(); itr != field_map.end(); itr++)
{
@@ -103,6 +104,16 @@ void ConduitDataCollection::Save()
}
}
// wrap all quadrature functions
QFieldMapConstIterator qf_itr;
for ( qf_itr = q_field_map.begin(); qf_itr != q_field_map.end(); qf_itr++)
{
std::string name = qf_itr->first;
QuadratureFunction *qf = qf_itr->second;
QuadratureFunctionToBlueprintField(qf,
n_mesh["fields"][name]);
}
// save mesh data
SaveMeshAndFields(myid,
n_mesh,
@@ -157,6 +168,16 @@ ConduitDataCollection::SetProtocol(const std::string &protocol)
relay_protocol = protocol;
}
// Conduit data type id for the MFEM precision
constexpr conduit::index_t mfem_precision_conduit_id =
#if defined(MFEM_USE_DOUBLE)
CONDUIT_NATIVE_DOUBLE_ID;
#elif defined(MFEM_USE_SINGLE)
CONDUIT_NATIVE_FLOAT_ID;
#else
#error Unknown MFEM precision
#endif
//------------------------------
// begin static public methods
//------------------------------
@@ -206,42 +227,41 @@ ConduitDataCollection::BlueprintMeshToMesh(const Node &n_mesh,
// get the number of points
int num_verts = n_coordset_vals[0].dtype().number_of_elements();
// get vals for points
const double *verts_ptr = NULL;
const real_t *verts_ptr = NULL;
// the mfem mesh constructor needs coords with interleaved (aos) type
// ordering, even for 1d + 2d we always need 3 doubles b/c it uses
// Array<Vertex> and Vertex is a pod of 3 doubles. we check for this
// ordering, even for 1d + 2d we always need 3 real_t (double/float) b/c it
// uses Array<Vertex> and Vertex is a pod of 3 real_t. we check for this
// case, if we don't have it we convert the data
if (ndims == 3 &&
n_coordset_vals[0].dtype().is_double() &&
n_coordset_vals[0].dtype().id() == mfem_precision_conduit_id &&
blueprint::mcarray::is_interleaved(n_coordset_vals) )
{
// already interleaved mcarray of 3 doubles,
// already interleaved mcarray of 3 real_t (double/float),
// return ptr to beginning
verts_ptr = n_coordset_vals[0].value();
}
else
{
Node n_tmp;
// check all vals, if we don't have doubles convert
// to doubles
// check all vals, if we don't have real_t (double/float) convert
// to real_t
NodeConstIterator itr = n_coordset_vals.children();
while (itr.has_next())
{
const Node &c_vals = itr.next();
std::string c_name = itr.name();
if ( c_vals.dtype().is_double() )
if ( c_vals.dtype().id() == mfem_precision_conduit_id )
{
// zero copy current coords
n_tmp[c_name].set_external(c_vals);
}
else
{
// convert
c_vals.to_double_array(n_tmp[c_name]);
c_vals.to_data_type(mfem_precision_conduit_id, n_tmp[c_name]);
}
}
@@ -250,13 +270,13 @@ ConduitDataCollection::BlueprintMeshToMesh(const Node &n_mesh,
if (ndims < 3)
{
// add dummy z
n_tmp["z"].set(DataType::c_double(num_verts));
n_tmp["z"].set(DataType(mfem_precision_conduit_id, num_verts));
}
if (ndims < 2)
{
// add dummy y
n_tmp["y"].set(DataType::c_double(num_verts));
n_tmp["y"].set(DataType(mfem_precision_conduit_id, num_verts));
}
Node &n_conv_coords_vals = n_conv["coordsets"][coords_name]["values"];
@@ -452,7 +472,7 @@ ConduitDataCollection::BlueprintMeshToMesh(const Node &n_mesh,
// if nodes gf is attached later, it resets the space dim based
// on the gf's fes.
Mesh *mesh = new Mesh(// from coordset
const_cast<double*>(verts_ptr),
const_cast<real_t*>(verts_ptr),
num_verts,
// from topology
const_cast<int*>(elem_indices),
@@ -519,7 +539,7 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
// can't return a gf that zero copies the conduit data
Node n_conv;
const double *vals_ptr = NULL;
const real_t *vals_ptr = NULL;
int vdim = 1;
@@ -529,10 +549,10 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
{
vdim = n_field["values"].number_of_children();
// need to check that we have doubles and
// need to check that we have real_t (double/float) and
// cover supported layouts
if ( n_field["values"][0].dtype().is_double() )
if ( n_field["values"][0].dtype().id() == mfem_precision_conduit_id )
{
// check for contig
if (n_field["values"].is_contiguous())
@@ -556,27 +576,26 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
vals_ptr = n_conv["values"].child(0).value();
}
}
else // convert to doubles and use contig
else // convert to real_t (double/float) and use contig
{
Node n_tmp;
// check all vals, if we don't have doubles convert
// to doubles
// check all vals, if we don't have real_t (double/float) convert
// to real_t
NodeConstIterator itr = n_field["values"].children();
while (itr.has_next())
{
const Node &c_vals = itr.next();
std::string c_name = itr.name();
if ( c_vals.dtype().is_double() )
if ( c_vals.dtype().id() == mfem_precision_conduit_id )
{
// zero copy current coords
n_tmp[c_name].set_external(c_vals);
}
else
{
// convert
c_vals.to_double_array(n_tmp[c_name]);
c_vals.to_data_type(mfem_precision_conduit_id, n_tmp[c_name]);
}
}
@@ -589,14 +608,15 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
}
else
{
if (n_field["values"].dtype().is_double() &&
if (n_field["values"].dtype().id() == mfem_precision_conduit_id &&
n_field["values"].is_compact())
{
vals_ptr = n_field["values"].value();
}
else
{
n_field["values"].to_double_array(n_conv["values"]);
n_field["values"].to_data_type(mfem_precision_conduit_id,
n_conv["values"]);
vals_ptr = n_conv["values"].value();
}
}
@@ -620,14 +640,14 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
if (zero_copy)
{
res = new GridFunction(fes,const_cast<double*>(vals_ptr));
res = new GridFunction(fes,const_cast<real_t*>(vals_ptr));
}
else
{
// copy case, this constructor will alloc the space for the GF data
res = new GridFunction(fes);
// create an mfem vector that wraps the conduit data
Vector vals_vec(const_cast<double*>(vals_ptr),fes->GetVSize());
Vector vals_vec(const_cast<real_t*>(vals_ptr),fes->GetVSize());
// copy values into the result
(*res) = vals_vec;
}
@@ -639,6 +659,155 @@ ConduitDataCollection::BlueprintFieldToGridFunction(Mesh *mesh,
return res;
}
//---------------------------------------------------------------------------//
mfem::QuadratureFunction *
ConduitDataCollection::BlueprintFieldToQuadratureFunction(Mesh *mesh,
const Node &n_field,
bool zero_copy)
{
// n_conv holds converted data (when necessary for mfem api)
// if n_conv is used ( !n_conv.dtype().empty() ) we
// know that some data allocation was necessary, so we
// can't return a qf that zero copies the conduit data
Node n_conv;
const real_t *vals_ptr = NULL;
int vdim = 1;
if (n_field["values"].dtype().is_object())
{
vdim = n_field["values"].number_of_children();
// need to check that we have real_t (double/float) and
// cover supported layouts
if ( n_field["values"][0].dtype().id() == mfem_precision_conduit_id )
{
// quad funcs use what mfem calls byVDIM
// and what conduit calls interleaved
// check for interleaved
if (blueprint::mcarray::is_interleaved(n_field["values"]))
{
// conduit mcarray interleaved == mfem byVDIM
vals_ptr = n_field["values"].child(0).value();
}
else
{
// for mcarray generic case -- default to byVDIM
// aka interleaved
blueprint::mcarray::to_interleaved(n_field["values"],
n_conv["values"]);
vals_ptr = n_conv["values"].child(0).value();
}
}
else // convert to real_t (double/float) and use interleaved
{
Node n_tmp;
// check all vals, if we don't have real_t (double/float) convert
// to real_t
NodeConstIterator itr = n_field["values"].children();
while (itr.has_next())
{
const Node &c_vals = itr.next();
std::string c_name = itr.name();
if ( c_vals.dtype().id() == mfem_precision_conduit_id )
{
// zero copy current coords
n_tmp[c_name].set_external(c_vals);
}
else
{
// convert
c_vals.to_data_type(mfem_precision_conduit_id, n_tmp[c_name]);
}
}
// for mcarray generic case -- default to byVDIM
// aka interleaved
blueprint::mcarray::to_interleaved(n_tmp,
n_conv["values"]);
vals_ptr = n_conv["values"].child(0).value();
}
}
else // scalar case
{
if (n_field["values"].dtype().id() == mfem_precision_conduit_id &&
n_field["values"].is_compact())
{
vals_ptr = n_field["values"].value();
}
else
{
n_field["values"].to_data_type(mfem_precision_conduit_id,
n_conv["values"]);
vals_ptr = n_conv["values"].value();
}
}
if (zero_copy && !n_conv.dtype().is_empty())
{
//Info: "Cannot zero-copy since data conversions were necessary"
zero_copy = false;
}
// we need basis name to create the proper mfem quad space and quad func
// the pattern used to encode the quad space params is:
// QF_{ORDER}_{VDIM}
// ORDER is the degree of the polynomials for the quad rule
// VDIM is the number of components at each quad point (scalar, vector, etc)
int qf_order = 0;
int qf_vdim = 0;
std::string qf_name = n_field["basis"].as_string();
const char *qf_name_cstr = qf_name.c_str();
if (!strncmp(qf_name_cstr, "QF_", 3))
{
// parse {ORDER}
qf_order = atoi(qf_name_cstr + 3);
// find second `_`
const char *qf_vdim_cstr = strstr(qf_name_cstr+3,"_");
if (qf_vdim_cstr == NULL)
{
MFEM_ABORT("Error parsing quadrature function description string: "
<< qf_name << std::endl
<< "Expected: QF_{ORDER}_{VDIM}");
}
// parse {VDIM}
qf_vdim = atoi(qf_vdim_cstr+1);
}
else
{
MFEM_ABORT("Error parsing quadrature function description string: "
<< qf_name << std::endl
<< "Expected: QF_{ORDER}_{VDIM}");
}
MFEM_VERIFY(qf_vdim == vdim, "vector dimension mismatch: vdim = " << vdim
<< ", qf_vdim = " << qf_vdim);
mfem::QuadratureSpace *quad_space = new mfem::QuadratureSpace(mesh, qf_order);
mfem::QuadratureFunction *res = new mfem::QuadratureFunction();
if (zero_copy)
{
res->SetSpace(quad_space, const_cast<real_t*>(vals_ptr), vdim);
res->SetOwnsSpace(true);
}
else
{
res->SetSpace(quad_space, vdim);
res->SetOwnsSpace(true);
// copy case, this constructor will alloc the space for the quad data
// create an mfem vector that wraps the conduit data
Vector vals_vec(const_cast<real_t*>(vals_ptr),res->Size());
// copy values into the result
(*res) = vals_vec;
}
return res;
}
//---------------------------------------------------------------------------//
void
ConduitDataCollection::MeshToBlueprintMesh(Mesh *mesh,
@@ -656,20 +825,20 @@ ConduitDataCollection::MeshToBlueprintMesh(Mesh *mesh,
// Setup main coordset
////////////////////////////////////////////
// Assumes mfem::Vertex has the layout of a double array.
// Assumes mfem::Vertex has the layout of a real_t (double/float) array.
// this logic assumes an mfem vertex is always 3 doubles wide
// this logic assumes an mfem vertex is always 3 real_t (double/float) wide
int stride = sizeof(mfem::Vertex);
int num_vertices = mesh->GetNV();
MFEM_ASSERT( ( stride == 3 * sizeof(double) ),
MFEM_ASSERT( ( stride == 3 * sizeof(real_t) ),
"Unexpected stride for Vertex");
Node &n_mesh_coords = n_mesh["coordsets"][coordset_name];
n_mesh_coords["type"] = "explicit";
double *coords_ptr = mesh->GetVertex(0);
real_t *coords_ptr = mesh->GetVertex(0);
n_mesh_coords["values/x"].set_external(coords_ptr,
num_vertices,
@@ -680,14 +849,14 @@ ConduitDataCollection::MeshToBlueprintMesh(Mesh *mesh,
{
n_mesh_coords["values/y"].set_external(coords_ptr,
num_vertices,
sizeof(double),
sizeof(real_t),
stride);
}
if (dim >= 3)
{
n_mesh_coords["values/z"].set_external(coords_ptr,
num_vertices,
sizeof(double) * 2,
sizeof(real_t) * 2,
stride);
}
@@ -942,6 +1111,59 @@ ConduitDataCollection::GridFunctionToBlueprintField(mfem::GridFunction *gf,
}
//---------------------------------------------------------------------------//
void
ConduitDataCollection::QuadratureFunctionToBlueprintField(
mfem::QuadratureFunction *qf,
Node &n_field,
const std::string &main_topology_name)
{
// For quadrature functions, use basis pattern:
// QF_{ORDER}_{VDIM}
int qf_vdim = qf->GetVDim();
int qf_order = qf->GetSpace()->GetOrder();
int qf_size = qf->GetSpace()->GetSize();
{
std::ostringstream oss;
oss << "QF_" << qf_order << "_" << qf_vdim;
n_field["basis"] = oss.str();
n_field["topology"] = main_topology_name;
}
if (qf_vdim == 1) // scalar case
{
n_field["values"].set_external(const_cast<real_t *>(qf->HostRead()),
qf_size);
}
else // vector case
{
// deal with striding of all components
// quadrature functions are always byVDIM
// or what conduit calls interleaved
index_t offset = 0;
index_t stride = sizeof(real_t) * qf_vdim;
for (int d = 0; d < qf_vdim; d++)
{
std::ostringstream oss;
oss << "v" << d;
std::string comp_name = oss.str();
n_field["values"][comp_name].set_external(const_cast<real_t *>(qf->HostRead()),
qf_size,
offset,
stride);
offset += sizeof(real_t);
}
}
}
//------------------------------
// end static public methods
//------------------------------
@@ -967,7 +1189,7 @@ ConduitDataCollection::RootFileName()
//---------------------------------------------------------------------------//
std::string
ConduitDataCollection::MeshFileName(int domain_id,
const std::string &relay_protocol)
const std::string &relay_protocol_)
{
std::string res = prefix_path +
name +
@@ -976,7 +1198,7 @@ ConduitDataCollection::MeshFileName(int domain_id,
"/domain_" +
to_padded_string(domain_id, pad_digits_rank) +
"." +
relay_protocol;
relay_protocol_;
return res;
}
@@ -994,7 +1216,7 @@ ConduitDataCollection::MeshDirectoryName()
//---------------------------------------------------------------------------//
std::string
ConduitDataCollection::MeshFilePattern(const std::string &relay_protocol)
ConduitDataCollection::MeshFilePattern(const std::string &relay_protocol_)
{
std::ostringstream oss;
oss << name
@@ -1003,7 +1225,7 @@ ConduitDataCollection::MeshFilePattern(const std::string &relay_protocol)
<< "/domain_%0"
<< pad_digits_rank
<< "d."
<< relay_protocol;
<< relay_protocol_;
return oss.str();
}
@@ -1013,14 +1235,14 @@ ConduitDataCollection::MeshFilePattern(const std::string &relay_protocol)
void
ConduitDataCollection::SaveRootFile(int num_domains,
const Node &n_mesh,
const std::string &relay_protocol)
const std::string &relay_protocol_)
{
// default to json root file, except for hdf5 case
std::string root_proto = "json";
if (relay_protocol == "hdf5")
if (relay_protocol_ == "hdf5")
{
root_proto = relay_protocol;
root_proto = relay_protocol_;
}
Node n_root;
@@ -1051,14 +1273,14 @@ ConduitDataCollection::SaveRootFile(int num_domains,
}
}
// add extra header info
n_root["protocol/name"] = relay_protocol;
n_root["protocol/name"] = relay_protocol_;
n_root["protocol/version"] = "0.3.1";
// we will save one file per domain, so trees == files
n_root["number_of_files"] = num_domains;
n_root["number_of_trees"] = num_domains;
n_root["file_pattern"] = MeshFilePattern(relay_protocol);
n_root["file_pattern"] = MeshFilePattern(relay_protocol_);
n_root["tree_pattern"] = "";
// Add the time, time step, and cycle
@@ -1073,9 +1295,9 @@ ConduitDataCollection::SaveRootFile(int num_domains,
void
ConduitDataCollection::SaveMeshAndFields(int domain_id,
const Node &n_mesh,
const std::string &relay_protocol)
const std::string &relay_protocol_)
{
relay::io::save(n_mesh, MeshFileName(domain_id, relay_protocol));
relay::io::save(n_mesh, MeshFileName(domain_id, relay_protocol_));
}
//---------------------------------------------------------------------------//
@@ -1172,13 +1394,13 @@ ConduitDataCollection::LoadRootFile(Node &root_out)
//---------------------------------------------------------------------------//
void
ConduitDataCollection::LoadMeshAndFields(int domain_id,
const std::string &relay_protocol)
const std::string &relay_protocol_)
{
// Note: This path doesn't use any info from the root file
// it uses the implicit mfem ConduitDataCollection layout
Node n_mesh;
relay::io::load( MeshFileName(domain_id, relay_protocol), n_mesh);
relay::io::load( MeshFileName(domain_id, relay_protocol_), n_mesh);
Node verify_info;
+33 -7
View File
@@ -50,11 +50,11 @@ namespace mfem
Those that construct MFEM objects from Conduit Nodes (Conduit Blueprint to
MFEM) provide a zero-copy option. Zero-copy is only possible if the
blueprint data matches the data types provided by the MFEM API, for example:
ints for connectivity arrays, doubles for field value arrays, allocations
that match MFEM's striding options, etc. If these constraints are not met,
MFEM objects that own the data are created and returned. In either case
pointers to new MFEM object instances are returned, the zero-copy only
applies to data backing the MFEM object instances.
ints for connectivity arrays, real_t (double/float) for field value arrays,
allocations that match MFEM's striding options, etc. If these constraints
are not met, MFEM objects that own the data are created and returned. In
either case pointers to new MFEM object instances are returned, the
zero-copy only applies to data backing the MFEM object instances.
@note QuadratureFunction%s (q-fields) are not supported.
@@ -183,6 +183,21 @@ public:
conduit::Node &out,
const std::string &main_topology_name = "main");
/// Describes a MFEM quadrature function using the mesh blueprint
/** Sets up passed conduit::Node out to describe the given quadrature function
using the mesh field blueprint.
Zero-copies as much data as possible.
@a main_toplogy_name is used to set the associated topology name.
With the default setting, the resulting field is associated with the
topology `main`.
*/
static void QuadratureFunctionToBlueprintField(QuadratureFunction *qf,
conduit::Node &out,
const std::string &main_topology_name = "main");
/// Constructs and MFEM mesh from a Conduit Blueprint Description
/** @a main_topology_name is used to select which topology to use, when
empty ("") the first topology entry will be used.
@@ -190,7 +205,7 @@ public:
If zero_copy == true, tries to construct a mesh that points to the data
described by the conduit node. This is only possible if the data in the
node matches the data types needed for the MFEM API (ints for
connectivity, doubles for field values, etc). If these constraints are
connectivity, real_t for field values, etc). If these constraints are
not met, a mesh that owns the data is created and returned.
*/
static Mesh *BlueprintMeshToMesh(const conduit::Node &n_mesh,
@@ -200,7 +215,7 @@ public:
/// Constructs and MFEM Grid Function from a Conduit Blueprint Description
/** If zero_copy == true, tries to construct a grid function that points to
the data described by the conduit node. This is only possible if the data
in the node matches the data types needed for the MFEM API (doubles for
in the node matches the data types needed for the MFEM API (real_t for
field values, allocated in soa or aos ordering, etc). If these
constraints are not met, a grid function that owns the data is created
and returned.
@@ -208,6 +223,17 @@ public:
static GridFunction *BlueprintFieldToGridFunction(Mesh *mesh,
const conduit::Node &n_field,
bool zero_copy = false);
/// Constructs and MFEM Quadrature Function from a Conduit Blueprint Description
/** If zero_copy == true, tries to construct a quadrature function that points to
the data described by the conduit node. This is only possible if the data
in the node matches the data types needed for the MFEM API (real_t for
field values, allocated in an interleavred/byVDIM order, etc). If these
constraints are not met, a grid function that owns the data is created
and returned.
*/
static QuadratureFunction *BlueprintFieldToQuadratureFunction(Mesh *mesh,
const conduit::Node &n_field,
bool zero_copy = false);
private:
/// Converts from MFEM element type enum to mesh bp shape name
+40 -5
View File
@@ -430,7 +430,9 @@ void VisItDataCollection::RegisterField(const std::string& name,
}
DataCollection::RegisterField(name, gf);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD,
gf->FESpace()->FEColl()->Name(),
gf->FESpace()->FEColl()->GetOrder());
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
@@ -449,7 +451,14 @@ void VisItDataCollection::RegisterQField(const std::string& name,
}
DataCollection::RegisterQField(name, qf);
field_info_map[name] = VisItFieldInfo("elements", 1, LOD);
// For quadrature functions, use basis pattern:
// QF_{ORDER}_{VDIM}
int qf_vdim = qf->GetVDim();
int qf_order = qf->GetSpace()->GetOrder();
std::ostringstream oss;
oss << "QF_" << qf_order << "_" << qf_vdim;
field_info_map[name] = VisItFieldInfo("quadrature", qf->GetVDim(), LOD,
oss.str(), qf_order);
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
@@ -623,7 +632,8 @@ void VisItDataCollection::LoadFields()
{
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
}
else if ((it->second).association == "elements")
else if ((it->second).association == "elements" || // old style
(it->second).association == "quadrature") // new style
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
@@ -637,7 +647,8 @@ void VisItDataCollection::LoadFields()
it->first,
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
}
else if ((it->second).association == "elements")
else if ((it->second).association == "elements" || // old style
(it->second).association == "quadrature") // new style
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
@@ -676,6 +687,8 @@ std::string VisItDataCollection::GetVisItRootString()
ftags["assoc"] = picojson::value((it->second).association);
ftags["comps"] = picojson::value(to_string((it->second).num_components));
ftags["lod"] = picojson::value(to_string((it->second).lod));
ftags["basis"] = picojson::value((it->second).basis);
ftags["order"] = picojson::value(to_string((it->second).order));
field["path"] = picojson::value(path_str + it->first + file_ext_format);
field["tags"] = picojson::value(ftags);
fields[it->first] = picojson::value(field);
@@ -752,9 +765,31 @@ void VisItDataCollection::ParseVisItRootString(const std::string& json)
it != fields_obj.end(); ++it)
{
picojson::value tags = it->second.get("tags");
// defaults that allow us to parse older mfem_root files
int lod = 1;
std::string basis = "";
int order = -1;
if (tags.contains("lod"))
{
lod = to_int(tags.get("lod").get<std::string>());
}
if (tags.contains("basis"))
{
basis = tags.get("comps").get<std::string>();
}
if (tags.contains("order"))
{
order = to_int(tags.get("comps").get<std::string>());
}
field_info_map[it->first] =
VisItFieldInfo(tags.get("assoc").get<std::string>(),
to_int(tags.get("comps").get<std::string>()));
to_int(tags.get("comps").get<std::string>()),
lod, basis, order);
}
}
}
+12 -6
View File
@@ -408,12 +408,18 @@ public:
class VisItFieldInfo
{
public:
std::string association;
int num_components;
int lod;
VisItFieldInfo() { association = ""; num_components = 0; lod = 1;}
VisItFieldInfo(std::string association_, int num_components_, int lod_ = 1)
{ association = association_; num_components = num_components_; lod =lod_;}
std::string association = "";
int num_components = 0;
int lod = 1;
std::string basis = "";
int order = -1;
VisItFieldInfo() = default;
VisItFieldInfo(std::string association_, int num_components_, int lod_ = 1,
std::string basis_ = "", int order_ = -1)
{
association = association_; num_components = num_components_; lod =lod_;
basis = basis_; order = order_;
}
};
/// Data collection with VisIt I/O routines
+403
View File
@@ -0,0 +1,403 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "util.hpp"
namespace mfem::future
{
/// @brief Assemble element matrix for three dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_t3d(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, const real_t>& qpdc,
const DeviceTensor<1, const real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& q1d,
const int& td1d)
{
constexpr int dimension = 3;
// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, num_qp]
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
// [num_test_dof, ...]
const auto num_test_dof = A.GetShape()[0];
for (int Jx = 0; Jx < td1d; Jx++)
{
for (int Jy = 0; Jy < td1d; Jy++)
{
for (int Jz = 0; Jz < td1d; Jz++)
{
const int J = Jx + td1d * (Jy + td1d * Jz);
for (int j = 0; j < trial_vdim; j++)
{
for (int tv = 0; tv < test_vdim; tv++)
{
for (int tod = 0; tod < test_op_dim; tod++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
fhat(tv, tod, q) = 0.0;
}
}
}
}
}
// MSVC lambda capture workaround
[[maybe_unused]] const auto& inputs_ref = inputs;
int m_offset = 0;
for_constexpr<num_inputs>([&](auto s)
{
using fop_t = std::decay_t<decltype(get<s>(inputs_ref))>;
const int trial_op_dim = static_cast<int>(itod(static_cast<int>(s)));
if (trial_op_dim == 0)
{
// This is inside a lambda so we have to return
// instead of idiomatic 'continue'.
return;
}
auto& B = input_dtqmaps[s].B;
auto& G = input_dtqmaps[s].G;
if constexpr (is_value_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy) * B(qz, 0, Jz);
}
}
}
}
}
}
}
else if constexpr (is_gradient_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
if (m == 0)
{
fhat(i, k, q) += f * G(qx, 0, Jx) * B(qy, 0, Jy) * B(qz, 0, Jz);
}
else if (m == 1)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * G(qy, 0, Jy) * B(qz, 0, Jz);
}
else if (m == 2)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy) * G(qz, 0, Jz);
}
}
}
}
}
}
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("sum factorized sparse matrix assemble routine "
"not implemented for field operator");
#endif
}
MFEM_SYNC_THREAD;
m_offset += trial_op_dim;
});
auto bvtfhat = Reshape(&A(0, 0, J, j), num_test_dof, test_vdim);
map_quadrature_data_to_fields(bvtfhat, fhat, output, output_dtqmap,
scratch_shmem, dimension, true);
}
}
}
}
}
/// @brief Assemble element matrix for two dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_t2d(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, const real_t>& qpdc,
const DeviceTensor<1, const real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& q1d,
const int& td1d)
{
constexpr int dimension = 2;
// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, num_qp]
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
// [num_test_dof, ...]
const auto num_test_dof = A.GetShape()[0];
for (int Jx = 0; Jx < td1d; Jx++)
{
for (int Jy = 0; Jy < td1d; Jy++)
{
const int J = Jy + Jx * td1d;
for (int j = 0; j < trial_vdim; j++)
{
for (int tv = 0; tv < test_vdim; tv++)
{
for (int tod = 0; tod < test_op_dim; tod++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
fhat(tv, tod, q) = 0.0;
}
}
}
}
// MSVC lambda capture workaround
[[maybe_unused]] const auto& inputs_ref = inputs;
int m_offset = 0;
for_constexpr<num_inputs>([&](auto s)
{
using fop_t = std::decay_t<decltype(get<s>(inputs_ref))>;
const int trial_op_dim = static_cast<int>(itod(static_cast<int>(s)));
if (trial_op_dim == 0)
{
// This is inside a lambda so we have to return
// instead of idiomatic 'continue'.
return;
}
auto& B = input_dtqmaps[s].B;
auto& G = input_dtqmaps[s].G;
if constexpr (is_value_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy);
}
}
}
}
}
}
else if constexpr (is_gradient_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
if (m == 0)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * G(qy, 0, Jy);
}
else
{
fhat(i, k, q) += f * G(qx, 0, Jx) * B(qy, 0, Jy);
}
}
}
}
}
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("sum factorized sparse matrix assemble routine "
"not implemented for field operator");
#endif
}
MFEM_SYNC_THREAD;
m_offset += trial_op_dim;
});
auto bvtfhat = Reshape(&A(0, 0, J, j), num_test_dof, test_vdim);
map_quadrature_data_to_fields(bvtfhat, fhat, output, output_dtqmap,
scratch_shmem, dimension, true);
}
}
}
}
/// @brief Assemble element matrix for two or three dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param dimension The spatial dimension.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
/// @param use_sum_factorization Indicator if sum factorization is used.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_naive(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, const real_t>& qpdc,
const DeviceTensor<1, const real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& dimension,
const int& q1d,
const int& td1d,
const bool& use_sum_factorization)
{
if (use_sum_factorization)
{
if (dimension == 2)
{
assemble_element_mat_t2d(A, fhat, qpdc, itod, inputs, output,
input_dtqmaps, output_dtqmap, scratch_shmem, q1d, td1d);
}
else if (dimension == 3)
{
assemble_element_mat_t3d(A, fhat, qpdc, itod, inputs, output,
input_dtqmaps, output_dtqmap, scratch_shmem, q1d, td1d);
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("element matrix assemble not implemented for non tensor "
"product basis");
#endif
}
}
} // namespace mfem::future
+483 -21
View File
@@ -22,6 +22,7 @@
#include "interpolate.hpp"
#include "integrate.hpp"
#include "qfunction_apply.hpp"
#include "assemble.hpp"
namespace mfem::future
{
@@ -30,14 +31,23 @@ namespace mfem::future
using action_t =
std::function<void(std::vector<Vector> &, const std::vector<Vector> &, Vector &)>;
/// @brief Type alias for a function that computes the cache for the action of a derivative
using derivative_setup_t =
std::function<void(std::vector<Vector> &, const Vector &)>;
/// @brief Type alias for a function that computes the action of a derivative
using derivative_action_t =
std::function<void(std::vector<Vector> &, const Vector &, Vector &)>;
/// @brief Type alias for a function that assembles the sparse matrix of a
/// @brief Type alias for a function that assembles the SparseMatrix of a
/// derivative operator
using assemble_derivative_sparsematrix_callback_t =
std::function<void(std::vector<Vector> &, SparseMatrix *&)>;
/// @brief Type alias for a function that assembles the HypreParMatrix of a
/// derivative operator
using assemble_derivative_hypreparmatrix_callback_t =
std::function<void(std::vector<Vector> &, HypreParMatrix &)>;
std::function<void(std::vector<Vector> &, HypreParMatrix *&)>;
/// @brief Type alias for a function that applies the appropriate restriction to
/// the solution and parameters
@@ -81,6 +91,8 @@ public:
const std::vector<Vector *> &parameters_l,
const restriction_callback_t &restriction_callback,
const std::function<void(Vector &, Vector &)> &prolongation_transpose,
const std::vector<assemble_derivative_sparsematrix_callback_t>
&assemble_derivative_sparsematrix_callbacks,
const std::vector<assemble_derivative_hypreparmatrix_callback_t>
&assemble_derivative_hypreparmatrix_callbacks) :
Operator(height, width),
@@ -91,6 +103,8 @@ public:
derivative_actions_transpose(derivative_actions_transpose),
transpose_direction(transpose_direction),
prolongation_transpose(prolongation_transpose),
assemble_derivative_sparsematrix_callbacks(
assemble_derivative_sparsematrix_callbacks),
assemble_derivative_hypreparmatrix_callbacks(
assemble_derivative_hypreparmatrix_callbacks)
{
@@ -156,14 +170,29 @@ public:
prolongation_transpose(daction_l, result_t);
};
/// @brief Assemble the derivative operator into a SparseMatrix.
///
/// @param A The SparseMatrix to assemble the derivative operator into. Can
/// be an uninitialized object.
void Assemble(SparseMatrix *&A)
{
MFEM_ASSERT(!assemble_derivative_sparsematrix_callbacks.empty(),
"derivative can't be assembled into a SparseMatrix");
for (const auto &f : assemble_derivative_sparsematrix_callbacks)
{
f(fields_e, A);
}
}
/// @brief Assemble the derivative operator into a HypreParMatrix.
///
/// @param A The HypreParMatrix to assemble the derivative operator into. Can
/// be an uninitialized object.
void Assemble(HypreParMatrix &A)
void Assemble(HypreParMatrix *&A)
{
MFEM_ASSERT(!assemble_derivative_hypreparmatrix_callbacks.empty(),
"derivative can't be assembled into a matrix");
"derivative can't be assembled into a HypreParMatrix");
for (const auto &f : assemble_derivative_hypreparmatrix_callbacks)
{
@@ -196,6 +225,10 @@ private:
std::function<void(Vector &, Vector &)> prolongation_transpose;
/// Callbacks that assemble derivatives into a SparseMatrix.
std::vector<assemble_derivative_sparsematrix_callback_t>
assemble_derivative_sparsematrix_callbacks;
/// Callbacks that assemble derivatives into a HypreParMatrix.
std::vector<assemble_derivative_hypreparmatrix_callback_t>
assemble_derivative_hypreparmatrix_callbacks;
@@ -398,6 +431,34 @@ public:
const size_t derivative_idx = FindIdx(derivative_id, fields);
std::vector<Vector> s_l(solutions_l.size());
for (size_t i = 0; i < s_l.size(); i++)
{
s_l[i] = *sol_l[i];
}
std::vector<Vector> p_l(parameters_l.size());
for (size_t i = 0; i < p_l.size(); i++)
{
p_l[i] = *par_l[i];
}
fields_e.resize(solutions_l.size() + parameters_l.size());
restriction_callback(s_l, p_l, fields_e);
// Dummy
Vector dir_l;
if (derivative_idx > s_l.size())
{
dir_l = p_l[derivative_idx - s_l.size()];
}
else
{
dir_l = s_l[derivative_idx];
}
derivative_setup_callbacks[derivative_id][0](fields_e, dir_l);
return std::make_shared<DerivativeOperator>(
height,
GetTrueVSize(fields[derivative_idx]),
@@ -411,6 +472,7 @@ public:
par_l,
restriction_callback,
prolongation_transpose,
assemble_derivative_sparsematrix_callbacks[derivative_id],
assemble_derivative_hypreparmatrix_callbacks[derivative_id]);
}
@@ -420,10 +482,14 @@ private:
MultLevel mult_level = TVECTOR;
std::vector<action_t> action_callbacks;
std::map<size_t, std::vector<derivative_setup_t>> derivative_setup_callbacks;
std::map<size_t,
std::vector<derivative_action_t>> derivative_action_callbacks;
std::map<size_t,
std::vector<derivative_action_t>> daction_transpose_callbacks;
std::map<size_t,
std::vector<assemble_derivative_sparsematrix_callback_t>>
assemble_derivative_sparsematrix_callbacks;
std::map<size_t,
std::vector<assemble_derivative_hypreparmatrix_callback_t>>
assemble_derivative_hypreparmatrix_callbacks;
@@ -444,6 +510,8 @@ private:
std::function<void(Vector &, Vector &)> output_restriction_transpose;
restriction_callback_t restriction_callback;
std::map<size_t, Vector> derivative_qp_caches;
std::map<size_t, size_t> assembled_vector_sizes;
bool use_tensor_product_structure = true;
@@ -563,6 +631,13 @@ void DifferentiableOperator::AddIntegrator(
auto output_to_field =
create_descriptors_to_fields_map<entity_t>(fields, outputs);
// TODO: factor out
std::vector<int> inputs_vdim(num_inputs);
for_constexpr<num_inputs>([&](auto i)
{
inputs_vdim[i] = get<i>(inputs).vdim;
});
const Array<int> *elem_attributes = nullptr;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
@@ -829,7 +904,8 @@ void DifferentiableOperator::AddIntegrator(
// print_shared_memory_info(shmem_info);
Vector direction_e;
Vector direction_e(get_restriction<entity_t>(fields[d_field_idx],
element_dof_ordering)->Height());
Vector derivative_action_e(output_e_size);
derivative_action_e = 0.0;
@@ -841,6 +917,152 @@ void DifferentiableOperator::AddIntegrator(
}
const auto input_is_dependent = it->second;
// Trial operator dimension for each input.
// The trial operator dimension is set for each input that is
// dependent and if it is independent the dimension is 0.
Vector inputs_trial_op_dim(num_inputs);
int total_trial_op_dim = 0;
{
auto itod = Reshape(inputs_trial_op_dim.HostReadWrite(), num_inputs);
int idx = 0;
for_constexpr<num_inputs>([&](auto s)
{
if (!input_is_dependent[s])
{
itod(idx) = 0;
}
else
{
// TODO: BUG! Make this a general function that works for all kinds of inputs.
itod(idx) = input_size_on_qp[s] / get<s>(inputs).vdim;
}
total_trial_op_dim += static_cast<int>(itod(idx));
idx++;
});
}
// First Input index of the derivative
const size_t d_input_idx = [d_field_idx, &input_to_field]
{
for (size_t i = 0; i < input_to_field.size(); i++)
{
if (input_to_field[i] == d_field_idx)
{
return i;
}
}
return size_t(SIZE_MAX);
}();
const int trial_vdim = GetVDim(fields[d_field_idx]);
const int num_trial_dof =
get_restriction<entity_t>(fields[d_field_idx], element_dof_ordering)->Height() /
inputs_vdim[d_input_idx] / num_entities;
const int num_trial_dof_1d =
input_dtq_maps[d_input_idx].B.GetShape()[DofToQuadMap::Index::DOF];
Vector Ae_mem(num_test_dof * test_vdim * num_trial_dof * trial_vdim *
num_entities);
Ae_mem = 0.0;
// Quadrature point local derivative cache for each element, with data
// layout:
// [test_vdim, test_op_dim, trial_vdim, trial_op_dim, qp, num_entities].
derivative_qp_caches[derivative_id] = Vector(test_vdim * test_op_dim *
trial_vdim *
total_trial_op_dim * num_qp * num_entities);
// Create local references for MSVC lambda capture compatibility
auto& fields_ref = this->fields;
auto& derivative_qp_caches_ref = this->derivative_qp_caches[derivative_id];
// In each of the callbacks we're saving the derivatives in the quadrature point
// caches. This trades memory with computational effort but also minimizes
// data movement on each multiplication of the gradient with a directional
// vector.
derivative_setup_callbacks[derivative_id].push_back(
[
// capture by copy:
dimension, // int
num_entities, // int
num_qp, // int
q1d, // int
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
input_to_field, // std::array<int, s>
qfunc, // qfunc_t
thread_blocks, // ThreadBlocks
shmem_cache, // Vector (local)
shmem_info, // SharedMemoryInfo
// TODO: make this Array<int> a member of the DifferentiableOperator
// and capture it by ref.
elem_attributes, // Array<int>
element_dof_ordering, // ElementDofOrdering
direction, // FieldDescriptor
direction_e, // Vector
da_size_on_qp, // int
total_trial_op_dim,
trial_vdim,
inputs_trial_op_dim,
// capture by ref:
&qpdc_mem = derivative_qp_caches_ref
](std::vector<Vector> &f_e, const Vector &dir_l) mutable
{
restriction<entity_t>(direction, dir_l, direction_e,
element_dof_ordering);
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
num_entities);
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(),
shmem_info.direction_size,
num_entities);
auto qpdc = Reshape(qpdc_mem.ReadWrite(), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp, num_entities);
auto itod = Reshape(inputs_trial_op_dim.Read(), num_inputs);
const auto d_elem_attr = elem_attributes->Read();
const bool has_attr = attributes.Size() > 0;
const auto d_domain_attr = attributes.Read();
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
{
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
direction_shmem, input_shmem,
shadow_shmem_, residual_shmem,
scratch_shmem] =
unpack_shmem(shmem, shmem_info, input_dtq_maps, output_dtq_maps,
wrapped_fields_e, wrapped_direction_e, num_qp, e);
auto &shadow_shmem = shadow_shmem_;
map_fields_to_quadrature_data(
input_shmem, fields_shmem, input_dtq_shmem, input_to_field,
inputs, ir_weights, scratch_shmem, dimension,
use_sum_factorization);
set_zero(shadow_shmem);
auto qpdc_e = Reshape(&qpdc(0, 0, 0, 0, 0, e), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp);
call_qfunction_derivative<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem, qpdc_e, itod, da_size_on_qp,
q1d, dimension, use_sum_factorization);
}, num_entities, thread_blocks, shmem_info.total_size,
shmem_cache.ReadWrite());
});
// The derivative action only uses the quadrature point caches and applies
// them to an input vector before integrating with the desired trial operator.
derivative_action_callbacks[derivative_id].push_back(
[
// capture by copy:
@@ -857,9 +1079,7 @@ void DifferentiableOperator::AddIntegrator(
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
input_to_field, // std::array<int, s>
output_fop, // class derived from FieldOperator
qfunc, // qfunc_t
thread_blocks, // ThreadBlocks
shmem_cache, // Vector (local)
shmem_info, // SharedMemoryInfo
@@ -872,9 +1092,11 @@ void DifferentiableOperator::AddIntegrator(
direction_e, // Vector
derivative_action_e, // Vector
element_dof_ordering, // ElementDofOrdering
da_size_on_qp, // int
inputs_trial_op_dim,
total_trial_op_dim,
trial_vdim,
// capture by ref:
&qpdc_mem = derivative_qp_caches_ref,
&or_transpose
](
std::vector<Vector> &f_e, const Vector &dir_l,
@@ -890,6 +1112,11 @@ void DifferentiableOperator::AddIntegrator(
shmem_info.direction_size,
num_entities);
auto qpdc = Reshape(qpdc_mem.Read(), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp, num_entities);
auto itod = Reshape(inputs_trial_op_dim.Read(), num_inputs);
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
@@ -907,25 +1134,20 @@ void DifferentiableOperator::AddIntegrator(
wrapped_fields_e, wrapped_direction_e, num_qp, e);
auto &shadow_shmem = shadow_shmem_;
map_fields_to_quadrature_data(
input_shmem, fields_shmem, input_dtq_shmem, input_to_field,
inputs, ir_weights, scratch_shmem, dimension,
use_sum_factorization);
// TODO: Probably redundant
set_zero(shadow_shmem);
map_direction_to_quadrature_data_conditional(
shadow_shmem, direction_shmem, input_dtq_shmem, inputs,
ir_weights, scratch_shmem, input_is_dependent, dimension,
use_sum_factorization);
call_qfunction_derivative_action<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem,
da_size_on_qp, num_qp, q1d, dimension, use_sum_factorization);
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim,
test_op_dim, num_qp);
auto qpdce = Reshape(&qpdc(0, 0, 0, 0, 0, e), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp);
apply_qpdc(fhat, shadow_shmem, qpdce, itod, q1d, dimension,
use_sum_factorization);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields(
y, fhat, output_fop, output_dtq_shmem[0],
@@ -934,6 +1156,246 @@ void DifferentiableOperator::AddIntegrator(
shmem_cache.ReadWrite());
or_transpose(derivative_action_e, der_action_l);
});
assemble_derivative_sparsematrix_callbacks[derivative_id].push_back(
[
// capture by copy:
dimension, // int
num_entities, // int
num_test_dof, // int
num_qp, // int
q1d, // int
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
attributes, // Array<int>
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
input_to_field, // std::array<int, s>
output_fop, // class derived from FieldOperator
thread_blocks, // ThreadBlocks
shmem_cache, // Vector (local)
shmem_info, // SharedMemoryInfo
// TODO: make this Array<int> a member of the DifferentiableOperator
// and capture it by ref.
elem_attributes, // Array<int>
input_is_dependent, // std::array<bool, num_inputs>
direction_e, // Vector
total_trial_op_dim,
trial_vdim,
num_trial_dof,
num_trial_dof_1d,
inputs_trial_op_dim,
Ae_mem,
output_to_field,
// capture by ref:
&qpdc_mem = derivative_qp_caches_ref,
&fields = fields_ref
](std::vector<Vector> &f_e, SparseMatrix *&A) mutable
{
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
num_entities);
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(),
shmem_info.direction_size,
num_entities);
auto qpdc = Reshape(qpdc_mem.Read(), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp, num_entities);
auto itod = Reshape(inputs_trial_op_dim.Read(), num_inputs);
auto Ae = Reshape(Ae_mem.ReadWrite(), num_test_dof, test_vdim, num_trial_dof,
trial_vdim, num_entities);
const auto d_elem_attr = elem_attributes->Read();
const bool has_attr = attributes.Size() > 0;
const auto d_domain_attr = attributes.Read();
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
{
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
direction_shmem, input_shmem,
shadow_shmem_, residual_shmem,
scratch_shmem] =
unpack_shmem(shmem, shmem_info, input_dtq_maps, output_dtq_maps,
wrapped_fields_e, wrapped_direction_e, num_qp, e);
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto Aee = Reshape(&Ae(0, 0, 0, 0, e), num_test_dof, test_vdim, num_trial_dof,
trial_vdim);
auto qpdce = Reshape(&qpdc(0, 0, 0, 0, 0, e), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp);
assemble_element_mat_naive(Aee, fhat, qpdce, itod, inputs, output_fop,
input_dtq_shmem, output_dtq_shmem[0], scratch_shmem, dimension, q1d,
num_trial_dof_1d, use_sum_factorization);
}, num_entities, thread_blocks, shmem_info.total_size,
shmem_cache.ReadWrite());
FieldDescriptor *trial_field = nullptr;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
A = new SparseMatrix(test_fes->GetVSize(), trial_fes->GetVSize());
auto tmp = Reshape(Ae_mem.HostReadWrite(), num_test_dof * test_vdim,
num_trial_dof * trial_vdim, num_entities);
for (int e = 0; e < num_entities; e++)
{
DenseMatrix Aee(&tmp(0, 0, e), num_test_dof * test_vdim,
num_trial_dof * trial_vdim);
Array<int> test_vdofs, trial_vdofs;
test_fes->GetElementVDofs(e, test_vdofs);
trial_fes->GetElementVDofs(e, trial_vdofs);
if (use_sum_factorization)
{
Array<int> test_vdofs_mapped(test_vdofs.Size());
const Array<int> &test_dofmap =
dynamic_cast<const TensorBasisElement&>(*test_fes->GetFE(0)).GetDofMap();
if (test_dofmap.Size() == 0)
{
test_vdofs_mapped = test_vdofs;
}
else
{
MFEM_ASSERT(test_dofmap.Size() == num_test_dof,
"internal error: dof map of the test space does not "
"match previously determined number of test space dofs");
for (int vd = 0; vd < test_vdim; vd++)
{
for (int i = 0; i < num_test_dof; i++)
{
test_vdofs_mapped[i + vd * num_test_dof] =
test_vdofs[test_dofmap[i] + vd * num_test_dof];
}
}
}
Array<int> trial_vdofs_mapped(trial_vdofs.Size());
const Array<int> &trial_dofmap =
dynamic_cast<const TensorBasisElement&>(*trial_fes->GetFE(0)).GetDofMap();
if (trial_dofmap.Size() == 0)
{
trial_vdofs_mapped = trial_vdofs;
}
else
{
MFEM_ASSERT(trial_dofmap.Size() == num_trial_dof,
"internal error: dof map of the test space does not "
"match previously determined number of test space dofs");
for (int vd = 0; vd < trial_vdim; vd++)
{
for (int i = 0; i < num_trial_dof; i++)
{
trial_vdofs_mapped[i + vd * num_trial_dof] =
trial_vdofs[trial_dofmap[i] + vd * num_trial_dof];
}
}
}
A->AddSubMatrix(test_vdofs_mapped, trial_vdofs_mapped, Aee, 1);
}
else
{
A->AddSubMatrix(test_vdofs, trial_vdofs, Aee, 1);
}
}
A->Finalize();
});
// Create local references for MSVC lambda capture compatibility
auto& assemble_derivative_sparsematrix_callbacks_ref =
this->assemble_derivative_sparsematrix_callbacks[derivative_id];
assemble_derivative_hypreparmatrix_callbacks[derivative_id].push_back(
[
input_is_dependent,
input_to_field,
output_to_field,
&spmatcb = assemble_derivative_sparsematrix_callbacks_ref,
&fields = fields_ref
](std::vector<Vector> &f_e, HypreParMatrix *&A) mutable
{
SparseMatrix *spmat = nullptr;
for (const auto &f : spmatcb)
{
f(f_e, spmat);
}
if (spmat == nullptr)
{
MFEM_ABORT("internal error");
}
bool same_test_and_trial = false;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
if (output_to_field[0] == input_to_field[s])
{
same_test_and_trial = true;
break;
}
}
}
FieldDescriptor *trial_field = nullptr;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
if (same_test_and_trial)
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
spmat);
A = RAP(&tmp, test_fes->Dof_TrueDof_Matrix());
}
else
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
trial_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
trial_fes->GetDofOffsets(),
spmat);
A = RAP(test_fes->Dof_TrueDof_Matrix(), &tmp,
trial_fes->Dof_TrueDof_Matrix());
}
delete spmat;
});
}, derivative_ids);
}
}
+4 -3
View File
@@ -511,7 +511,7 @@ void map_fields_to_quadrature_data(
std::array<DeviceTensor<2>, num_inputs> &fields_qp,
const std::array<DeviceTensor<1>, num_fields> &fields_e,
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
const std::array<int, num_inputs> &input_to_field,
const std::array<size_t, num_inputs> &input_to_field,
const field_operator_ts &fops,
const DeviceTensor<1, const real_t> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem,
@@ -526,7 +526,8 @@ void map_fields_to_quadrature_data(
for_constexpr<num_inputs>([&](auto i)
{
const DeviceTensor<1> &field_e =
(input_to_field[i] == -1) ? dummy_field_weight : fields_e[input_to_field[i]];
(input_to_field[i] == SIZE_MAX) ? dummy_field_weight :
fields_e[input_to_field[i]];
if (use_sum_factorization)
{
@@ -637,7 +638,7 @@ void map_direction_to_quadrature_data_conditional(
const std::array<DeviceTensor<1>, 6> &scratch_mem,
const std::array<bool, num_inputs> &conditions,
const int &dimension,
const bool &use_sum_factorization = false)
const bool &use_sum_factorization)
{
for_constexpr<num_inputs>([&](auto i)
{
+308 -14
View File
@@ -46,7 +46,7 @@ void call_qfunction(
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(q, x, q1d)
{
auto qf_args = decay_tuple<qf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), rs_qp);
@@ -55,9 +55,9 @@ void call_qfunction(
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
const int q = qx + q1d * qy;
auto qf_args = decay_tuple<qf_param_ts> {};
@@ -68,11 +68,11 @@ void call_qfunction(
}
else if (dimension == 3)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
MFEM_FOREACH_THREAD_DIRECT(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto qf_args = decay_tuple<qf_param_ts> {};
@@ -92,7 +92,7 @@ void call_qfunction(
}
else
{
MFEM_FOREACH_THREAD(q, x, num_qp)
MFEM_FOREACH_THREAD_DIRECT(q, x, num_qp)
{
auto qf_args = decay_tuple<qf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), rs_qp);
@@ -134,7 +134,7 @@ void call_qfunction_derivative_action(
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(q, x, q1d)
{
auto r = Reshape(&residual_shmem(0, q), das_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
@@ -149,9 +149,9 @@ void call_qfunction_derivative_action(
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
const int q = qx + q1d * qy;
auto r = Reshape(&residual_shmem(0, q), das_qp);
@@ -168,11 +168,11 @@ void call_qfunction_derivative_action(
}
else if (dimension == 3)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
MFEM_FOREACH_THREAD_DIRECT(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto r = Reshape(&residual_shmem(0, q), das_qp);
@@ -195,7 +195,7 @@ void call_qfunction_derivative_action(
}
else
{
MFEM_FOREACH_THREAD(q, x, num_qp)
MFEM_FOREACH_THREAD_DIRECT(q, x, num_qp)
{
auto r = Reshape(&residual_shmem(0, q), das_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
@@ -211,6 +211,300 @@ void call_qfunction_derivative_action(
MFEM_SYNC_THREAD;
}
namespace detail
{
template <
typename qf_param_ts,
typename qfunc_t,
std::size_t num_fields>
MFEM_HOST_DEVICE inline
void call_qfunction_derivative(
qfunc_t &qfunc,
const std::array<DeviceTensor<2>, num_fields> &input_shmem,
const std::array<DeviceTensor<2>, num_fields> &shadow_shmem,
DeviceTensor<2> &residual_shmem,
DeviceTensor<5> &qpdc,
const DeviceTensor<1, const real_t> &itod,
const int &das_qp,
const int &q)
{
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
const int num_qp = qpdc.GetShape()[4];
const size_t num_inputs = itod.GetShape()[0];
for (int j = 0; j < trial_vdim; j++)
{
int m_offset = 0;
for (size_t s = 0; s < num_inputs; s++)
{
const int trial_op_dim = static_cast<int>(itod(s));
if (trial_op_dim == 0)
{
continue;
}
auto d_qp = Reshape(&(shadow_shmem[s])[0], trial_vdim, trial_op_dim, num_qp);
for (int m = 0; m < trial_op_dim; m++)
{
d_qp(j, m, q) = 1.0;
auto r = Reshape(&residual_shmem(0, q), das_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
#ifdef MFEM_USE_ENZYME
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
shadow_shmem, q);
#else
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
d_qp(j, m, q) = 0.0;
auto f = Reshape(&r(0), test_vdim, test_op_dim);
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
qpdc(i, k, j, m + m_offset, q) = f(i, k);
}
}
}
m_offset += trial_op_dim;
}
}
}
}
/// @brief Call a qfunction with the given parameters and
/// compute it's derivative represented by the Jacobian on
/// each quadrature point.
///
/// @param qfunc the qfunction to call.
/// @param input_shmem the input shared memory.
/// @param shadow_shmem the shadow shared memory.
/// @param residual_shmem the residual shared memory.
/// @param qpdc the quadrature point data cache holding the resulting
/// Jacobians on each quadrature point.
/// @param itod inputs trial operator dimension.
/// If input is dependent the value corresponds to the spatial dimension, otherwise
/// a zero indicates non-dependence on the variable.
/// @param das_qp the size of the derivative action.
/// @param q1d the number of quadrature points in 1D.
/// @param dimension the spatial dimension.
/// @param use_sum_factorization whether to use sum factorization.
/// @tparam qf_param_ts the tuple type of the qfunction parameters.
template <
typename qf_param_ts,
typename qfunc_t,
std::size_t num_fields>
MFEM_HOST_DEVICE inline
void call_qfunction_derivative(
qfunc_t &qfunc,
const std::array<DeviceTensor<2>, num_fields> &input_shmem,
const std::array<DeviceTensor<2>, num_fields> &shadow_shmem,
DeviceTensor<2> &residual_shmem,
DeviceTensor<5> &qpdc,
const DeviceTensor<1, const real_t> &itod,
const int &das_qp,
const int &q1d,
const int &dimension,
const bool &use_sum_factorization)
{
if (use_sum_factorization)
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD_DIRECT(q, x, q1d)
{
detail::call_qfunction_derivative<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem, qpdc, itod, das_qp, q);
}
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
const int q = qx + q1d * qy;
detail::call_qfunction_derivative<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem, qpdc, itod, das_qp, q);
}
}
}
else if (dimension == 3)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
detail::call_qfunction_derivative<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem, qpdc, itod, das_qp, q);
}
}
}
}
else
{
MFEM_ABORT_KERNEL("unsupported dimension");
}
}
else
{
const int num_qp = qpdc.GetShape()[4];
MFEM_FOREACH_THREAD_DIRECT(q, x, num_qp)
{
detail::call_qfunction_derivative<qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem, qpdc, itod, das_qp, q);
}
}
MFEM_SYNC_THREAD;
}
namespace detail
{
/// @brief Apply the quadrature point data cache (qpdc) to a vector
/// (usually a direction) on quadrature point q.
///
/// The qpdc consists of compatible data to be used for integration with a test
/// operator, e.g. Jacobians of a linearization from a FE operation with a trial
/// function including integration weights and necessesary transformations.
///
/// @param fhat the qpdc applied to a vector in shadow_memory.
/// @param shadow_shmem the shadow shared memory.
/// @param qpdc the quadrature point data cache holding the resulting
/// Jacobians on each quadrature point.
/// @param itod inputs trial operator dimension.
/// If input is dependent the value corresponds to the spatial dimension, otherwise
/// a zero indicates non-dependence on the variable.
/// @param q the current quadrature point index.
template <size_t num_fields>
MFEM_HOST_DEVICE inline
void apply_qpdc(
DeviceTensor<3> &fhat,
const std::array<DeviceTensor<2>, num_fields> &shadow_shmem,
const DeviceTensor<5, const real_t> &qpdc,
const DeviceTensor<1, const real_t> &itod,
const int &q)
{
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
const int num_qp = qpdc.GetShape()[4];
const size_t num_inputs = itod.GetShape()[0];
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
real_t sum = 0.0;
int m_offset = 0;
for (size_t s = 0; s < num_inputs; s++)
{
const int trial_op_dim = static_cast<int>(itod(s));
if (trial_op_dim == 0)
{
continue;
}
const auto d_qp =
Reshape(&(shadow_shmem[s])[0], trial_vdim, trial_op_dim, num_qp);
for (int j = 0; j < trial_vdim; j++)
{
for (int m = 0; m < trial_op_dim; m++)
{
sum += qpdc(i, k, j, m + m_offset, q) * d_qp(j, m, q);
}
}
m_offset += trial_op_dim;
}
fhat(i, k, q) = sum;
}
}
}
}
/// @brief Apply the quadrature point data cache (qpdc) to a vector
/// (usually a direction).
///
/// The qpdc consists of compatible data to be used for integration with a test
/// operator, e.g. Jacobians of a linearization from a FE operation with a trial
/// function including integration weights and necessesary transformations.
///
/// @param fhat the qpdc applied to a vector in shadow_memory.
/// @param shadow_shmem the shadow shared memory.
/// @param qpdc the quadrature point data cache holding the resulting
/// Jacobians on each quadrature point.
/// @param itod inputs trial operator dimension.
/// If input is dependent the value corresponds to the spatial dimension, otherwise
/// a zero indicates non-dependence on the variable.
/// @param q1d number of quadrature points in 1D.
/// @param dimension spatial dimension.
/// @param use_sum_factorization whether to use sum factorization.
template <size_t num_fields>
MFEM_HOST_DEVICE inline
void apply_qpdc(
DeviceTensor<3> &fhat,
const std::array<DeviceTensor<2>, num_fields> &shadow_shmem,
const DeviceTensor<5, const real_t> &qpdc,
const DeviceTensor<1, const real_t> &itod,
const int &q1d,
const int &dimension,
const bool &use_sum_factorization)
{
if (use_sum_factorization)
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD_DIRECT(q, x, q1d)
{
detail::apply_qpdc(fhat, shadow_shmem, qpdc, itod, q);
}
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
const int q = qx + q1d * qy;
detail::apply_qpdc(fhat, shadow_shmem, qpdc, itod, q);
}
}
}
else if (dimension == 3)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
detail::apply_qpdc(fhat, shadow_shmem, qpdc, itod, q);
}
}
}
}
else
{
MFEM_ABORT_KERNEL("unsupported dimension");
}
}
else
{
const int num_qp = qpdc.GetShape()[4];
MFEM_FOREACH_THREAD_DIRECT(q, x, num_qp)
{
detail::apply_qpdc(fhat, shadow_shmem, qpdc, itod, q);
}
}
}
template <typename qfunc_t, typename args_ts, size_t num_args>
MFEM_HOST_DEVICE inline
void apply_kernel(
+62 -33
View File
@@ -20,6 +20,7 @@
#include <vector>
#include <type_traits>
#include <numeric>
#include <iomanip>
#include "../../general/communication.hpp"
#include "../../general/forall.hpp"
@@ -107,25 +108,32 @@ constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg)
indices{});
}
template <typename... input_ts, std::size_t... Is>
auto make_dependency_map_impl(
tuple<input_ts...> inputs,
std::index_sequence<Is...>)
template <std::size_t I, typename Tuple, std::size_t... Is>
std::array<bool, sizeof...(Is)>
make_dependency_array(const Tuple& inputs, std::index_sequence<Is...>)
{
auto make_dependency_array = [&](auto i)
{
return std::array<bool, sizeof...(input_ts)>
{
(get<i>(inputs).GetFieldId() == get<Is>(inputs).GetFieldId())...
};
};
return { (get<I>(inputs).GetFieldId() == get<Is>(inputs).GetFieldId())... };
}
std::unordered_map<int, std::array<bool, sizeof...(input_ts)>> map;
for_constexpr<sizeof...(input_ts)>([&](auto i)
template <typename... input_ts, std::size_t... Is>
auto make_dependency_map_impl(tuple<input_ts...> inputs,
std::index_sequence<Is...>)
{
constexpr std::size_t N = sizeof...(input_ts);
if constexpr (N == 0)
return std::unordered_map<int, std::array<bool, 0>> {};
std::unordered_map<int, std::array<bool, N>> map;
(void)std::initializer_list<int>
{
map[get<i>(inputs).GetFieldId()] =
make_dependency_array(std::integral_constant<std::size_t, i> {});
});
(
map[get<Is>(inputs).GetFieldId()] =
make_dependency_array<Is>(inputs, std::make_index_sequence<N>{}),
0
)...
};
return map;
}
@@ -200,24 +208,45 @@ void print_tuple(const std::tuple<Args...>& t)
/// ..., vmn]]
/// which is compatible with numpy syntax.
///
/// @param m mfem::DenseMatrix to print
/// @param out ostream to print to
/// @param A mfem::DenseMatrix to print
inline
void pretty_print(const mfem::DenseMatrix& m)
void pretty_print(std::ostream &out, const mfem::DenseMatrix &A)
{
out << "[";
for (int i = 0; i < m.NumRows(); i++)
// Determine the max width of any entry in scientific notation
int max_width = 0;
for (int i = 0; i < A.NumRows(); ++i)
{
for (int j = 0; j < m.NumCols(); j++)
for (int j = 0; j < A.NumCols(); ++j)
{
out << m(i, j);
if (j < m.NumCols() - 1)
std::ostringstream oss;
oss << std::scientific << std::setprecision(2) << A(i, j);
max_width = std::max(max_width, static_cast<int>(oss.str().length()));
}
}
out << "[\n";
for (int i = 0; i < A.NumRows(); ++i)
{
out << " [";
for (int j = 0; j < A.NumCols(); ++j)
{
out << std::setw(max_width) << std::scientific << std::setprecision(2) <<
A(i, j);
if (j < A.NumCols() - 1)
{
out << ", ";
}
}
if (i < m.NumRows() - 1)
out << "]";
if (i < A.NumRows() - 1)
{
out << ", ";
out << ",\n";
}
else
{
out << "\n";
}
}
out << "]\n";
@@ -356,7 +385,7 @@ void print_mpi_sync(const std::string& msg)
else
{
// Other ranks: Send message to rank 0
MPI_Send(msg.c_str(), static_cast<int>(msg_len), MPI_CHAR,
MPI_Send(const_cast<char*>(msg.c_str()), static_cast<int>(msg_len), MPI_CHAR,
0, 0, MPI_COMM_WORLD);
}
@@ -1404,12 +1433,12 @@ int GetSizeOnQP(const field_operator_t &, const FieldDescriptor &f)
/// @tparam entity_t the entity type (see Entity).
/// @returns an array mapping field operator types to field descriptor indices.
template <typename entity_t, typename field_operator_ts>
std::array<int, tuple_size<field_operator_ts>::value>
std::array<size_t, tuple_size<field_operator_ts>::value>
create_descriptors_to_fields_map(
const std::vector<FieldDescriptor> &fields,
field_operator_ts &fops)
{
std::array<int, tuple_size<field_operator_ts>::value> map;
std::array<size_t, tuple_size<field_operator_ts>::value> map;
auto find_id = [](const std::vector<FieldDescriptor> &fields, std::size_t i)
{
@@ -1421,9 +1450,9 @@ create_descriptors_to_fields_map(
if (it == fields.end())
{
return -1;
return SIZE_MAX;
}
return static_cast<int>(it - fields.begin());
return static_cast<size_t>(it - fields.begin());
};
auto f = [&](auto &fop, auto &map)
@@ -1434,7 +1463,7 @@ create_descriptors_to_fields_map(
fop.dim = GetDimension<entity_t>(fields[0]);
fop.vdim = 1;
fop.size_on_qp = 1;
map = -1;
map = SIZE_MAX;
}
else
{
@@ -2220,7 +2249,7 @@ template <
std::array<DofToQuadMap, N> create_dtq_maps_impl(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqs,
const std::array<int, N> &field_map,
const std::array<size_t, N> &field_map,
std::index_sequence<Is...>)
{
auto f = [&](auto fop, std::size_t idx)
@@ -2305,7 +2334,7 @@ template <
std::array<DofToQuadMap, num_fields> create_dtq_maps(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqmaps,
const std::array<int, num_fields> &to_field_map)
const std::array<size_t, num_fields> &to_field_map)
{
return create_dtq_maps_impl<entity_t>(
fops, dtqmaps,
+1 -1
View File
@@ -81,7 +81,7 @@ void NURBS1DFiniteElement::CalcHessian (const IntegrationPoint &ip,
sum = 1.0/sum;
add(sum, hess, -2*dsum*sum*sum, grad, hess);
add(1.0, hess, (-d2sum + 2*dsum*dsum*sum)*sum*sum, shape_x, hess);
add((real_t)1.0, hess, (-d2sum + 2*dsum*dsum*sum)*sum*sum, shape_x, hess);
}
+1 -1
View File
@@ -1574,7 +1574,7 @@ void FuentesPyramid::V_R(int p, Vector s, const DenseMatrix &grad_s,
{
// dphi_E_i.GetRow(i, dphi);
for (int l=0; l<3; l++) { dphi[l] = dphi_E_i(i, l); }
add(t * t, dphi, 2.0 * t * phi_E_i(i), dt3, dphit2);
add(t * t, dphi, 2 * t * phi_E_i(i), dt3, dphit2);
dphit2.cross3D(dmu3, dphixdmu);
// u.SetRow(i, dphixdmu);
for (int l=0; l<3; l++) { u(i, l) = dphixdmu(l); }
+1 -1
View File
@@ -509,7 +509,7 @@ GetFace(int &nv, v_t &v, int &ne, e_t &e, eo_t &eo,
int v0 = v[f_consts::Edges[i][0]];
int v1 = v[f_consts::Edges[i][1]];
int eor = 0;
if (v0 > v1) { swap(v0, v1); eor = 1; }
if (v0 > v1) { std::swap(v0, v1); eor = 1; }
for (int j = g_consts::VertToVert::I[v0]; true; j++)
{
MFEM_ASSERT(j < g_consts::VertToVert::I[v0+1],
+1
View File
@@ -50,6 +50,7 @@
#include "dgmassinv.hpp"
#include "hyperbolic.hpp"
#include "bounds.hpp"
#include "particleset.hpp"
#include "dfem/doperator.hpp"
+15 -31
View File
@@ -27,37 +27,6 @@ using namespace std;
namespace mfem
{
template <>
void Ordering::DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = 1; vd < vdim; vd++)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byNODES>(ndofs, vdim, dofs[i], vd);
}
}
}
template <>
void Ordering::DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = vdim-1; vd >= 0; vd--)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byVDIM>(ndofs, vdim, dofs[i], vd);
}
}
}
FiniteElementSpace::FiniteElementSpace()
: mesh(NULL), fec(NULL), vdim(0), ordering(Ordering::byNODES),
@@ -1583,6 +1552,11 @@ const FaceRestriction *FiniteElementSpace::GetFaceRestriction(
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const IntegrationRule &ir) const
{
if (!QuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
for (int i = 0; i < E2Q_array.Size(); i++)
{
const QuadratureInterpolator *qi = E2Q_array[i];
@@ -1597,6 +1571,11 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const QuadratureSpace &qs) const
{
if (!QuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
for (int i = 0; i < E2Q_array.Size(); i++)
{
const QuadratureInterpolator *qi = E2Q_array[i];
@@ -1612,6 +1591,11 @@ const FaceQuadratureInterpolator
*FiniteElementSpace::GetFaceQuadratureInterpolator(
const IntegrationRule &ir, FaceType type) const
{
if (!FaceQuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
if (type==FaceType::Interior)
{
for (int i = 0; i < E2IFQ_array.Size(); i++)
+13 -40
View File
@@ -13,6 +13,7 @@
#define MFEM_FESPACE
#include "../config/config.hpp"
#include "../linalg/ordering.hpp"
#include "../linalg/sparsemat.hpp"
#include "../mesh/mesh.hpp"
#include "fe_coll.hpp"
@@ -24,29 +25,6 @@
namespace mfem
{
/** @brief The ordering method used when the number of unknowns per mesh node
(vector dimension) is bigger than 1. */
class Ordering
{
public:
/// %Ordering methods:
enum Type
{
byNODES, /**< loop first over the nodes (inner loop) then over the vector
dimension (outer loop); symbolically it can be represented
as: XXX...,YYY...,ZZZ... */
byVDIM /**< loop first over the vector dimension (inner loop) then over
the nodes (outer loop); symbolically it can be represented
as: XYZ,XYZ,XYZ,... */
};
template <Type Ord>
static inline int Map(int ndofs, int vdim, int dof, int vd);
template <Type Ord>
static void DofsToVDofs(int ndofs, int vdim, Array<int> &dofs);
};
/// @brief Type describing possible layouts for Q-vectors.
/// @sa QuadratureInterpolator and FaceQuadratureInterpolator.
enum class QVectorLayout
@@ -64,20 +42,6 @@ enum class QVectorLayout
byVDIM
};
template <> inline int
Ordering::Map<Ordering::byNODES>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? dof+ndofs*vd : dof-ndofs*vd;
}
template <> inline int
Ordering::Map<Ordering::byVDIM>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? vd+vdim*dof : -1-(vd+vdim*(-1-dof));
}
/// Constants describing the possible orderings of the DOFs in one element.
enum class ElementDofOrdering
{
@@ -799,7 +763,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by QuadratureInterpolator, nullptr is
returned. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const IntegrationRule &ir) const;
@@ -815,7 +782,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by QuadratureInterpolator, nullptr is
returned. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const QuadratureSpace &qs) const;
@@ -825,7 +795,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by FaceQuadratureInterpolator,
nullptr is returned. */
const FaceQuadratureInterpolator *GetFaceQuadratureInterpolator(
const IntegrationRule &ir, FaceType type) const;
+7 -7
View File
@@ -4610,7 +4610,7 @@ GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
void GridFunction::GetElementBoundsAtControlPoints(const int elem,
const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
const FiniteElement *fe = fes->GetFE(elem);
int fes_dim = fes->GetVDim();
@@ -4626,7 +4626,7 @@ void GridFunction::GetElementBoundsAtControlPoints(const int elem,
fes->GetElementDofs(elem, dof_idx);
int ndofs = dof_idx.Size();
int n_c_pts = std::pow(plb.GetNControlPoints(), rdim);
int n_c_pts = static_cast<int>(std::pow(plb.GetNControlPoints(), rdim));
lower.SetSize(n_c_pts*(vdim > 0 ? 1 : fes_dim));
upper.SetSize(n_c_pts*(vdim > 0 ? 1 : fes_dim));
@@ -4658,13 +4658,13 @@ void GridFunction::GetElementBoundsAtControlPoints(const int elem,
void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
Vector lowerC, upperC;
GetElementBoundsAtControlPoints(elem, plb, lowerC, upperC, vdim);
const FiniteElement *fe = fes->GetFE(elem);
int rdim = fe->GetDim();
int n_c_pts = std::pow(plb.GetNControlPoints(), rdim);
int n_c_pts = static_cast<int>(std::pow(plb.GetNControlPoints(), rdim));
int fes_dim = fes->GetVDim();
lower.SetSize((vdim > 0 ? 1 :fes_dim));
upper.SetSize((vdim > 0 ? 1 :fes_dim));
@@ -4681,7 +4681,7 @@ void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
void GridFunction::GetElementBounds(const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
int nel = fes->GetNE();
int fes_dim = fes->GetVDim();
@@ -4704,7 +4704,7 @@ void GridFunction::GetElementBounds(const PLBound &plb,
PLBound GridFunction::GetElementBounds(Vector &lower,
Vector &upper,
const int ref_factor,
const int vdim)
const int vdim) const
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
@@ -4713,7 +4713,7 @@ PLBound GridFunction::GetElementBounds(Vector &lower,
}
PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
const int ref_factor, const int vdim) const
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
+5 -5
View File
@@ -1604,7 +1604,7 @@ public:
/// We compute the bounds for each vdim if @a vdim < 1.
/// Note: For most cases, this method/interface will be sufficient.
virtual PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
const int ref_factor=1, const int vdim=-1) const;
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each element
@@ -1614,27 +1614,27 @@ public:
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
PLBound GetElementBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
const int ref_factor=1, const int vdim=-1) const;
/// Compute piecewise linear bounds on the given element at the grid of
/// [plb.ncp x plb.ncp x plb.ncp] control points for each of the vdim
/// components of the gridfunction.
void GetElementBoundsAtControlPoints(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
const int vdim = -1) const;
/// Compute bounds on the grid function for the given element.
/// The bounds are stored in @b lower and @b upper.
void GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
const int vdim = -1) const;
/// Compute bounds on the grid function for all the elements. The bounds
/// are returned in @b lower and @b upper, ordered byVDim:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
const int vdim=-1);
const int vdim=-1) const;
///@}
/// Destroys grid function.
+1
View File
@@ -324,6 +324,7 @@ void FindPointsGSLIB::FindPoints(const Vector &point_pos,
gsl_elem[i] = 0;
for (int d = 0; d < dim; d++) { gsl_ref(i*dim + d) = -1.; }
gsl_code[i] = 2;
gsl_proc[i] = gsl_comm->id;
}
}
+95 -64
View File
@@ -25,9 +25,9 @@ static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
const GeometricFactors &el_geom,
const FaceGeometricFactors &face_geom,
const FaceNeighborGeometricFactors *nbr_geom,
const Vector &q, const real_t sigma,
const real_t kappa, Vector &pa_data,
const Array<int> &face_info_)
const Vector &q, const int coeff_dim,
const real_t sigma, const real_t kappa,
Vector &pa_data, const Array<int> &face_info_)
{
const auto J_loc = Reshape(el_geom.J.Read(), Q1D, Q1D, 2, 2, NE);
const auto detJe_loc = Reshape(el_geom.detJ.Read(), Q1D, Q1D, NE);
@@ -41,9 +41,9 @@ static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
const auto detJf = Reshape(face_geom.detJ.Read(), Q1D, NF);
const auto n = Reshape(face_geom.normal.Read(), Q1D, 2, NF);
const bool const_q = (q.Size() == 1);
const auto Q =
const_q ? Reshape(q.Read(), 1, 1) : Reshape(q.Read(), Q1D, NF);
const bool const_q = (q.Size() == coeff_dim);
const auto Q = const_q ? Reshape(q.Read(), coeff_dim, 1, 1)
: Reshape(q.Read(), coeff_dim, Q1D, NF);
const auto W = w.Read();
@@ -53,6 +53,12 @@ static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
// (q, 1/h, J0_0, J0_1, J1_0, J1_1)
auto pa = Reshape(pa_data.Write(), 6, Q1D, NF);
auto get_coeff = [const_q] MFEM_HOST_DEVICE (const decltype(Q) &Q, int i,
int qx, int e)
{
return const_q ? Q(i,0,0) : Q(i,qx,e);
};
mfem::forall(NF, [=] MFEM_HOST_DEVICE(int f) -> void
{
const int normal_dir[] = {face_info(0, f), face_info(1, f)};
@@ -71,10 +77,26 @@ static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
for (int p = 0; p < Q1D; ++p)
{
const real_t Qp = const_q ? Q(0, 0) : Q(p, f);
pa(0, p, f) = kappa * Qp * W[p] * detJf(p, f);
real_t qh = 0.0;
real_t hi = 0.0;
real_t Qtn[2];
if (coeff_dim > 1)
{
// matrix coefficient
Qtn[0] = get_coeff(Q,0,p,f)*n(p,0,f) + get_coeff(Q,1,p,f)*n(p,1,f);
Qtn[1] = get_coeff(Q,2,p,f)*n(p,0,f) + get_coeff(Q,3,p,f)*n(p,1,f);
qh = Qtn[0]*n(p,0,f) + Qtn[1]*n(p,1,f);
}
else
{
qh = get_coeff(Q, 0, p, f);
Qtn[0] = qh*n(p,0,f);
Qtn[1] = qh*n(p,1,f);
}
pa(0, p, f) = kappa * qh * W[p] * detJf(p, f);
for (int side = 0; side < nsides; ++side)
{
int i, j;
@@ -89,15 +111,13 @@ static void PADGDiffusionSetup2D(const int Q1D, const int NE, const int NF,
const auto &detJ = (side == 1 && shared) ? detJ_shared : detJe_loc;
real_t nJi[2];
nJi[0] =
n(p, 0, f) * J(i, j, 1, 1, e) - n(p, 1, f) * J(i, j, 0, 1, e);
nJi[1] =
-n(p, 0, f) * J(i, j, 1, 0, e) + n(p, 1, f) * J(i, j, 0, 0, e);
nJi[0] = Qtn[0]*J(i, j, 1, 1, e) - Qtn[1]*J(i, j, 0, 1, e);
nJi[1] = -Qtn[0]*J(i, j, 1, 0, e) + Qtn[1]*J(i, j, 0, 0, e);
const real_t dJe = detJ(i, j, e);
const real_t dJf = detJf(p, f);
const real_t w = factor * Qp * W[p] * dJf / dJe;
const real_t w = factor * W[p] * dJf / dJe;
const int ni = normal_dir[side];
const int ti = 1 - ni;
@@ -126,9 +146,9 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
const GeometricFactors &el_geom,
const FaceGeometricFactors &face_geom,
const FaceNeighborGeometricFactors *nbr_geom,
const Vector &q, const real_t sigma,
const real_t kappa, Vector &pa_data,
const Array<int> &face_info_)
const Vector &q, const int coeff_dim,
const real_t sigma, const real_t kappa,
Vector &pa_data, const Array<int> &face_info_)
{
const auto J_loc = Reshape(el_geom.J.Read(), Q1D, Q1D, Q1D, 3, 3, NE);
const auto detJe_loc = Reshape(el_geom.detJ.Read(), Q1D, Q1D, Q1D, NE);
@@ -142,9 +162,9 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
const auto detJf = Reshape(face_geom.detJ.Read(), Q1D, Q1D, NF);
const auto n = Reshape(face_geom.normal.Read(), Q1D, Q1D, 3, NF);
const bool const_q = (q.Size() == 1);
const auto Q =
const_q ? Reshape(q.Read(), 1, 1, 1) : Reshape(q.Read(), Q1D, Q1D, NF);
const bool const_q = (q.Size() == coeff_dim);
const auto Q = const_q ? Reshape(q.Read(), coeff_dim, 1, 1, 1)
: Reshape(q.Read(), coeff_dim, Q1D, Q1D, NF);
const auto W = Reshape(w.Read(), Q1D, Q1D);
@@ -157,6 +177,12 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
// (J00, J01, J02, J10, J11, J12, q/h)
const auto pa = Reshape(pa_data.Write(), 7, Q1D, Q1D, NF);
auto get_coeff = [const_q] MFEM_HOST_DEVICE (const decltype(Q) &Q, int i,
int qx, int qy, int e)
{
return const_q ? Q(i,0,0,0) : Q(i,qx,qy,e);
};
mfem::forall_2D(NF, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int f) -> void
{
MFEM_SHARED int perm[2][3];
@@ -192,11 +218,32 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
{
MFEM_FOREACH_THREAD(p2, y, Q1D)
{
const real_t Qp = const_q ? Q(0, 0, 0) : Q(p1, p2, f);
const real_t dJf = detJf(p1, p2, f);
real_t hi = 0.0;
real_t Qtn[3];
real_t qh = 0.0;
if (coeff_dim > 1)
{
// matrix coefficient
for (int d = 0; d < 3; ++d)
{
Qtn[d] = get_coeff(Q,0+3*d,p1,p2,f)*n(p1,p2,0,f)
+ get_coeff(Q,1+3*d,p1,p2,f)*n(p1,p2,1,f)
+ get_coeff(Q,2+3*d,p1,p2,f)*n(p1,p2,2,f);
qh += Qtn[d] * n(p1,p2,d,f);
}
}
else
{
qh = get_coeff(Q,0,p1,p2,f);
Qtn[0] = qh * n(p1,p2,0,f);
Qtn[1] = qh * n(p1,p2,1,f);
Qtn[2] = qh * n(p1,p2,2,f);
}
for (int side = 0; side < nsides; ++side)
{
int i, j, k;
@@ -210,38 +257,29 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
// *INDENT-OFF*
real_t nJi[3];
nJi[0] = (-J(i, j, k, 1, 2, e) * J(i, j, k, 2, 1, e) +
J(i, j, k, 1, 1, e) * J(i, j, k, 2, 2, e)) *
n(p1, p2, 0, f) +
(J(i, j, k, 0, 2, e) * J(i, j, k, 2, 1, e) -
J(i, j, k, 0, 1, e) * J(i, j, k, 2, 2, e)) *
n(p1, p2, 1, f) +
(-J(i, j, k, 0, 2, e) * J(i, j, k, 1, 1, e) +
J(i, j, k, 0, 1, e) * J(i, j, k, 1, 2, e)) *
n(p1, p2, 2, f);
J(i, j, k, 1, 1, e) * J(i, j, k, 2, 2, e)) * Qtn[0] +
(J(i, j, k, 0, 2, e) * J(i, j, k, 2, 1, e) -
J(i, j, k, 0, 1, e) * J(i, j, k, 2, 2, e)) * Qtn[1] +
(-J(i, j, k, 0, 2, e) * J(i, j, k, 1, 1, e) +
J(i, j, k, 0, 1, e) * J(i, j, k, 1, 2, e)) * Qtn[2];
nJi[1] = (J(i, j, k, 1, 2, e) * J(i, j, k, 2, 0, e) -
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 2, e)) *
n(p1, p2, 0, f) +
(-J(i, j, k, 0, 2, e) * J(i, j, k, 2, 0, e) +
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 2, e)) *
n(p1, p2, 1, f) +
(J(i, j, k, 0, 2, e) * J(i, j, k, 1, 0, e) -
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 2, e)) *
n(p1, p2, 2, f);
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 2, e)) * Qtn[0] +
(-J(i, j, k, 0, 2, e) * J(i, j, k, 2, 0, e) +
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 2, e)) * Qtn[1] +
(J(i, j, k, 0, 2, e) * J(i, j, k, 1, 0, e) -
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 2, e)) * Qtn[2];
nJi[2] = (-J(i, j, k, 1, 1, e) * J(i, j, k, 2, 0, e) +
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 1, e)) *
n(p1, p2, 0, f) +
(J(i, j, k, 0, 1, e) * J(i, j, k, 2, 0, e) -
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 1, e)) *
n(p1, p2, 1, f) +
(-J(i, j, k, 0, 1, e) * J(i, j, k, 1, 0, e) +
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 1, e)) *
n(p1, p2, 2, f);
J(i, j, k, 1, 0, e) * J(i, j, k, 2, 1, e)) * Qtn[0] +
(J(i, j, k, 0, 1, e) * J(i, j, k, 2, 0, e) -
J(i, j, k, 0, 0, e) * J(i, j, k, 2, 1, e)) * Qtn[1] +
(-J(i, j, k, 0, 1, e) * J(i, j, k, 1, 0, e) +
J(i, j, k, 0, 0, e) * J(i, j, k, 1, 1, e)) * Qtn[2];
// *INDENT-ON*
const real_t dJe = detJe(i, j, k, e);
const real_t val = factor * Qp * W(p1, p2) * dJf / dJe;
const real_t val = factor * W(p1, p2) * dJf / dJe;
for (int d = 0; d < 3; ++d)
{
@@ -260,7 +298,7 @@ static void PADGDiffusionSetup3D(const int Q1D, const int NE, const int NF,
pa(5, p1, p2, f) = 0.0;
}
pa(6, p1, p2, f) = kappa * hi * Qp * W(p1, p2) * dJf;
pa(6, p1, p2, f) = kappa * hi * qh * W(p1, p2) * dJf;
}
}
});
@@ -502,19 +540,12 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
// Evaluate the coefficient at the face quadrature points.
FaceQuadratureSpace fqs(mesh, ir, type);
CoefficientVector q(fqs, CoefficientStorage::COMPRESSED);
if (Q)
{
q.Project(*Q);
}
else if (MQ)
{
MFEM_ABORT("Not yet implemented"); /* q.Project(*MQ); */
}
else
{
q.SetConstant(1.0);
}
CoefficientVector q(fqs, CoefficientStorage::CONSTANTS);
if (Q) { q.Project(*Q); }
else if (MQ) { q.Project(*MQ); }
else { q.SetConstant(1.0); }
const int coeff_dim = q.GetVDim();
Array<int> face_info;
if (dim == 1)
@@ -525,15 +556,15 @@ void DGDiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
{
PADGDiffusionSetupFaceInfo2D(nf, mesh, type, face_info);
PADGDiffusionSetup2D(quad1D, ne, nf, ir.GetWeights(), *el_geom,
*face_geom, nbr_geom.get(), q, sigma, kappa, pa_data,
face_info);
*face_geom, nbr_geom.get(), q, coeff_dim, sigma,
kappa, pa_data, face_info);
}
else if (dim == 3)
{
PADGDiffusionSetupFaceInfo3D(nf, mesh, type, face_info);
PADGDiffusionSetup3D(quad1D, ne, nf, ir.GetWeights(), *el_geom,
*face_geom, nbr_geom.get(), q, sigma, kappa, pa_data,
face_info);
*face_geom, nbr_geom.get(), q, coeff_dim, sigma,
kappa, pa_data, face_info);
}
}
+40 -29
View File
@@ -134,14 +134,19 @@ void PADiffusionSetup2D<2>(const int Q1D,
Vector &d)
{
const bool symmetric = (coeffDim != 4);
const bool const_c = c.Size() == 1;
MFEM_VERIFY(coeffDim < 3 ||
!const_c, "Constant matrix coefficient not supported");
const bool const_c = c.Size() == coeffDim;
const auto W = Reshape(w.Read(), Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,2,2,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
const auto C = const_c ? Reshape(c.Read(), coeffDim,1,1,1) :
Reshape(c.Read(), coeffDim,Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D, symmetric ? 3 : 4, NE);
auto get_coeff = [const_c] MFEM_HOST_DEVICE
(const decltype(C) &C, int i, int qx, int qy, int e)
{
return const_c ? C(i,0,0,0) : C(i,qx,qy,e);
};
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
@@ -156,10 +161,11 @@ void PADiffusionSetup2D<2>(const int Q1D,
if (coeffDim == 3 || coeffDim == 4) // Matrix coefficient
{
// First compute entries of R = MJ^{-T}, without det J factor.
const real_t M11 = C(0,qx,qy,e);
const real_t M12 = C(1,qx,qy,e);
const real_t M21 = symmetric ? M12 : C(2,qx,qy,e);
const real_t M22 = symmetric ? C(2,qx,qy,e) : C(3,qx,qy,e);
const real_t M11 = get_coeff(C,0,qx,qy,e);
const real_t M12 = get_coeff(C,1,qx,qy,e);
const real_t M21 = symmetric ? M12 : get_coeff(C,2,qx,qy,e);
const real_t M22 = symmetric ? get_coeff(C,2,qx,qy,e)
: get_coeff(C,3,qx,qy,e);
const real_t R11 = M11*J22 - M12*J12;
const real_t R21 = M21*J22 - M22*J12;
const real_t R12 = -M11*J21 + M12*J11;
@@ -177,9 +183,8 @@ void PADiffusionSetup2D<2>(const int Q1D,
}
else // Vector or scalar coefficient
{
const real_t C1 = const_c ? C(0,0,0,0) : C(0,qx,qy,e);
const real_t C2 = const_c ? C(0,0,0,0) :
(coeffDim == 2 ? C(1,qx,qy,e) : C(0,qx,qy,e));
const real_t C1 = get_coeff(C,0,qx,qy,e);
const real_t C2 = get_coeff(C,coeffDim==2?1:0,qx,qy,e);
D(qx,qy,0,e) = w_detJ * (C2*J12*J12 + C1*J22*J22); // 1,1
D(qx,qy,1,e) = -w_detJ * (C2*J12*J11 + C1*J22*J21); // 1,2
@@ -244,14 +249,19 @@ void PADiffusionSetup3D(const int Q1D,
Vector &d)
{
const bool symmetric = (coeffDim != 9);
const bool const_c = c.Size() == 1;
MFEM_VERIFY(coeffDim < 6 ||
!const_c, "Constant matrix coefficient not supported");
const bool const_c = c.Size() == coeffDim;
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1,1) :
const auto C = const_c ? Reshape(c.Read(), coeffDim,1,1,1,1) :
Reshape(c.Read(), coeffDim,Q1D,Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, symmetric ? 6 : 9, NE);
auto get_coeff = [const_c] MFEM_HOST_DEVICE
(const decltype(C) &C, int i, int qx, int qy, int qz, int e)
{
return const_c ? C(i,0,0,0,0) : C(i,qx,qy,qz,e);
};
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
@@ -287,15 +297,18 @@ void PADiffusionSetup3D(const int Q1D,
if (coeffDim == 6 || coeffDim == 9) // Matrix coefficient version
{
// Compute entries of R = MJ^{-T} = M adj(J)^T, without det J.
const real_t M11 = C(0, qx,qy,qz, e);
const real_t M12 = C(1, qx,qy,qz, e);
const real_t M13 = C(2, qx,qy,qz, e);
const real_t M21 = (!symmetric) ? C(3, qx,qy,qz, e) : M12;
const real_t M22 = (!symmetric) ? C(4, qx,qy,qz, e) : C(3, qx,qy,qz, e);
const real_t M23 = (!symmetric) ? C(5, qx,qy,qz, e) : C(4, qx,qy,qz, e);
const real_t M31 = (!symmetric) ? C(6, qx,qy,qz, e) : M13;
const real_t M32 = (!symmetric) ? C(7, qx,qy,qz, e) : M23;
const real_t M33 = (!symmetric) ? C(8, qx,qy,qz, e) : C(5, qx,qy,qz, e);
const real_t M11 = get_coeff(C, 0, qx,qy,qz, e);
const real_t M12 = get_coeff(C, 1, qx,qy,qz, e);
const real_t M13 = get_coeff(C, 2, qx,qy,qz, e);
const real_t M21 = (!symmetric) ? get_coeff(C, 3, qx,qy,qz, e) : M12;
const real_t M22 = (!symmetric) ? get_coeff(C, 4, qx,qy,qz, e)
: get_coeff(C, 3, qx,qy,qz, e);
const real_t M23 = (!symmetric) ? get_coeff(C, 5, qx,qy,qz, e)
: get_coeff(C, 4, qx,qy,qz, e);
const real_t M31 = (!symmetric) ? get_coeff(C, 6, qx,qy,qz, e) : M13;
const real_t M32 = (!symmetric) ? get_coeff(C, 7, qx,qy,qz, e) : M23;
const real_t M33 = (!symmetric) ? get_coeff(C, 8, qx,qy,qz, e)
: get_coeff(C, 5, qx,qy,qz, e);
const real_t R11 = M11*A11 + M12*A12 + M13*A13;
const real_t R12 = M11*A21 + M12*A22 + M13*A23;
@@ -335,11 +348,9 @@ void PADiffusionSetup3D(const int Q1D,
}
else // Vector or scalar coefficient version
{
const real_t C1 = const_c ? C(0,0,0,0,0) : C(0,qx,qy,qz,e);
const real_t C2 = const_c ? C(0,0,0,0,0) :
(coeffDim == 3 ? C(1,qx,qy,qz,e) : C(0,qx,qy,qz,e));
const real_t C3 = const_c ? C(0,0,0,0,0) :
(coeffDim == 3 ? C(2,qx,qy,qz,e) : C(0,qx,qy,qz,e));
const real_t C1 = get_coeff(C,0,qx,qy,qz,e);
const real_t C2 = get_coeff(C,coeffDim==3?1:0,qx,qy,qz,e);
const real_t C3 = get_coeff(C,coeffDim==3?2:0,qx,qy,qz,e);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
D(qx,qy,qz,0,e) = w_detJ * (C1*A11*A11 + C2*A12*A12 + C3*A13*A13); // 1,1
+1 -1
View File
@@ -201,7 +201,7 @@ inline void SmemPADiffusionDiagonal2D(const int NE,
MFEM_SHARED real_t BG[2][MQ1*MD1];
real_t (*B)[MD1] = (real_t (*)[MD1]) (BG+0);
real_t (*G)[MD1] = (real_t (*)[MD1]) (BG+1);
MFEM_SHARED real_t QD[3][NBZ][MD1][MQ1];
MFEM_SHARED real_t QD[3][NBZ][MQ1][MD1];
real_t (*QD0)[MD1] = (real_t (*)[MD1])(QD[0] + tidz);
real_t (*QD1)[MD1] = (real_t (*)[MD1])(QD[1] + tidz);
real_t (*QD2)[MD1] = (real_t (*)[MD1])(QD[2] + tidz);
-1
View File
@@ -18,7 +18,6 @@
namespace mfem
{
class Operator;
class LinearForm;
/// Class extending the LinearForm class to support assembly on devices.
+950
View File
@@ -0,0 +1,950 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "particleset.hpp"
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
// Ignore warnings from the gslib header (GCC version)
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
namespace gslib
{
extern "C"
{
#include <gslib.h>
} // extern C
} // namespace gslib
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
#endif // MFEM_USE_MPI && MFEM_USE_GSLIB
namespace mfem
{
Particle::Particle(int dim, const Array<int> &field_vdims, int num_tags)
: coords(dim), fields(), tags()
{
coords = 0.0;
fields.reserve(field_vdims.Size());
for (int f = 0; f < field_vdims.Size(); f++)
{
fields.emplace_back(field_vdims[f]);
fields.back() = 0.0;
}
tags.reserve(num_tags);
for (int t = 0; t < num_tags; t++)
{
tags.emplace_back(1);
tags.back()[0] = 0;
}
}
void Particle::SetTagRef(int t, int *tag_data)
{
MFEM_ASSERT(t >= 0 &&
static_cast<size_t>(t) < tags.size(), "Invalid tag index");
tags[t].MakeRef(tag_data, 1);
}
void Particle::SetFieldRef(int f, real_t *field_data)
{
MFEM_ASSERT(f >= 0 &&
static_cast<size_t>(f) < fields.size(), "Invalid field "
"index");
Vector temp(field_data, fields[f].Size());
fields[f].MakeRef(temp, 0, fields[f].Size());
}
bool Particle::operator==(const Particle &rhs) const
{
// Compare coordinate size and values
if (coords.Size() != rhs.coords.Size())
{
return false;
}
for (int d = 0; d < coords.Size(); d++)
{
if (coords[d] != rhs.coords[d])
{
return false;
}
}
// Compare fields vdim and values
if (fields.size() != rhs.fields.size())
{
return false;
}
for (size_t f = 0; f < fields.size(); f++)
{
if (fields[f].Size() != rhs.fields[f].Size())
{
return false;
}
for (int c = 0; c < fields[f].Size(); c++)
{
if (fields[f][c] != rhs.fields[f][c])
{
return false;
}
}
}
// Compare tags size and values
if (tags.size() != rhs.tags.size())
{
return false;
}
for (size_t t = 0; t < tags.size(); t++)
{
if (tags[t][0] != rhs.tags[t][0])
{
return false;
}
}
return true;
}
void Particle::Print(std::ostream &os) const
{
os << "Coords: (";
for (int d = 0; d < coords.Size(); d++)
{
os << coords[d] << ( (d+1 < coords.Size()) ? "," : ")\n");
}
for (size_t f = 0; f < fields.size(); f++)
{
os << "Field " << f << ": (";
for (int c = 0; c < fields[f].Size(); c++)
{
os << fields[f][c] << ( (c+1 < fields[f].Size()) ? "," : ")\n");
}
}
for (size_t t = 0; t < tags.size(); t++)
{
os << "Tag " << t << ": " << tags[t][0] << "\n";
}
}
Array<Ordering::Type> ParticleSet::GetOrderingArray(Ordering::Type o, int N)
{
Array<Ordering::Type> ordering_arr(N);
ordering_arr = o;
return ordering_arr;
}
std::string ParticleSet::GetDefaultFieldName(int i)
{
return "Field_" + std::to_string(i);
}
std::string ParticleSet::GetDefaultTagName(int i)
{
return "Tag_" + std::to_string(i);
}
Array<const char*> ParticleSet::GetEmptyNameArray(int N)
{
Array<const char*> names(N);
names = nullptr;
return names;
}
#ifdef MFEM_USE_MPI
int ParticleSet::GetRank(MPI_Comm comm_)
{
int r; MPI_Comm_rank(comm_, &r);
return r;
}
int ParticleSet::GetSize(MPI_Comm comm_)
{
int s; MPI_Comm_size(comm_, &s);
return s;
}
#endif // MFEM_USE_MPI
void ParticleSet::Reserve(int res)
{
ids.Reserve(res);
// Reserve fields
for (int f = -1; f < GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? coords : *fields[f]);
pv.Reserve(res*pv.GetVDim());
}
// Reserve tags
for (int t = 0; t < GetNTags(); t++)
{
tags[t]->Reserve(res);
}
}
const Array<int> ParticleSet::GetFieldVDims() const
{
Array<int> field_vdims(GetNFields());
for (int f = 0; f < GetNFields(); f++)
{
field_vdims[f] = Field(f).GetVDim();
}
return field_vdims;
}
void ParticleSet::AddParticles(const Array<IDType> &new_ids,
Array<int> *new_indices)
{
int num_add = new_ids.Size();
int old_np = GetNParticles();
int new_np = old_np + num_add;
// Set indices of new particles
if (new_indices)
{
new_indices->SetSize(num_add);
for (int i = 0; i < num_add; i++)
{
(*new_indices)[i] = ids.Size() + i;
}
}
// Add new ids
ids.Append(new_ids);
// Update data
for (int f = -1; f < GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? coords : *fields[f]);
pv.SetNumParticles(new_np); // does not delete existing data
}
// Update tags
for (int t = 0; t < GetNTags(); t++)
{
tags[t]->SetSize(new_np);
}
}
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
/// \cond DO_NOT_DOCUMENT
template<size_t NBytes>
void ParticleSet::TransferParticlesImpl(ParticleSet &pset,
const Array<int> &send_idxs,
const Array<unsigned int> &send_ranks)
{
struct pdata_t
{
alignas(real_t) std::array<std::byte, NBytes> data;
IDType id;
};
int nreals = pset.GetFieldVDims().Sum() + pset.Coords().GetVDim();
int ntags = pset.GetNTags();
size_t nbytes = nreals*sizeof(real_t) + ntags*sizeof(int);
MFEM_VERIFY(nbytes <= NBytes, "More data than can be packed.");
using parr_t = pdata_t;
gslib::array gsl_arr;
parr_t *pdata_arr;
array_init(parr_t, &gsl_arr, send_idxs.Size());
pdata_arr = (parr_t*) gsl_arr.ptr;
gsl_arr.n = send_idxs.Size();
for (int i = 0; i < send_idxs.Size(); i++)
{
parr_t &pdata = pdata_arr[i];
pdata.id = pset.GetIDs()[send_idxs[i]];
// Copy particle data directly into pdata
size_t counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
for (int c = 0; c < pv.GetVDim(); c++)
{
std::memcpy(pdata.data.data() + counter, &pv(send_idxs[i], c),
sizeof(real_t));
counter += sizeof(real_t);
}
}
// Copy tags
for (int t = 0; t < pset.GetNTags(); t++)
{
Array<int> &tag_arr = pset.Tag(t);
std::memcpy(pdata.data.data() + counter, &tag_arr[send_idxs[i]],
sizeof(int));
counter += sizeof(int);
}
}
int nparticles = pset.GetNParticles();
int nsend = send_idxs.Size();
// Transfer particles
sarray_transfer_ext(parr_t, &gsl_arr, send_ranks.GetData(),
sizeof(unsigned int), pset.cr);
// Make sure we have enough space for received particles
int nrecv = (int) gsl_arr.n;
int ndelete = nsend - nrecv;
if (ndelete > 0)
{
// Remove unneeded particles
auto datap = const_cast<int*>(send_idxs.GetData());
Array<int> delete_idxs(datap + nrecv, ndelete);
pset.RemoveParticles(delete_idxs);
}
else
{
pset.Reserve(nparticles-ndelete);
}
pdata_arr = (parr_t*) gsl_arr.ptr;
// Add newly-recvd data directly to active state
for (int i = 0; i < nrecv; i++)
{
parr_t &pdata = pdata_arr[i];
IDType id = pdata.id;
int new_loc_idx;
if (i < nsend) // update existing particle
{
new_loc_idx = send_idxs[i];
pset.UpdateID(new_loc_idx, id);
}
else
{
// add new particle
Array<int> idx_temp;
pset.AddParticles(Array<IDType>({id}), &idx_temp);
new_loc_idx = idx_temp[0]; // Get index of newly-added particle
}
size_t counter = 0;
for (int f = -1; f < pset.GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? pset.Coords() : pset.Field(f));
for (int c = 0; c < pv.GetVDim(); c++)
{
real_t& val = pv(new_loc_idx, c);
std::memcpy(&val, pdata.data.data() + counter, sizeof(real_t));
counter += sizeof(real_t);
}
}
for (int t = 0; t < pset.GetNTags(); t++)
{
Array<int> &tag_arr = pset.Tag(t);
std::memcpy(&tag_arr[new_loc_idx],
pdata.data.data() + counter, sizeof(int));
counter += sizeof(int);
}
}
array_free(&gsl_arr);
}
template<size_t NBytes>
ParticleSet::TransferParticlesType ParticleSet::TransferParticles::Kernel()
{
return &ParticleSet::TransferParticlesImpl<NBytes>;
}
ParticleSet::Kernels::Kernels()
{
constexpr size_t sizd = sizeof(real_t);
TransferParticles::Specialization<2*sizd>::Add();
TransferParticles::Specialization<3*sizd>::Add();
TransferParticles::Specialization<4*sizd>::Add();
TransferParticles::Specialization<8*sizd>::Add();
TransferParticles::Specialization<12*sizd>::Add();
TransferParticles::Specialization<16*sizd>::Add();
TransferParticles::Specialization<20*sizd>::Add();
TransferParticles::Specialization<24*sizd>::Add();
TransferParticles::Specialization<28*sizd>::Add();
TransferParticles::Specialization<32*sizd>::Add();
TransferParticles::Specialization<36*sizd>::Add();
TransferParticles::Specialization<40*sizd>::Add();
}
auto ParticleSet::TransferParticles::Fallback(size_t bufsize)
-> ParticleSet::TransferParticlesType
{
constexpr size_t sizd = sizeof(real_t);
if (bufsize < 4*sizd)
{
return &ParticleSet::TransferParticlesImpl<4*sizd>;
}
else if (bufsize < 8*sizd)
{
return &ParticleSet::TransferParticlesImpl<8*sizd>;
}
else if (bufsize < 12*sizd)
{
return &ParticleSet::TransferParticlesImpl<12*sizd>;
}
else if (bufsize < 16*sizd)
{
return &ParticleSet::TransferParticlesImpl<16*sizd>;
}
else if (bufsize < 20*sizd)
{
return &ParticleSet::TransferParticlesImpl<20*sizd>;
}
else if (bufsize < 24*sizd)
{
return &ParticleSet::TransferParticlesImpl<24*sizd>;
}
else if (bufsize < 28*sizd)
{
return &ParticleSet::TransferParticlesImpl<28*sizd>;
}
else if (bufsize < 32*sizd)
{
return &ParticleSet::TransferParticlesImpl<32*sizd>;
}
else if (bufsize < 36*sizd)
{
return &ParticleSet::TransferParticlesImpl<36*sizd>;
}
else if (bufsize < 40*sizd)
{
return &ParticleSet::TransferParticlesImpl<40*sizd>;
}
return &ParticleSet::TransferParticlesImpl<60*sizd>;
}
/// \endcond DO_NOT_DOCUMENT
void ParticleSet::Redistribute(const Array<unsigned int> &rank_list)
{
MFEM_ASSERT(rank_list.Size() == GetNParticles(),
"rank_list must be of size GetNParticles().");
int rank = GetRank(comm);
// Get particles to be transferred
// (Avoid unnecessary copies of particle data into and out of buffers)
Array<int> send_idxs;
Array<unsigned int> send_ranks;
send_idxs.Reserve(rank_list.Size());
send_ranks.Reserve(rank_list.Size());
for (int i = 0; i < rank_list.Size(); i++)
{
if (rank != static_cast<int>(rank_list[i]))
{
send_idxs.Append(i);
send_ranks.Append(rank_list[i]);
}
}
// Compute number of bytes of a single particle
int nreals = GetFieldVDims().Sum() + coords.GetVDim();
int ntags = GetNTags();
size_t nbytes = nreals*sizeof(real_t) + ntags*sizeof(int);
// Dispatch to appropriate redistribution function for this size
TransferParticles::Run(nbytes, *this, send_idxs, send_ranks);
}
#endif // MFEM_USE_MPI && MFEM_USE_GSLIB
Particle ParticleSet::CreateParticle() const
{
return Particle(GetDim(), GetFieldVDims(), GetNTags());
}
void ParticleSet::WriteToFile(const char *fname,
const std::stringstream &ss_header, const std::stringstream &ss_data)
{
#ifdef MFEM_USE_MPI
// Parallel:
int rank = GetRank(comm);
MPI_File_delete(fname, MPI_INFO_NULL); // delete old file if it exists
MPI_File file;
int mpi_err = MPI_File_open(comm, fname, MPI_MODE_CREATE | MPI_MODE_WRONLY,
MPI_INFO_NULL, &file);
MFEM_VERIFY(mpi_err == MPI_SUCCESS, "MPI_File_open failed.");
// Print header
if (rank == 0)
{
MPI_File_write_at(file, 0, ss_header.str().data(), ss_header.str().size(),
MPI_CHAR, MPI_STATUS_IGNORE);
}
// Compute the data size in bytes
MPI_Offset data_size = ss_data.str().size();
MPI_Offset offset;
// Compute the offsets using an exclusive scan
MPI_Exscan(&data_size, &offset, 1, MPI_OFFSET, MPI_SUM, comm);
if (rank == 0)
{
offset = 0;
}
// Add offset from the header
offset += ss_header.str().size();
// Write data collectively
MPI_File_write_at_all(file, offset, ss_data.str().data(),
data_size, MPI_BYTE, MPI_STATUS_IGNORE);
// Close file
MPI_File_close(&file);
#else
// Serial:
std::ofstream ofs(fname);
MFEM_VERIFY(ofs.is_open() && !ofs.fail(),
"Error: Could not open file " << fname << " for writing.");
ofs << ss_header.str() << ss_data.str();
ofs.close();
#endif // MFEM_USE_MPI
}
ParticleSet::ParticleSet(int id_stride_, IDType id_counter_, int num_particles,
int dim, Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
: id_stride(id_stride_),
id_counter(id_counter_),
coords(dim, coords_ordering)
{
// Initialize fields
for (int f = 0; f < field_vdims.Size(); f++)
{
AddField(field_vdims[f], field_orderings[f], field_names_[f]);
}
// Initialize tags
for (int t = 0; t < num_tags; t++)
{
AddTag(tag_names_[t]);
}
// Add num_particles
Array<IDType> init_ids(num_particles);
for (int i = 0; i < num_particles; i++)
{
init_ids[i] = id_counter;
id_counter += id_stride;
}
AddParticles(init_ids);
}
bool ParticleSet::IsValidParticle(const Particle &p) const
{
if (p.GetDim() != GetDim())
{
return false;
}
if (p.GetNFields() != GetNFields())
{
return false;
}
for (int f = 0; f < GetNFields(); f++)
{
if (p.GetFieldVDim(f) != Field(f).GetVDim())
{
return false;
}
}
if (p.GetNTags() != GetNTags())
{
return false;
}
return true;
}
ParticleSet::ParticleSet(int num_particles, int dim,
Ordering::Type coords_ordering)
: ParticleSet(1, 0, num_particles, dim, coords_ordering, Array<int>(),
Array<Ordering::Type>(), Array<const char*>(), 0,
Array<const char*>())
{
}
ParticleSet::ParticleSet(int num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering)
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
GetEmptyNameArray(field_vdims.Size()), num_tags,
GetEmptyNameArray(num_tags))
{
}
ParticleSet::ParticleSet(int num_particles, int dim,
const Array<int> &field_vdims, const Array<const
char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_,
Ordering::Type all_ordering)
: ParticleSet(1, 0, num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
field_names_, num_tags,
tag_names_)
{
}
ParticleSet::ParticleSet(int num_particles, int dim,
Ordering::Type coords_ordering,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
: ParticleSet(1, 0, num_particles, dim, coords_ordering, field_vdims,
field_orderings, field_names_, num_tags, tag_names_)
{
}
#ifdef MFEM_USE_MPI
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering)
: ParticleSet(comm_, rank_num_particles, dim, coords_ordering, Array<int>(),
Array<Ordering::Type>(), Array<const char*>(), 0,
Array<const char*>())
{
};
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering)
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
GetEmptyNameArray(field_vdims.Size()), num_tags,
GetEmptyNameArray(num_tags))
{
}
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, const Array<const
char*> &field_names_,
int num_tags, const Array<const char*> &tag_names_,
Ordering::Type all_ordering)
: ParticleSet(comm_, rank_num_particles, dim, all_ordering, field_vdims,
GetOrderingArray(all_ordering, field_vdims.Size()),
field_names_, num_tags,
tag_names_)
{
}
ParticleSet::ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_)
: ParticleSet(GetSize(comm_), (IDType)GetRank(comm_),
rank_num_particles,
dim,
coords_ordering,
field_vdims,
field_orderings,
field_names_,
num_tags,
tag_names_)
{
comm = comm_;
#ifdef MFEM_USE_GSLIB
gsl_comm = new gslib::comm;
cr = new gslib::crystal;
comm_init(gsl_comm, comm);
crystal_init(cr, gsl_comm);
#endif // MFEM_USE_GSLIB
}
#endif // MFEM_USE_MPI
ParticleSet::IDType ParticleSet::GetGlobalNParticles() const
{
IDType total = (IDType)GetNParticles();
#ifdef MFEM_USE_MPI
MPI_Allreduce(MPI_IN_PLACE, &total, 1, MPI_UNSIGNED_LONG_LONG,
MPI_SUM, comm);
#endif // MFEM_USE_MPI
return total;
}
int ParticleSet::AddField(int vdim, Ordering::Type field_ordering,
const char* field_name)
{
std::string field_name_str(field_name ? field_name : "");
if (!field_name)
{
field_name_str = GetDefaultFieldName(field_names.size());
}
fields.emplace_back(std::make_unique<ParticleVector>(vdim, field_ordering,
GetNParticles()));
field_names.emplace_back(field_name_str);
return GetNFields() - 1;
}
int ParticleSet::AddTag(const char* tag_name)
{
std::string tag_name_str(tag_name ? tag_name : "");
if (!tag_name)
{
tag_name_str = GetDefaultTagName(tag_names.size());
}
tags.emplace_back(std::make_unique<Array<int>>(GetNParticles()));
tag_names.emplace_back(tag_name_str);
return GetNTags() - 1;
}
void ParticleSet::AddParticle(const Particle &p)
{
MFEM_ASSERT(IsValidParticle(p),
"Particle is incompatible with ParticleSet.");
// Add the particle
Array<int> idxs;
AddParticles(Array<IDType>({id_counter}), &idxs);
id_counter += id_stride;
// Set the new particle data
int idx = idxs[0];
SetParticle(idx, p);
}
void ParticleSet::AddParticles(int num_particles, Array<int> *new_indices)
{
Array<IDType> add_ids(num_particles);
for (int i = 0; i < num_particles; i++)
{
add_ids[i] = id_counter;
id_counter += id_stride;
}
AddParticles(add_ids, new_indices);
}
void ParticleSet::RemoveParticles(const Array<int> &list)
{
// Delete IDs
ids.DeleteAt(list);
// Delete data
for (int f = -1; f < GetNFields(); f++)
{
ParticleVector &pv = (f == -1 ? coords : *fields[f]);
pv.DeleteParticles(list);
}
// Delete tags
for (int t = 0; t < GetNTags(); t++)
{
tags[t]->DeleteAt(list);
}
}
Particle ParticleSet::GetParticle(int i) const
{
Particle p = CreateParticle();
Coords().GetValues(i, p.Coords());
for (int f = 0; f < GetNFields(); f++)
{
Field(f).GetValues(i, p.Field(f));
}
for (int t = 0; t < GetNTags(); t++)
{
p.Tag(t) = Tag(t)[i];
}
return p;
}
bool ParticleSet::IsParticleRefValid() const
{
if (coords.GetOrdering() == Ordering::byNODES)
{
return false;
}
for (int f = 0; f < GetNFields(); f++)
{
if (fields[f]->GetOrdering() == Ordering::byNODES)
{
return false;
}
}
return true;
}
Particle ParticleSet::GetParticleRef(int i)
{
Particle p = CreateParticle();
Coords().GetValuesRef(i, p.Coords());
for (int f = 0; f < GetNFields(); f++)
{
MFEM_ASSERT(Field(f).GetOrdering() == Ordering::byVDIM,
"GetParticleRef only valid when all fields ordered byVDIM.");
p.SetFieldRef(f, Field(f).GetData() + i*Field(f).GetVDim());
}
for (int t = 0; t < GetNTags(); t++)
{
p.SetTagRef(t, &(*tags[t])[i]);
}
return p;
}
void ParticleSet::SetParticle(int i, const Particle &p)
{
MFEM_ASSERT(IsValidParticle(p),
"Particle is incompatible with ParticleSet.");
Coords().SetValues(i, p.Coords());
for (int f = 0; f < GetNFields(); f++)
{
Field(f).SetValues(i, p.Field(f));
}
for (int t = 0; t < GetNTags(); t++)
{
Tag(t)[i] = p.Tag(t);
}
}
void ParticleSet::PrintCSV(const char *fname, int precision)
{
Array<int> all_field_idxs(GetNFields()), all_tag_idxs(GetNTags());
for (int f = 0; f < GetNFields(); f++)
{
all_field_idxs[f] = f;
}
for (int t = 0; t < GetNTags(); t++)
{
all_tag_idxs[t] = t;
}
PrintCSV(fname, all_field_idxs, all_tag_idxs, precision);
}
void ParticleSet::PrintCSV(const char *fname, const Array<int> &field_idxs,
const Array<int> &tag_idxs, int precision)
{
std::stringstream ss_header;
// Configure header:
ss_header << "id";
#ifdef MFEM_USE_MPI
ss_header << ",rank";
#endif // MFEM_USE_MPI
std::array<char, 3> ax = {'X', 'Y', 'Z'};
for (int c = 0; c < coords.GetVDim(); c++)
{
ss_header << "," << ax[c];
}
for (int f = 0; f < field_idxs.Size(); f++)
{
ParticleVector &pv = *fields[field_idxs[f]];
for (int c = 0; c < pv.GetVDim(); c++)
{
ss_header << "," << field_names[field_idxs[f]] <<
(pv.GetVDim() > 1 ? "_" + std::to_string(c) : "");
}
}
for (int t = 0; t < tag_idxs.Size(); t++)
{
ss_header << "," << tag_names[tag_idxs[t]];
}
ss_header << "\n";
// Configure data
std::stringstream ss_data;
ss_data.precision(precision);
#ifdef MFEM_USE_MPI
int rank = GetRank(comm);
#endif // MFEM_USE_MPI
for (int i = 0; i < GetNParticles(); i++)
{
ss_data << ids[i];
#ifdef MFEM_USE_MPI
ss_data << "," << rank;
#endif // MFEM_USE_MPI
for (int c = 0; c < coords.GetVDim(); c++)
{
ss_data << "," << coords(i, c);
}
for (int f = 0; f < field_idxs.Size(); f++)
{
ParticleVector &pv = *fields[field_idxs[f]];
for (int c = 0; c < pv.GetVDim(); c++)
{
ss_data << "," << pv(i, c);
}
}
for (int t = 0; t < tag_idxs.Size(); t++)
{
ss_data << "," << (*tags[tag_idxs[t]])[i];
}
ss_data << "\n";
}
// Write
WriteToFile(fname, ss_header, ss_data);
}
ParticleSet::~ParticleSet()
{
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
if (gsl_comm)
{
if (!Mpi::IsFinalized()) // currently segfaults inside gslib otherwise
{
crystal_free(cr);
comm_free(gsl_comm);
delete gsl_comm;
delete cr;
}
}
#endif // MFEM_USE_MPI && MFEM_USE_GSLIB
}
} // namespace mfem
+685
View File
@@ -0,0 +1,685 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_PARTICLESET
#define MFEM_PARTICLESET
#include "../config/config.hpp"
#include "../linalg/linalg.hpp"
#include "gslib.hpp"
#include "kernel_dispatch.hpp"
namespace mfem
{
/** @brief Container for data associated with a single particle.
*
* @note This class mainly serves as a convenience interface to individual
* particle data from ParticleSet. We recommend seeing ParticleSet first.
*
* @details As described in ParticleSet documentation, each particle has a
* position (\ref coords), arbitrary number of scalar or vector \ref real_t
* data (\ref fields), and arbitrary number of integers (\ref tags)
* associated with it.
*
* \ref fields can thus hold data such as mass, momentum, and velocity, while
* \ref tags can hold integer data such as particle type, color, etc.
*
* Each particle also has a unique global ID, but that is managed by the
* ParticleSet class and not stored in this Particle class. Simiarly, the names
* of the fields and tags, typically useful for output purposes, are managed by
* the ParticleSet class.
*
*
* For clarity, we will use the particles below to illustrate the data layout
* for \ref coords, \ref fields, and \ref tags
*
* @anchor sample_particle_data
* @code
* Particle_0: coords = (x0, y0),
* fields = {'mass'=m0, 'vel' = (vx0, vy0)},
* tags = {'type'=t0, 'color'=color0}
* Particle_1: coords = (x1, y1),
* fields = {'mass'=m1, 'vel' = (vx1, vy1)},
* tags = {'type'=t1, 'color'=color1}
* Particle_2: coords = (x2, y2),
* fields = {'mass'=m2, 'vel' = (vx2, vy2)},
* tags = {'type'=t2, 'color'=color2}
* @endcode
*
*/
class Particle
{
protected:
/** @brief Spatial coordinates
*
* @details For the \ref sample_particle_data, \ref coords would hold
* (x_i, y_i) for each particle i.
*/
Vector coords;
/** @brief A std::vector of Vector where each Vector holds data for a given
* field (e.g., mass, momentum or velocity) associated with the particle.
*
* @details For the \ref sample_particle_data, \ref fields would be
* fields[0]=(m_i), fields[1]=(vx_i,vy_i) for each particle i.
*/
std::vector<Vector> fields;
/** @brief A std::vector of Array<int> where each Array<int> holds data
* for a given tag.
*
* @details For the \ref sample_particle_data, \ref tags would be
* tags[0]=(type_i), tags[1]=(color_i) for each particle i. \n
*
* @note An Array of length 1 is used for EACH tag, strictly for
* its owning/non-owning semantics (see Array<T>::MakeRef).
*/
std::vector<Array<int>> tags;
public:
/** @brief Construct a Particle instance.
* @param[in] dim Spatial dimension (size of #coords).
* @param[in] field_vdims Vector dimensions of particle fields.
* @param[in] num_tags Number of integer tags.
*/
Particle(int dim, const Array<int> &field_vdims, int num_tags);
// Force default constructors and destructor
Particle(const Particle&) = default;
Particle& operator=(const Particle&) = default;
Particle(Particle&&) = default;
Particle& operator=(Particle&&) = default;
~Particle() = default;
/// Get the spatial dimension of this particle.
int GetDim() const { return coords.Size(); }
/// Get the number of fields associated with this particle.
int GetNFields() const { return fields.size(); }
/// Get the vector dimension of field \p f .
int GetFieldVDim(int f) const { return fields[f].Size(); }
/// Get the number of tags associated with this particle.
int GetNTags() const { return tags.size(); }
/// Get reference to particle coordinates Vector.
Vector& Coords() { return coords; }
/// Get const reference to particle coordinates Vector.
const Vector& Coords() const { return coords; }
/// Get reference to field \p f , component \p c value.
real_t& FieldValue(int f, int c=0)
{
MFEM_ASSERT(f >= 0 && static_cast<std::size_t>(f) < fields.size(),
"Invalid field index");
MFEM_ASSERT(c >= 0 && c < fields[f].Size(),
"Invalid component index");
return fields[f][c];
}
/// Get const reference to field \p f , component \p c value.
const real_t& FieldValue(int f, int c=0) const
{
MFEM_ASSERT(f >= 0 && static_cast<std::size_t>(f) < fields.size(),
"invalid field index");
MFEM_ASSERT(c >= 0 && c < fields[f].Size(),
"invalid component index");
return fields[f][c];
}
/// Get reference to field \p f Vector.
Vector& Field(int f)
{
MFEM_ASSERT(f >= 0 && static_cast<std::size_t>(f) < fields.size(),
"invalid field index");
return fields[f];
}
/// Get const reference to field \p f Vector.
const Vector& Field(int f) const
{
MFEM_ASSERT(f >= 0 && static_cast<std::size_t>(f) < fields.size(),
"invalid field index");
return fields[f];
}
/// Get reference to tag \p t .
int& Tag(int t)
{
MFEM_ASSERT(t >= 0 && static_cast<std::size_t>(t) < tags.size(),
"invalid tag index");
return tags[t][0];
}
/// Get const reference to tag \p t .
const int& Tag(int t) const
{
MFEM_ASSERT(t >= 0 && static_cast<std::size_t>(t) < tags.size(),
"invalid tag index");
return tags[t][0];
}
/// Set tag \p t to reference external data.
void SetTagRef(int t, int *tag_data);
/// Set field \p f to reference external data.
void SetFieldRef(int f, real_t *field_data);
/// Particle equality operator.
bool operator==(const Particle &rhs) const;
/// Particle inequality operator.
bool operator!=(const Particle &rhs) const { return !operator==(rhs); }
/// Print all particle data to \p os.
void Print(std::ostream &os=mfem::out) const;
};
/** @brief ParticleSet initializes and manages data associated with particles.
*
* @details Particles are inherently initialized to have a position and an ID,
* and optionally can have any number of Vector (of arbitrary vdim) and scalar
* integer data in the form of @b fields and @b tags respectively. All particle
* data are internally stored in a Struct-of-Arrays fashion, as elaborated on
* below.
*
* @par Coordinates:
* All particle coordinates are stored in a ParticleVector with vector
* dimension equal to the spatial dimension, ordered either byNODES or byVDIM.
* The ParticleVector \ref coords contains the coordinates of all particles.
*
* @par IDs:
* Each particle is assigned a unique global ID of type IDType. In parallel,
* IDs are initialized starting with @b rank and striding by @b size. The IDs
* of all particles owned by this rank are stored in \ref ids.
*
* @par Fields:
* Fields represent scalar or vector \ref real_t data to be associated with
* each particles, such as mass, momentum, or moment. For a given field, all
* particle data is stored in a single ParticleVector with a given
* vector dimension (1 for scalar data) and Ordering::Type (byNODES or
* byVDIM). The unique_ptrs to all the ParticleVectors are stored in the
* std::vector \ref fields.
*
* @par Tags:
* Tags represent integers associated with each particle. For a given tag,
* all particle data are stored in a single Array<int>. The unique_ptrs to all
* the Array<int> is stored in the std::vector \ref tags.
*
* @par Names:
* Each field and tag can optionally be given a name (string) to be used when
* printing particle data in CSV format using PrintCSV(). The names of all
* fields and tags are stored in the std::vectors \ref field_names and
* \ref tag_names, respectively.
*
* @note We assume that all particles in a ParticleSet have the same number
* of fields and tags.
*
* Following the example in the Particle class, we will use the
* particles below to illustrate the data layout for \ref coords, \ref ids,
* \ref fields, \ref tags, \ref field_names, and \ref tag_names.
* In each case, the name of the field and tag is enclosed in '...' for
* clarity. Additionally, we assume for this example that the particle
* coordinates and the 'vel' field are ordered byVDIM in their respective
* ParticleVector.
* @anchor sample_particleset_data
* @code
* Particle_0: id = id0, coords = (x0, y0),
* fields = {'mass'=m0, 'vel' = (vx0, vy0)},
* tags = {'type'=t0, 'color'=c0}
* Particle_1: id = id1, coords = (x1, y1),
* fields = {'mass'=m1, 'vel' = (vx1, vy1)},
* tags = {'type'=t1, 'color'=c1}
* Particle_2: id = id2, coords = (x2, y2),
* fields = {'mass'=m2, 'vel' = (vx2, vy2)},
* tags = {'type'=t2, 'color'=c2}
* @endcode
*/
class ParticleSet
{
public:
using IDType = unsigned long long;
private:
/// Constructs an Array of size N filled with Ordering::Type o.
static Array<Ordering::Type> GetOrderingArray(Ordering::Type o, int N);
/// Returns default field name for field index i. "Field_{i}"
static std::string GetDefaultFieldName(int i);
/// Returns default tag name for tag index i. "Tag_{i}"
static std::string GetDefaultTagName(int i);
/// Constructs an Array of size N filled with nullptr.
static Array<const char*> GetEmptyNameArray(int N);
#ifdef MFEM_USE_MPI
static int GetRank(MPI_Comm comm_);
static int GetSize(MPI_Comm comm_);
#endif // MFEM_USE_MPI
protected:
/// Stride for IDs (used internally when new particles are added).
/** In parallel, this defaults to the number of MPI ranks. */
const int id_stride;
/// Current globally unique ID to be assigned to the next particle added.
/** In parallel, this starts locally as the rank and increments with
* id_stride, ensuring a global unique identifier whenever a particle is
* added.
*/
IDType id_counter;
/** @brief Global unique IDs of particles owned by this rank.
*
* @details For the \ref sample_particleset_data, \ref ids would be
* ids[0]=id0, ids[1]=id1, ids[2]=id2.
*/
Array<IDType> ids;
/** @brief Spatial coordinates of particles owned by this rank.
*
* @details For the \ref sample_particleset_data, \ref coords would be
* coords=(x0,y0,x1,y1,x2,y2) assuming coords ordering is byVDIM.
*/
ParticleVector coords;
/** @brief All particle fields for particles owned by this rank.
*
* @details For the \ref sample_particleset_data, \ref fields would be
* *fields[0]=(m0,m1,m2), *fields[1]=(vx0,vy0,vx1,vy1,vx2,vy2)
* assuming fields[1] ordering is byVDIM.
*/
std::vector<std::unique_ptr<ParticleVector>> fields;
/** @brief All particle tags for particles owned by this rank.
*
* @details For the \ref sample_particleset_data, \ref tags would be
* *tags[0]=(t0,t1,t2), *tags[1]=(c0,c1,c2).
*/
std::vector<std::unique_ptr<Array<int>>> tags;
/** @brief Field names, to be written when PrintCSV() is called.
*
* @details For the \ref sample_particleset_data, \ref field_names would be
* field_names[0]='mass', field_names[1]='vel'.
*/
std::vector<std::string> field_names;
/** @brief Tag names, to be written when PrintCSV() is called.
*
* @details For the \ref sample_particleset_data, \ref tag_names would be
* tag_names[0]='type', tag_names[1]='color'.
*/
std::vector<std::string> tag_names;
/** @brief Add particles with global identifiers \p new_ids and
* optionally get the local indices of new particles in \p new_indices .
*
* @details Note the data of new particles is uninitialized and must be
* set.
*/
void AddParticles(const Array<IDType> &new_ids,
Array<int> *new_indices=nullptr);
#ifdef MFEM_USE_MPI
MPI_Comm comm;
#endif // MFEM_USE_MPI
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
struct gslib::crystal *cr = nullptr; // gslib's internal data
struct gslib::comm *gsl_comm = nullptr; // gslib's internal data
/// \cond DO_NOT_DOCUMENT
template<std::size_t NBytes>
static void TransferParticlesImpl(ParticleSet &pset,
const Array<int> &send_idxs,
const Array<unsigned int> &send_ranks);
using TransferParticlesType = void (*)(ParticleSet &pset,
const Array<int> &send_idxs,
const Array<unsigned int> &send_ranks);
// Specialization parameter: NBytes
MFEM_REGISTER_KERNELS(TransferParticles, TransferParticlesType, (size_t));
friend TransferParticles;
struct Kernels
{
Kernels();
};
/// \endcond
#endif // MFEM_USE_MPI && MFEM_USE_GSLIB
/** @brief Update global ID of a particle.
*
* @details This method updates the global ID of the particle at given
* local index after Redistribute().
*
* @note This method must be used very carefully as it updates global
* ID of a particle.
*/
void UpdateID(int local_idx, IDType new_global_id)
{ ids[local_idx] = new_global_id; }
/** @brief Create a Particle object with the same spatial dimension,
* number of fields and field vdims, and number of tags as this ParticleSet.
*/
Particle CreateParticle() const;
/** @brief Write string in \p ss_header , followed by \p ss_data , to a
* single file; compatible in parallel.
*/
void WriteToFile(const char *fname, const std::stringstream &ss_header,
const std::stringstream &ss_data);
/** @brief Check if a particle could belong in this ParticleSet by
* comparing field and tag dimension.
*/
bool IsValidParticle(const Particle &p) const;
/** @brief Hidden main constructor of ParticleSet
*
* @param[in] id_stride_ ID stride.
* @param[in] id_counter_ Starting ID counter.
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering Ordering of coordinates
* @param[in] field_vdims Array of field vector dimensions
* @param[in] field_orderings Array of field ordering types.
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
*/
ParticleSet(int id_stride_, IDType id_counter_, int num_particles, int dim,
Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
public:
/** @brief Construct a serial ParticleSet.
*
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering Ordering of coordinates.
*/
ParticleSet(int num_particles, int dim,
Ordering::Type coords_ordering=Ordering::byVDIM);
/** @brief Construct a serial ParticleSet with specified fields and tags at
* construction.
*
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] field_vdims Array of field vector dimensions.
* @param[in] num_tags Number of tags to register.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
*/
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
int num_tags, Ordering::Type all_ordering=Ordering::byVDIM);
/** @brief Construct a serial ParticleSet with specified fields and tags at
* construction, with names.
*
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] field_vdims Array of field vector dimensions.
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
*/
ParticleSet(int num_particles, int dim, const Array<int> &field_vdims,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_,
Ordering::Type all_ordering=Ordering::byVDIM);
/** @brief Comprehensive serial constructor of ParticleSet.
*
* @param[in] num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering Ordering of coordinates.
* @param[in] field_vdims Array of field vector dimensions.
* @param[in] field_orderings Array of field ordering types.
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
*/
ParticleSet(int num_particles, int dim, Ordering::Type coords_ordering,
const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
#ifdef MFEM_USE_MPI
/** @brief Construct a parallel ParticleSet.
*
* @param[in] comm_ MPI communicator.
* @param[in] rank_num_particles Number of particles to initialize.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering (Optional) Ordering of coordinates.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering=Ordering::byVDIM);
/** @brief Construct a parallel ParticleSet with specified fields and tags
* at construction.
*
* @param[in] comm_ MPI communicator.
* @param[in] rank_num_particles # of particles to initialize on this rank.
* @param[in] dim Particle spatial dimension.
* @param[in] field_vdims Array of field vector dimensions.
* @param[in] num_tags Number of tags to register.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims, int num_tags,
Ordering::Type all_ordering=Ordering::byVDIM);
/** @brief Construct a parallel ParticleSet with specified fields and tags
* at construction, with names (for PrintCSV()).
*
* @param[in] comm_ MPI communicator.
* @param[in] rank_num_particles # of particles to initialize on this rank.
* @param[in] dim Particle spatial dimension.
* @param[in] field_vdims Array of field vector dimension.
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
* @param[in] all_ordering (Optional) Ordering of coordinates and
* field ParticleVector.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
const Array<int> &field_vdims,
const Array<const char*> &field_names_,
int num_tags, const Array<const char*> &tag_names_,
Ordering::Type all_ordering=Ordering::byVDIM);
/** @brief Comprehensive parallel constructor of ParticleSet.
*
* @param[in] comm_ MPI communicator.
* @param[in] rank_num_particles # of particles to initialize on this rank.
* @param[in] dim Particle spatial dimension.
* @param[in] coords_ordering Ordering of coordinates.
* @param[in] field_vdims Array of field vector dimensions.
* @param[in] field_orderings Array of field ordering types.
* @param[in] field_names_ Array of field names.
* @param[in] num_tags Number of tags to register.
* @param[in] tag_names_ Array of tag names.
*/
ParticleSet(MPI_Comm comm_, int rank_num_particles, int dim,
Ordering::Type coords_ordering, const Array<int> &field_vdims,
const Array<Ordering::Type> &field_orderings,
const Array<const char*> &field_names_, int num_tags,
const Array<const char*> &tag_names_);
/// Get the MPI communicator for this ParticleSet.
MPI_Comm GetComm() const { return comm; };
#endif // MFEM_USE_MPI
/// Get the global number of active particles across all ranks.
IDType GetGlobalNParticles() const;
/// Get the spatial dimension.
int GetDim() const { return coords.GetVDim(); }
/// Get the global IDs of the active particles owned by this ParticleSet.
const Array<IDType>& GetIDs() const { return ids; }
/** @brief Add a field to the ParticleSet.
*
* @param[in] vdim Vector dimension of the field.
* @param[in] field_ordering (Optional) Ordering::Type of the field.
* @param[in] field_name (Optional) Name of the field.
*
* @return Index of the newly-added field.
*/
int AddField(int vdim, Ordering::Type field_ordering=Ordering::byVDIM,
const char* field_name=nullptr);
/** @brief Add a field to the ParticleSet.
*
* @details Same as AddField() but with different parameter order
* for convenience
*/
int AddNamedField(int vdim, const char* field_name,
Ordering::Type field_ordering=Ordering::byVDIM)
{
return AddField(vdim, field_ordering, field_name);
}
/** @brief Add a tag to the ParticleSet.
*
* @param[in] tag_name (Optional) Name of the tag.
*
* @return Index of the newly-added tag.
*/
int AddTag(const char* tag_name=nullptr);
/// Reserve memory for \p res particles.
/** Can help to avoid re-allocation for adding + removing particles. */
void Reserve(int res);
/// Get the number of active particles currently held by this ParticleSet.
int GetNParticles() const { return ids.Size(); }
/// Get the number of fields registered to particles.
int GetNFields() const { return fields.size(); }
/// Get an Array<int> of the field vector-dimensions registered to particles.
const Array<int> GetFieldVDims() const;
/// Get Field vector-dimension
int FieldVDim(int f) const { return fields[f]->GetVDim(); }
/// Get the number of tags registered to particles.
int GetNTags() const { return tags.size(); }
/// Add a particle using Particle .
void AddParticle(const Particle &p);
/** @brief Add \p num_particles particles, and optionally get the local
* indices of new particles in \p new_indices .
*
* @details The data of new particles is uninitialized and must be
* set.
*/
void AddParticles(int num_particles, Array<int> *new_indices=nullptr);
/// Remove particle data specified by \p list of particle indices.
void RemoveParticles(const Array<int> &list);
/// Get a reference to the coordinates ParticleVector.
ParticleVector& Coords() { return coords; }
/// Get a const reference to the coordinates ParticleVector.
const ParticleVector& Coords() const { return coords; }
/// Get a reference to field \p f 's ParticleVector.
ParticleVector& Field(int f) { return *fields[f]; }
/// Get a const reference to field \p f 's ParticleVector.
const ParticleVector& Field(int f) const { return *fields[f]; }
/// Get a reference to tag \p t 's Array<int>.
Array<int>& Tag(int t) { return *tags[t]; }
/// Get a const reference to tag \p t 's Array<int>.
const Array<int>& Tag(int t) const { return *tags[t]; }
/** @brief Get new Particle object with copy of data associated with
particle \p i . */
Particle GetParticle(int i) const;
/** @brief Get Particle object whose members reference the actual data
* associated with particle \p i in this ParticleSet.
*
* @see IsParticleRefValid for when this method can be used.
*
* @warning If particles are added, removed, or redistributed after
* invoking this, the returned Particle member references may be
* invalidated.
*/
Particle GetParticleRef(int i);
/** @brief Determine if GetParticleRef is valid.
*
* If coordinates and all fields are ordered byVDIM, then returns true.
* Otherwise, false.
*/
bool IsParticleRefValid() const;
/// Set data for particle at index \p i with data from provided particle \p p
void SetParticle(int i, const Particle &p);
/** @brief Print all particle data to a comma-delimited CSV file.
*
* The first row contains the header. We include the particle ID,
* owning rank (in parallel), coordinates, followed by all fields and
* tags.
*
* The output can be visualized in Paraview by loading the csv files, and
* applying the "Table To Points" filter.
*/
void PrintCSV(const char *fname, int precision=16);
/** @brief Print only particle field and tags given by \p field_idxs and
\p tag_idxs respectively to a CSV file. */
void PrintCSV(const char *fname, const Array<int> &field_idxs,
const Array<int> &tag_idxs, int precision=16);
#if defined(MFEM_USE_MPI) && defined(MFEM_USE_GSLIB)
/** @brief Redistribute particle data to \p rank_list
@param[in] rank_list Array of size GetNParticles() denoting ultimate
destination of particle data. Index = this rank
means no data is moved.
*/
void Redistribute(const Array<unsigned int> &rank_list);
#endif // MFEM_USE_MPI && MFEM_USE_GSLIB
/// Destructor
~ParticleSet();
ParticleSet(const ParticleSet&) = delete;
ParticleSet& operator=(const ParticleSet&) = delete;
};
} // namespace mfem
#endif // MFEM_PARTICLESET
+9 -4
View File
@@ -5271,7 +5271,8 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
MFEM_ASSERT(R->Finalized(), "");
const int tdofs = R->Height();
MFEM_ASSERT(tdofs == R->HostReadI()[tdofs], "");
ltdof_ldof = Array<int>(const_cast<int*>(R->HostReadJ()), tdofs);
ltdof_ldof.SetSize(tdofs);
ltdof_ldof.CopyFrom(R->HostReadJ());
{
Table nbr_ltdof;
gc.GetNeighborLTDofTable(nbr_ltdof);
@@ -5294,9 +5295,13 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
}
Table unique_shr;
Transpose(shared_ltdof, unique_shr, unique_ltdof.Size());
unq_ltdof = Array<int>(unique_ltdof, unique_ltdof.Size());
unq_shr_i = Array<int>(unique_shr.GetI(), unique_shr.Size()+1);
unq_shr_j = Array<int>(unique_shr.GetJ(), unique_shr.Size_of_connections());
unq_ltdof = unique_ltdof;
// Steal I and J arrays from the unique_shr table.
unq_shr_i.GetMemory() = unique_shr.GetIMemory();
unq_shr_i.SetSize(unique_shr.Size()+1);
unq_shr_j.GetMemory() = unique_shr.GetJMemory();
unq_shr_j.SetSize(unique_shr.Size_of_connections());
unique_shr.LoseData();
}
nbr_ltdof.GetJMemory().Delete();
nbr_ltdof.LoseData();
+1 -1
View File
@@ -1407,7 +1407,7 @@ real_t L2ZZErrorEstimator(BilinearFormIntegrator &flux_integrator,
}
PLBound ParGridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
const int ref_factor, const int vdim) const
{
PLBound plb = GridFunction::GetBounds(lower, upper, ref_factor, vdim);
int siz = vdim > 0 ? 1 : fes->GetVDim();
+1 -1
View File
@@ -587,7 +587,7 @@ public:
/// PLBound object used to compute the bounds. Note: if vdim < 1, we compute
/// the bounds for each vector dimension.
PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) override;
const int ref_factor=1, const int vdim=-1) const override;
/** Save the local portion of the ParGridFunction. This differs from the
serial GridFunction::Save in that it takes into account the signs of
+40 -7
View File
@@ -56,6 +56,20 @@ void QuadratureFunction::Save(std::ostream &os) const
os.flush();
}
void QuadratureFunction::ProjectGridFunctionFallback(const GridFunction &gf)
{
if (gf.VectorDim() == 1)
{
GridFunctionCoefficient coeff(&gf);
coeff.Coefficient::Project(*this);
}
else
{
VectorGridFunctionCoefficient coeff(&gf);
coeff.VectorCoefficient::Project(*this);
}
}
void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
{
SetVDim(gf.VectorDim());
@@ -68,14 +82,23 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
ElementDofOrdering::LEXICOGRAPHIC :
ElementDofOrdering::NATIVE;
// Use quadrature interpolator to go from E-vector to Q-vector
const QuadratureInterpolator *qi =
gf_fes.GetQuadratureInterpolator(*qs_elem);
// If quadrature interpolator doesn't support this space, then fallback
// on slower (non-device) version, and return early.
if (!qi)
{
ProjectGridFunctionFallback(gf);
return;
}
// Use element restriction to go from L-vector to E-vector
const Operator *R = gf_fes.GetElementRestriction(ordering);
Vector e_vec(R->Height());
R->Mult(gf, e_vec);
// Use quadrature interpolator to go from E-vector to Q-vector
const QuadratureInterpolator *qi =
gf_fes.GetQuadratureInterpolator(*qs_elem);
qi->SetOutputLayout(QVectorLayout::byVDIM);
qi->DisableTensorProducts(!use_tensor_products);
qi->PhysValues(e_vec, *this);
@@ -83,12 +106,25 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
else if (auto *qs_face = dynamic_cast<FaceQuadratureSpace*>(qspace))
{
const FiniteElementSpace &gf_fes = *gf.FESpace();
const FaceType face_type = qs_face->GetFaceType();
const bool use_tensor_products = UsesTensorBasis(gf_fes);
const ElementDofOrdering ordering = use_tensor_products ?
ElementDofOrdering::LEXICOGRAPHIC :
ElementDofOrdering::NATIVE;
const FaceType face_type = qs_face->GetFaceType();
// Use quadrature interpolator to go from E-vector to Q-vector
const FaceQuadratureInterpolator *qi =
gf_fes.GetFaceQuadratureInterpolator(qspace->GetIntRule(0), face_type);
// If quadrature interpolator doesn't support this space, then fallback
// on slower (non-device) version, and return early. Also, currently,
// ElementDofOrdering::NATIVE in FaceRestriction, so fall back in that
// case too.
if (qi == nullptr || ordering == ElementDofOrdering::NATIVE)
{
ProjectGridFunctionFallback(gf);
return;
}
// Use element restriction to go from L-vector to E-vector
const Operator *R = gf_fes.GetFaceRestriction(
@@ -96,9 +132,6 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
Vector e_vec(R->Height());
R->Mult(gf, e_vec);
// Use quadrature interpolator to go from E-vector to Q-vector
const FaceQuadratureInterpolator *qi =
gf_fes.GetFaceQuadratureInterpolator(qspace->GetIntRule(0), face_type);
qi->SetOutputLayout(QVectorLayout::byVDIM);
qi->DisableTensorProducts(!use_tensor_products);
qi->Values(e_vec, *this);
+2
View File
@@ -27,6 +27,8 @@ protected:
bool own_qspace; ///< Does this own the associated QuadratureSpaceBase?
int vdim; ///< Vector dimension.
void ProjectGridFunctionFallback(const GridFunction &gf);
public:
/// Default constructor, results in an empty vector.
QuadratureFunction() : qspace(nullptr), own_qspace(false), vdim(0)
+12 -6
View File
@@ -69,9 +69,7 @@ QuadratureInterpolator::QuadratureInterpolator(const FiniteElementSpace &fes,
d_buffer.UseDevice(true);
if (fespace->GetNE() == 0) { return; }
const FiniteElement *fe = fespace->GetTypicalFE();
MFEM_VERIFY(fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV,
MFEM_VERIFY(SupportsFESpace(fes),
"Only elements with MapType VALUE and H_DIV are supported!");
}
@@ -86,12 +84,20 @@ QuadratureInterpolator::QuadratureInterpolator(const FiniteElementSpace &fes,
{
d_buffer.UseDevice(true);
if (fespace->GetNE() == 0) { return; }
const FiniteElement *fe = fespace->GetTypicalFE();
MFEM_VERIFY(fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV,
MFEM_VERIFY(SupportsFESpace(fes),
"Only elements with MapType VALUE and H_DIV are supported!");
}
bool QuadratureInterpolator::SupportsFESpace(const FiniteElementSpace &fespace)
{
const FiniteElement *fe = fespace.GetTypicalFE();
const Mesh &mesh = *fespace.GetMesh();
return (fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV)
&& (!fespace.IsVariableOrder())
&& (!mesh.IsMixedMesh());
}
namespace internal
{
+3
View File
@@ -155,6 +155,9 @@ public:
void MultTranspose(unsigned eval_flags, const Vector &q_val,
const Vector &q_der, Vector &e_vec) const;
/// @brief Returns true if the given finite element space is supported by
/// QuadratureInterpolator.
static bool SupportsFESpace(const FiniteElementSpace &fespace);
using TensorEvalKernelType = void(*)(const int, const real_t *, const real_t *,
real_t *, const int, const int, const int);
+11 -11
View File
@@ -77,17 +77,17 @@ FaceQuadratureInterpolator::FaceQuadratureInterpolator(
if (fespace->GetNE() == 0) { return; }
GetSigns(*fespace, type, signs);
const FiniteElement *fe = fespace->GetTypicalFE();
const ScalarFiniteElement *sfe =
dynamic_cast<const ScalarFiniteElement*>(fe);
const TensorBasisElement *tfe =
dynamic_cast<const TensorBasisElement*>(fe);
MFEM_VERIFY(sfe != NULL, "Only scalar finite elements are supported");
MFEM_VERIFY(tfe != NULL &&
(tfe->GetBasisType()==BasisType::GaussLobatto ||
tfe->GetBasisType()==BasisType::Positive),
"Only Gauss-Lobatto and Bernstein basis are supported in "
"FaceQuadratureInterpolator.");
MFEM_VERIFY(SupportsFESpace(fes), "Unsupported finite element space");
}
bool FaceQuadratureInterpolator::SupportsFESpace(const FiniteElementSpace &fes)
{
const FiniteElement *fe = fes.GetTypicalFE();
const auto *sfe = dynamic_cast<const ScalarFiniteElement*>(fe);
const auto *tfe = dynamic_cast<const TensorBasisElement*>(fe);
return sfe != nullptr && tfe != nullptr && (
tfe->GetBasisType() == BasisType::GaussLobatto ||
tfe->GetBasisType() == BasisType::Positive);
}
template<const int T_VDIM, const int T_ND1D, const int T_NQ1D>
+4
View File
@@ -64,6 +64,10 @@ public:
FaceQuadratureInterpolator(const FiniteElementSpace &fes,
const IntegrationRule &ir, FaceType type);
/// @brief Returns true if the given finite element space is supported by
/// FaceQuadratureInterpolator.
static bool SupportsFESpace(const FiniteElementSpace &fes);
/** @brief Disable the use of tensor product evaluations, for tensor-product
elements, e.g. quads and hexes. */
/** Currently, tensor product evaluations are not implemented and this method
+5
View File
@@ -1611,6 +1611,11 @@ protected:
const TargetType target_type;
bool uses_phys_coords; // see UsesPhysicalCoordinates()
/// Cached copy of GeomToPerfGeomJac used on device.
mutable DenseMatrix current_W;
/// Geometry type of current W matrix (used for cache invalidation).
mutable Geometry::Type current_W_type = Geometry::INVALID;
#ifdef MFEM_USE_MPI
MPI_Comm comm;
#endif
+11 -11
View File
@@ -19,14 +19,15 @@ namespace mfem
struct TMOP_PA_Metric_001 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1();
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4];
@@ -34,14 +35,13 @@ struct TMOP_PA_Metric_001 : TMOP_PA_Metric_2D
kernels::Set(2, 2, 1.0, ie.Get_dI1(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// weight * ddI1
+8 -7
View File
@@ -19,14 +19,15 @@ namespace mfem
struct TMOP_PA_Metric_002 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return 0.5 * ie.Get_I1b() - 1.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1b[4], dI2b[4];
@@ -34,10 +35,10 @@ struct TMOP_PA_Metric_002 : TMOP_PA_Metric_2D
kernels::Set(2, 2, 1. / 2., ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx, const int qy, const int e, const real_t weight,
const real_t (&Jpt)[4], const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e, const real_t weight,
const real_t (&Jpt)[4], const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// 0.5 * weight * dI1b
+10 -11
View File
@@ -19,14 +19,15 @@ namespace mfem
struct TMOP_PA_Metric_007 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1() * (1.0 + 1.0 / ie.Get_I2()) - 4.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4], dI2[4], dI2b[4];
@@ -37,14 +38,12 @@ struct TMOP_PA_Metric_007 : TMOP_PA_Metric_2D
ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
+10 -11
View File
@@ -19,15 +19,16 @@ namespace mfem
struct TMOP_PA_Metric_056 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b + 1.0 / I2b) - 1.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
// 0.5*(1 - 1/I2b^2)*dI2b
@@ -37,14 +38,12 @@ struct TMOP_PA_Metric_056 : TMOP_PA_Metric_2D
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// (0.5 - 0.5/I2b^2)*ddI2b + (1/I2b^3)*(dI2b x dI2b)
+10 -11
View File
@@ -19,15 +19,16 @@ namespace mfem
struct TMOP_PA_Metric_077 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4];
@@ -36,14 +37,12 @@ struct TMOP_PA_Metric_077 : TMOP_PA_Metric_2D
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4], ddI2[4];
+10 -11
View File
@@ -19,7 +19,8 @@ namespace mfem
struct TMOP_PA_Metric_080 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *w) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
@@ -28,8 +29,8 @@ struct TMOP_PA_Metric_080 : TMOP_PA_Metric_2D
return w[0] * eval_w_02 + w[1] * eval_w_77;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
// w0 P_2 + w1 P_77
@@ -41,14 +42,12 @@ struct TMOP_PA_Metric_080 : TMOP_PA_Metric_2D
kernels::Add(2, 2, w[1] * 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
// w0 H_2 + w1 H_77
real_t ddI1[4], ddI1b[4], dI2[4], dI2b[4], ddI2[4];
+10 -11
View File
@@ -19,7 +19,8 @@ namespace mfem
struct TMOP_PA_Metric_094 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *w) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
@@ -28,8 +29,8 @@ struct TMOP_PA_Metric_094 : TMOP_PA_Metric_2D
return w[0] * eval_w_02 + w[1] * eval_w_56;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
// w0 P_2 + w1 P_56
real_t dI1b[4], dI2b[4];
@@ -40,14 +41,12 @@ struct TMOP_PA_Metric_094 : TMOP_PA_Metric_2D
P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
// w0 H_2 + w1 H_56
real_t ddI1[4], ddI1b[4], dI2b[4], ddI2b[4];
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_302 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -29,8 +29,8 @@ struct TMOP_PA_Metric_302 : TMOP_PA_Metric_3D
return ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9],
const real_t *w, real_t (&P)[9]) const final
{
MFEM_CONTRACT_VAR(w);
// (I1b/9)*dI2b + (I2b/9)*dI1b
@@ -43,17 +43,14 @@ struct TMOP_PA_Metric_302 : TMOP_PA_Metric_3D
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
MFEM_CONTRACT_VAR(Jrt);
MFEM_CONTRACT_VAR(Jpr);
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_303 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -29,8 +29,8 @@ struct TMOP_PA_Metric_303 : TMOP_PA_Metric_3D
return ie.Get_I1b() / 3. - 1.;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9],
const real_t *w, real_t (&P)[9]) const final
{
MFEM_CONTRACT_VAR(w);
// dI1b/3
@@ -41,17 +41,14 @@ struct TMOP_PA_Metric_303 : TMOP_PA_Metric_3D
kernels::Set(3, 3, 1. / 3., ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t B[9];
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_315 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -30,8 +30,8 @@ struct TMOP_PA_Metric_315 : TMOP_PA_Metric_3D
return a * a;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9],
const real_t *w, real_t (&P)[9]) const final
{
MFEM_CONTRACT_VAR(w);
// 2*(I3b - 1)*dI3b
@@ -42,17 +42,14 @@ struct TMOP_PA_Metric_315 : TMOP_PA_Metric_3D
kernels::Set(3, 3, 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t *dI3b = Jrt, *ddI3b = Jpr;
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_318 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -32,8 +32,8 @@ struct TMOP_PA_Metric_318 : TMOP_PA_Metric_3D
// P_318 = (I3b - 1/I3b^3)*dI3b.
// Uses the I3b form, as dI3 and ddI3 were not implemented at the time
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9],
const real_t *w, real_t (&P)[9]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI3b[9];
@@ -45,17 +45,14 @@ struct TMOP_PA_Metric_318 : TMOP_PA_Metric_3D
P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t *dI3b = Jrt, *ddI3b = Jpr;
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_321 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -29,8 +29,8 @@ struct TMOP_PA_Metric_321 : TMOP_PA_Metric_3D
return ie.Get_I1() + ie.Get_I2() / ie.Get_I3() - 6.0;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9],
const real_t *w, real_t (&P)[9]) const final
{
MFEM_CONTRACT_VAR(w);
// dI1 + (1/I3)*dI2 - (2*I2/I3b^3)*dI3b
@@ -46,17 +46,14 @@ struct TMOP_PA_Metric_321 : TMOP_PA_Metric_3D
kernels::Add(3, 3, ie.Get_dI1(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t B[9];
+10 -13
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_332 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -32,7 +32,7 @@ struct TMOP_PA_Metric_332 : TMOP_PA_Metric_3D
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) const final
{
// w0 P_302 + w1 P_315
real_t B[9];
@@ -47,17 +47,14 @@ struct TMOP_PA_Metric_332 : TMOP_PA_Metric_3D
kernels::Add(3, 3, w[1] * 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
real_t B[9];
real_t dI1b[9], /*ddI1[9],*/ ddI1b[9];
+11 -14
View File
@@ -20,7 +20,7 @@ namespace mfem
struct TMOP_PA_Metric_338 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
const real_t *w) const final
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
@@ -31,8 +31,8 @@ struct TMOP_PA_Metric_338 : TMOP_PA_Metric_3D
return w[0] * eval_w_302 + w[1] * eval_w_318;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[9], const real_t *w,
real_t (&P)[9]) const final
{
// w0 P_302 + w1 P_318
real_t B[9];
@@ -48,17 +48,14 @@ struct TMOP_PA_Metric_338 : TMOP_PA_Metric_3D
ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy, const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const final
{
real_t B[9];
real_t dI1b[9], ddI1b[9];
+29 -9
View File
@@ -25,17 +25,27 @@ struct TMOP_PA_Metric_2D
using Args = kernels::InvariantsEvaluator2D::Buffers;
virtual MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) = 0;
const real_t *w) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_2D::EvalW is not implemented");
return -0.0_r;
}
virtual MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[DIM * DIM],
const real_t *w,
real_t (&P)[DIM * DIM]) = 0;
real_t (&P)[DIM * DIM]) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_2D::EvalP is not implemented");
}
virtual MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e, const real_t weight,
const real_t (&Jpt)[DIM * DIM],
const real_t *w,
const DeviceTensor<7> &H) = 0;
const DeviceTensor<7> &H) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_2D::AssembleH is not implemented");
}
};
/// Abstract base class for the 3D metric TMOP PA kernels.
@@ -45,18 +55,28 @@ struct TMOP_PA_Metric_3D
using Args = kernels::InvariantsEvaluator3D::Buffers;
virtual MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) = 0;
const real_t *w) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_3D::EvalW is not implemented");
return -0.0_r;
}
virtual MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[DIM * DIM],
const real_t *w,
real_t (&P)[DIM * DIM]) = 0;
real_t (&P)[DIM * DIM]) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_3D::EvalP is not implemented");
}
virtual MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int qz, const int e, const real_t weight,
real_t *Jrt, real_t *Jpr,
const real_t (&Jpt)[DIM * DIM],
const real_t *w,
const DeviceTensor<5 + DIM> &H) const = 0;
const DeviceTensor<5 + DIM> &H) const
{
MFEM_ABORT_KERNEL("TMOP_PA_Metric_3D::AssembleH is not implemented");
}
};
namespace tmop
@@ -107,7 +127,7 @@ int KernelSpecializationsMDQ()
#define MFEM_TMOP_MDQ_SPECIALIZE(Name) \
namespace \
{ \
static bool k##Name{ (tmop::KernelSpecializationsMDQ<Name>(), true) }; \
[[maybe_unused]] static bool k##Name{ (tmop::KernelSpecializationsMDQ<Name>(), true) }; \
}
// Register TMOP kernels using two templated parameters (D1D, Q1D).
@@ -146,7 +166,7 @@ int KernelSpecializations()
#define MFEM_TMOP_ADD_SPECIALIZED_KERNELS(Name) \
namespace \
{ \
static bool k##Name{ (tmop::KernelSpecializations<Name>(), true) }; \
[[maybe_unused]] static bool k##Name{ (tmop::KernelSpecializations<Name>(), true) }; \
}
// Register TMOP kernels using a single templated parameter (Q1D).
@@ -172,7 +192,7 @@ int KernelSpecializations1()
#define MFEM_TMOP_ADD_SPECIALIZED_KERNELS_1(Name) \
namespace \
{ \
static bool k##Name{ (tmop::KernelSpecializations1<Name>(), true) }; \
[[maybe_unused]] static bool k##Name{ (tmop::KernelSpecializations1<Name>(), true) }; \
}
// Register TMOP kernels for a templated metric id.
+8 -3
View File
@@ -215,7 +215,12 @@ void DiscreteAdaptTC::ComputeAllElementTargets(const FiniteElementSpace &pa_fes,
"mixed meshes are not supported");
MFEM_VERIFY(!fes->IsVariableOrder(), "variable orders are not supported");
const FiniteElement &fe = *fes->GetTypicalFE();
const DenseMatrix &w = Geometries.GetGeomToPerfGeomJac(fe.GetGeomType());
if (current_W_type != fe.GetGeomType())
{
current_W_type = fe.GetGeomType();
current_W = Geometries.GetGeomToPerfGeomJac(current_W_type);
}
const DofToQuad::Mode mode = DofToQuad::TENSOR;
const DofToQuad &maps = fe.GetDofToQuad(ir, mode);
const int d = maps.ndof, q = maps.nqpt;
@@ -248,7 +253,7 @@ void DiscreteAdaptTC::ComputeAllElementTargets(const FiniteElementSpace &pa_fes,
if (dim == 2)
{
const auto W = Reshape(w.Read(), 2, 2);
const auto W = Reshape(current_W.Read(), 2, 2);
const auto X = Reshape(tspec_e.Read(), d, d, ncomp, NE);
auto J = Reshape(Jtr.Write(), 2, 2, q, q, NE);
TMOPDatc2Size::Run(d, q,
@@ -257,7 +262,7 @@ void DiscreteAdaptTC::ComputeAllElementTargets(const FiniteElementSpace &pa_fes,
}
else
{
const auto W = Reshape(w.Read(), 3, 3);
const auto W = Reshape(current_W.Read(), 3, 3);
const auto X = Reshape(tspec_e.Read(), d, d, d, ncomp, NE);
auto J = Reshape(Jtr.Write(), 3, 3, q, q, q, NE);
TMOPDatc3Size::Run(d, q,
+9 -3
View File
@@ -111,8 +111,14 @@ bool TargetConstructor::ComputeAllElementTargets<2>(
MFEM_VERIFY(!fes.IsVariableOrder(), "variable orders are not supported");
const FiniteElement &fe = *fes.GetFE(0);
MFEM_VERIFY(fe.GetGeomType() == Geometry::SQUARE, "");
const DenseMatrix &w = Geometries.GetGeomToPerfGeomJac(Geometry::SQUARE);
const real_t detW = w.Det();
if (current_W_type != Geometry::SQUARE)
{
current_W_type = Geometry::SQUARE;
current_W = Geometries.GetGeomToPerfGeomJac(current_W_type);
}
current_W.HostRead(); // Needed for det
const real_t detW = current_W.Det();
const DofToQuad::Mode mode = DofToQuad::TENSOR;
const DofToQuad &maps = fe.GetDofToQuad(ir, mode);
const int d = maps.ndof, q = maps.nqpt;
@@ -120,7 +126,7 @@ bool TargetConstructor::ComputeAllElementTargets<2>(
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto W = Reshape(w.Read(), 2, 2);
const auto W = Reshape(current_W.Read(), 2, 2);
const auto *b = maps.B.Read(), *g = maps.G.Read();
auto J = Reshape(Jtr.Write(), 2, 2, q, q, NE);
+9 -3
View File
@@ -112,8 +112,14 @@ bool TargetConstructor::ComputeAllElementTargets<3>(
MFEM_VERIFY(!fes.IsVariableOrder(), "variable orders are not supported");
const FiniteElement &fe = *fes.GetFE(0);
MFEM_VERIFY(fe.GetGeomType() == Geometry::CUBE, "");
const DenseMatrix &w = Geometries.GetGeomToPerfGeomJac(Geometry::CUBE);
const real_t detW = w.Det();
if (current_W_type != Geometry::CUBE)
{
current_W_type = Geometry::CUBE;
current_W = Geometries.GetGeomToPerfGeomJac(current_W_type);
}
current_W.HostRead(); // Needed for det
const real_t detW = current_W.Det();
const DofToQuad::Mode mode = DofToQuad::TENSOR;
const DofToQuad &maps = fe.GetDofToQuad(ir, mode);
const int d = maps.ndof, q = maps.nqpt;
@@ -121,7 +127,7 @@ bool TargetConstructor::ComputeAllElementTargets<3>(
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto W = Reshape(w.Read(), 3, 3);
const auto W = Reshape(current_W.Read(), 3, 3);
const auto *b = maps.B.Read(), *g = maps.G.Read();
auto J = Reshape(Jtr.Write(), 3, 3, q, q, q, NE);
+104 -46
View File
@@ -28,11 +28,12 @@
namespace mfem
{
template <class T>
class Array;
/** @brief Swap objects of type T. The operation is performed using the most
specialized `swap` function from the `mfem` namespace (or other visible
`swap` functions), or using the `std::swap` generic template and its
specializations in the standard library. */
template <class T> inline void Swap(T &a, T &b);
template <class T>
void Swap(Array<T> &, Array<T> &);
/**
Abstract data type Array.
@@ -60,8 +61,6 @@ public:
using reference = T&; ///< Type alias for stl.
using const_reference = const T&; ///< Type alias for stl.
friend void Swap<T>(Array<T> &, Array<T> &);
/// Creates an empty array
inline Array() : size(0) { }
@@ -103,18 +102,34 @@ public:
explicit inline Array(std::initializer_list<CT> values);
/// Move constructor ("steals" data from 'src')
inline Array(Array<T> &&src) : Array() { Swap(src, *this); }
Array(Array<T> &&src) : data(std::move(src.data)), size(src.size)
{
src.size = 0;
}
/// Destructor
inline ~Array() { data.Delete(); }
/// Assignment operator: deep copy from 'src'.
/// Copy assignment operator: deep copy from 'src'.
Array<T> &operator=(const Array<T> &src) { src.Copy(*this); return *this; }
/// Move assignment operator
Array<T> &operator=(Array<T> &&src)
{
if (this == &src) { return *this; }
Swap(src); // Swap does not use move assignment!
src.DeleteAll();
return *this;
}
/// Assignment operator (deep copy) from @a src, an Array of convertible type.
template <typename CT>
inline Array &operator=(const Array<CT> &src);
/// Swap the contents of the Array with @a other.
/** Implemented without using move assignment, avoiding DeleteAll() calls. */
inline void Swap(Array &other);
/// Return the data as 'T *'
inline operator T *() { return data; }
@@ -236,6 +251,15 @@ public:
/// Make this Array a reference to 'master'.
inline void MakeRef(const Array &master);
/// Reset the Array to use the given external Memory @a mem and size @a s.
/** If @a own_mem is false, the Array will not own any of the pointers of
@a mem.
Note that when @a own_mem is true, the @a mem object can be destroyed
immediately by the caller but `mem.Delete()` should NOT be called since
the Array object takes ownership of all pointers owned by @a mem. */
inline void NewMemoryAndSize(const Memory<T> &mem, int s, bool own_mem);
/**
* @brief Permute the array using the provided indices. Sorts the indices
* variable in the process, thereby destroying the permutation. The rvalue
@@ -400,19 +424,11 @@ inline bool operator!=(const Array<T> &LHS, const Array<T> &RHS)
template <typename T> const T &AsConst(const T &a) { return a; }
template <class T>
class Array2D;
template <class T>
void Swap(Array2D<T> &, Array2D<T> &);
/// Dynamic 2D array using row-major layout
template <class T>
class Array2D
{
private:
friend void Swap<T>(Array2D<T> &, Array2D<T> &);
Array<T> array1d;
int M, N; // number of rows and columns
@@ -423,6 +439,7 @@ public:
Array2D(int m, int n) : array1d(m*n) { M = m; N = n; }
Array2D(const Array2D &) = default;
Array2D(Array2D &&) = default;
/// Set the 2D array size to m x n.
void SetSize(int m, int n) { array1d.SetSize(m*n); M = m; N = n; }
@@ -450,10 +467,10 @@ public:
}
/** @brief Save the Array2D to the stream @a out using the format @a fmt.
The format @a fmt can be:
0 - write the number of rows and columns, followed by all entries
1 - write only the entries, using row-major layout
The format @a fmt can be:
- 0 - write the number of rows and columns, followed by all entries
- 1 - write only the entries, using row-major layout
*/
void Save(std::ostream &os, int fmt = 0) const
{
@@ -462,10 +479,10 @@ public:
}
/** @brief Read an Array2D from the stream @a in using format @a fmt.
The format @a fmt can be:
0 - read the number of rows and columns, then the entries
1 - read NumRows() x NumCols() entries, using row-major layout
The format @a fmt can be:
- 0 - read the number of rows and columns, then the entries
- 1 - read NumRows() x NumCols() entries, using row-major layout
*/
void Load(std::istream &in, int fmt = 0)
{
@@ -481,17 +498,24 @@ public:
void Load(int new_size0,int new_size1, std::istream &in)
{ SetSize(new_size0,new_size1); Load(in, 1); }
/// Create a copy of the internal array to the provided @a copy.
void Copy(Array2D &copy) const
{ copy.M = M; copy.N = N; array1d.Copy(copy.array1d); }
void Copy(Array2D &copy) const { copy = *this; }
/// Set all entries of the array to the provided constant.
inline void operator=(const T &a)
{ array1d = a; }
inline Array2D& operator=(const Array2D &a) = default;
/// Copy assignment.
Array2D& operator=(const Array2D &) = default;
/// Make this Array a reference to 'master'
/// Move assignment.
Array2D& operator=(Array2D &&) = default;
/// Swap the contents of the Array2D with @a other.
/** Implemented without using move assignment, avoiding some unnecessary
calls. */
inline void Swap(Array2D &other);
/// Make this Array2D a reference to 'master'
inline void MakeRef(const Array2D &master)
{ M = master.M; N = master.N; array1d.MakeRef(master.array1d); }
@@ -708,21 +732,22 @@ protected:
};
/// inlines ///
// Inlines
template <class T>
inline void Swap(T &a, T &b)
template <class T> inline void Swap(T &a, T &b)
{
T c = a;
a = b;
b = c;
using std::swap;
swap(a, b);
}
template <class T>
inline void Swap(Array<T> &a, Array<T> &b)
/** @brief Swap of Array<T> objects for use with standard library algorithms.
Also, used by mfem::Swap(). */
template <typename T>
inline void swap(Array<T> &a, Array<T> &b)
{
Swap(a.data, b.data);
Swap(a.size, b.size);
// swap without using move assignment
a.Swap(b);
}
template <class T>
@@ -756,6 +781,13 @@ inline Array<T>::Array(const CT (&values)[N]) : Array(N)
std::copy(values, values + N, begin());
}
template <class T>
inline void Array<T>::Swap(Array &other)
{
mfem::Swap(data, other.data);
std::swap(size, other.size);
}
template <class T>
inline void Array<T>::GrowSize(int minsize)
{
@@ -1001,8 +1033,7 @@ template <class T>
inline void Array<T>::DeleteAll()
{
const bool use_dev = data.UseDevice();
data.Delete();
data.Reset();
data.Delete(); // calls data.Reset(h_mt) as well
size = 0;
data.UseDevice(use_dev);
}
@@ -1010,9 +1041,12 @@ inline void Array<T>::DeleteAll()
template <typename T>
inline void Array<T>::Copy(Array &copy) const
{
copy.SetSize(Size(), data.GetMemoryType());
data.CopyTo(copy.data, Size());
copy.data.UseDevice(data.UseDevice());
copy.SetSize(Size());
const bool use_dev = UseDevice() || copy.UseDevice();
copy.data.UseDevice(use_dev);
// keep 'copy.data' where it is, unless 'use_dev' is true
if (use_dev) { copy.Write(); }
copy.data.CopyFrom(data, Size());
}
template <class T>
@@ -1039,6 +1073,22 @@ inline void Array<T>::MakeRef(const Array &master)
data.MakeAlias(master.GetMemory(), 0, size);
}
template <class T>
inline void Array<T>::NewMemoryAndSize(
const Memory<T> &mem, int s, bool own_mem)
{
data.Delete();
size = s;
if (own_mem)
{
data = mem;
}
else
{
data.MakeAlias(mem, 0, s);
}
}
template <class T>
inline void Array<T>::GetSubArray(int offset, int sa_size, Array<T> &sa) const
{
@@ -1103,12 +1153,20 @@ inline T *Array2D<T>::operator[](int i)
return &array1d[i*N];
}
template <class T>
inline void Swap(Array2D<T> &a, Array2D<T> &b)
inline void Array2D<T>::Swap(Array2D<T> &other)
{
Swap(a.array1d, b.array1d);
Swap(a.N, b.N);
mfem::Swap(array1d, other.array1d);
std::swap(M, other.M);
std::swap(N, other.N);
}
/** @brief Swap of Array2D<T> objects for use with standard library algorithms.
Also, used by mfem::Swap(). */
template <typename T>
inline void swap(Array2D<T> &a, Array2D<T> &b)
{
a.Swap(b);
}
+26 -9
View File
@@ -209,27 +209,27 @@ public:
Memory() { Reset(); }
/// Copy constructor: default.
Memory(const Memory &orig) = default;
Memory(const Memory &) = default;
/** Move constructor. Sets the pointers and associated ownership of validity
flags of @a *this to those of @a other. Resets @a other. */
Memory(Memory &&orig)
Memory(Memory &&other)
{
*this = orig;
orig.Reset();
*this = other;
other.Reset();
}
/// Copy-assignment operator: default.
Memory &operator=(const Memory &orig) = default;
Memory &operator=(const Memory &) = default;
/** Move assignment operator. Sets the pointers and associated ownership of
validity flags of @a *this to those of @a other. Resets @a other. */
Memory &operator=(Memory &&orig)
Memory &operator=(Memory &&other)
{
// Guard self-assignment:
if (this == &orig) { return *this; }
*this = orig;
orig.Reset();
if (this == &other) { return *this; }
*this = other;
other.Reset();
return *this;
}
@@ -280,6 +280,14 @@ public:
/** @note The destructor will NOT delete the current memory. */
~Memory() = default;
/// Swap without using move assignment, avoiding Reset() calls.
void Swap(Memory &other)
{
Memory tmp(*this);
*this = other;
other = tmp;
}
/** @brief Return true if the host pointer is owned. Ownership indicates
whether the pointer will be deleted by the method Delete(). */
bool OwnsHostPtr() const { return flags & OWNS_HOST; }
@@ -601,6 +609,15 @@ private:
};
/** @brief Swap of Memory<T> objects for use with standard library algorithms.
Also, used by mfem::Swap(). */
template <typename T>
void swap(Memory<T> &a, Memory<T> &b)
{
a.Swap(b);
}
/** The MFEM memory manager class. Host-side pointers are inserted into this
manager which keeps track of the associated device pointer, and where the
data currently resides. */
+1 -2
View File
@@ -14,12 +14,11 @@
#include "../config/config.hpp"
#include "array.hpp"
#include "../linalg/vector.hpp"
namespace mfem
{
class Vector;
/** Class for parsing command-line options.
The class is initialized with argc and argv, and new options are added with
+32 -75
View File
@@ -24,19 +24,6 @@ namespace mfem
using namespace std;
Table::Table(const Table &table)
{
size = table.size;
if (size >= 0)
{
const int nnz = table.I[size];
I.New(size+1, table.I.GetMemoryType());
J.New(nnz, table.J.GetMemoryType());
I.CopyFrom(table.I, size+1);
J.CopyFrom(table.J, nnz);
}
}
Table::Table(const Table &table1,
const Table &table2, int offset)
{
@@ -45,8 +32,8 @@ Table::Table(const Table &table1,
size = table1.size;
const int nnz = table1.I[size] + table2.I[size];
I.New(size+1, table1.I.GetMemoryType());
J.New(nnz, table1.J.GetMemoryType());
I.SetSize(size+1);
J.SetSize(nnz);
I[0] = 0;
Array<int> row;
@@ -80,8 +67,8 @@ Table::Table(const Table &table1,
size = table1.size;
const int nnz = table1.I[size] + table2.I[size] + table3.I[size];
I.New(size+1, table1.I.GetMemoryType());
J.New(nnz, table1.J.GetMemoryType());
I.SetSize(size+1);
J.SetSize(nnz);
I[0] = 0;
Array<int> row;
@@ -109,23 +96,13 @@ Table::Table(const Table &table1,
}
}
Table& Table::operator=(const Table &rhs)
{
Clear();
Table copy(rhs);
Swap(copy);
return *this;
}
Table::Table (int dim, int connections_per_row)
{
int i, j, sum = dim * connections_per_row;
size = dim;
I.New(size+1);
J.New(sum);
I.SetSize(size+1);
J.SetSize(sum);
I[0] = 0;
for (i = 1; i <= size; i++)
@@ -135,12 +112,12 @@ Table::Table (int dim, int connections_per_row)
}
}
Table::Table (int nrows, int *partitioning)
Table::Table(int nrows, int *partitioning)
{
size = nrows;
I.New(size+1);
J.New(size);
I.SetSize(size+1);
J.SetSize(size);
for (int i = 0; i < size; i++)
{
@@ -150,9 +127,9 @@ Table::Table (int nrows, int *partitioning)
I[size] = size;
}
void Table::MakeI (int nrows)
void Table::MakeI(int nrows)
{
SetDims (nrows, 0);
SetDims(nrows, 0);
for (int i = 0; i <= nrows; i++)
{
@@ -169,11 +146,10 @@ void Table::MakeJ()
j = I[i], I[i] = k, k += j;
}
J.Delete();
J.New(I[size]=k);
J.SetSize(I[size]=k);
}
void Table::AddConnections (int r, const int *c, int nc)
void Table::AddConnections(int r, const int *c, int nc)
{
int *jp = J+I[r];
@@ -217,14 +193,12 @@ void Table::SetDims(int rows, int nnz)
if (size != rows)
{
size = rows;
I.Delete();
(rows >= 0) ? I.New(rows+1) : I.Reset();
(rows >= 0) ? I.SetSize(rows+1) : I.DeleteAll();
}
if (j != nnz)
{
J.Delete();
(nnz > 0) ? J.New(nnz) : J.Reset();
(nnz > 0) ? J.SetSize(nnz) : J.DeleteAll();
}
if (size >= 0)
@@ -234,7 +208,7 @@ void Table::SetDims(int rows, int nnz)
}
}
int Table::operator() (int i, int j) const
int Table::operator()(int i, int j) const
{
if ( i>=size || i<0 )
{
@@ -278,14 +252,12 @@ void Table::SortRows()
void Table::SetIJ(int *newI, int *newJ, int newsize)
{
I.Delete();
J.Delete();
if (newsize >= 0)
{
size = newsize;
}
I.Wrap(newI, size+1, true);
J.Wrap(newJ, I[size], true);
I.MakeRef(newI, size + 1, true);
J.MakeRef(newJ, I[size], true);
}
int Table::Push(int i, int j)
@@ -326,7 +298,7 @@ void Table::Finalize()
if (sum != I[size])
{
int *NewJ = Memory<int>(sum);
Array<int> NewJ(sum);
for (i=0; i<size; i++)
{
@@ -341,9 +313,7 @@ void Table::Finalize()
}
I[size] = sum;
J.Delete();
J.Wrap(NewJ, sum, true);
mfem::Swap(J, NewJ);
MFEM_ASSERT(sum == n, "sum = " << sum << ", n = " << n);
}
@@ -356,8 +326,8 @@ void Table::MakeFromList(int nrows, const Array<Connection> &list)
size = nrows;
int nnz = list.Size();
I.New(size+1);
J.New(nnz);
I.SetSize(size+1);
J.SetSize(nnz);
for (int i = 0, k = 0; i <= size; i++)
{
@@ -433,17 +403,14 @@ void Table::Save(std::ostream &os) const
void Table::Load(std::istream &in)
{
I.Delete();
J.Delete();
in >> size;
I.New(size+1);
I.SetSize(size+1);
for (int i = 0; i <= size; i++)
{
in >> I[i];
}
int nnz = I[size];
J.New(nnz);
J.SetSize(nnz);
for (int j = 0; j < nnz; j++)
{
in >> J[j];
@@ -452,11 +419,9 @@ void Table::Load(std::istream &in)
void Table::Clear()
{
I.Delete();
J.Delete();
I.DeleteAll();
J.DeleteAll();
size = -1;
I.Reset();
J.Reset();
}
void Table::Copy(Table & copy) const
@@ -466,9 +431,7 @@ void Table::Copy(Table & copy) const
void Table::Swap(Table & other)
{
mfem::Swap(size, other.size);
mfem::Swap(I, other.I);
mfem::Swap(J, other.J);
mfem::Swap(*this, other);
}
std::size_t Table::MemoryUsage() const
@@ -477,13 +440,7 @@ std::size_t Table::MemoryUsage() const
return (size+1 + I[size]) * sizeof(int);
}
Table::~Table ()
{
I.Delete();
J.Delete();
}
void Transpose (const Table &A, Table &At, int ncols_A_)
void Transpose(const Table &A, Table &At, int ncols_A_)
{
const int *i_A = A.GetI();
const int *j_A = A.GetJ();
@@ -545,7 +502,7 @@ void Transpose(const Array<int> &A, Table &At, int ncols_A_)
At.ShiftUpI();
}
void Mult (const Table &A, const Table &B, Table &C)
void Mult(const Table &A, const Table &B, Table &C)
{
int i, j, k, l, m;
const int *i_A = A.GetI();
@@ -616,18 +573,18 @@ void Mult (const Table &A, const Table &B, Table &C)
}
Table * Mult (const Table &A, const Table &B)
Table * Mult(const Table &A, const Table &B)
{
Table * C = new Table;
Mult(A,B,*C);
return C;
}
STable::STable (int dim, int connections_per_row) :
STable::STable(int dim, int connections_per_row) :
Table(dim, connections_per_row)
{}
int STable::operator() (int i, int j) const
int STable::operator()(int i, int j) const
{
if (i < j)
{
+90 -103
View File
@@ -28,166 +28,162 @@ struct Connection
{
int from, to;
Connection() = default;
Connection(int from, int to) : from(from), to(to) {}
Connection(int from, int to) : from(from), to(to) { }
bool operator== (const Connection &rhs) const
bool operator==(const Connection &rhs) const
{ return (from == rhs.from) && (to == rhs.to); }
bool operator< (const Connection &rhs) const
bool operator<(const Connection &rhs) const
{ return (from == rhs.from) ? (to < rhs.to) : (from < rhs.from); }
};
/** Data type Table. Table stores the connectivity of elements of TYPE I
to elements of TYPE II, for example, it may be Element-To-Face
connectivity table, etc. */
/** @brief Table stores the connectivity of elements of TYPE I to elements of
TYPE II. For example, it may be the element-to-face connectivity table. */
class Table
{
protected:
/// size is the number of TYPE I elements.
int size;
int size; ///< The number of TYPE I elements.
/** Arrays for the connectivity information in the CSR storage.
I is of size "size+1", J is of size the number of connections
between TYPE I to TYPE II elements (actually stored I[size]). */
Memory<int> I, J;
/// @name Arrays for the connectivity information in the CSR storage.
/// @{
/// The length of the I array is 'size + 1',
Array<int> I;
/// @brief The length of the J array is equal to the number of connections
/// between TYPE I and TYPE II elements.
Array<int> J;
/// @}
public:
/// Creates an empty table
Table() { size = -1; }
/// Copy constructor
Table(const Table &);
/** Merge constructors
This is used to combine two or three tables into one table.*/
/// Merge constructor: combine two tables into one table.
Table(const Table &table1,
const Table &table2, int offset2);
/// Merge constructor: combine three tables into one table.
Table(const Table &table1,
const Table &table2, int offset2,
const Table &table3, int offset3);
/// Assignment operator: deep copy
Table& operator=(const Table &rhs);
/// Create a table with an upper limit for the number of connections.
explicit Table (int dim, int connections_per_row = 3);
explicit Table(int dim, int connections_per_row = 3);
/** Create a table from a list of connections, see MakeFromList(). */
/// Create a table from a list of connections, see MakeFromList().
Table(int nrows, Array<Connection> &list) : size(-1)
{ MakeFromList(nrows, list); }
/** Create a table with one entry per row with column indices given
by 'partitioning'. */
Table (int nrows, int *partitioning);
/// @brief Create a table with one entry per row with column indices given by
/// @a partitioning.
Table(int nrows, int *partitioning);
/// Next 7 methods are used together with the default constructor
void MakeI (int nrows);
void AddAColumnInRow (int r) { I[r]++; }
void AddColumnsInRow (int r, int ncol) { I[r] += ncol; }
/// @name Used together with the default constructor
/// @{
void MakeI(int nrows);
void AddAColumnInRow(int r) { I[r]++; }
void AddColumnsInRow(int r, int ncol) { I[r] += ncol; }
void MakeJ();
void AddConnection (int r, int c) { J[I[r]++] = c; }
void AddConnections (int r, const int *c, int nc);
void AddConnection(int r, int c) { J[I[r]++] = c; }
void AddConnections(int r, const int *c, int nc);
void ShiftUpI();
/// @}
/// Set the size and the number of connections for the table.
void SetSize(int dim, int connections_per_row);
/** Set the rows and the number of all connections for the table.
Does NOT initialize the whole array I ! (I[0]=0 and I[rows]=nnz only) */
/// @brief Set the rows and the number of all connections for the table.
///
/// Does NOT initialize the whole array I ! (I[0]=0 and I[rows]=nnz only)
void SetDims(int rows, int nnz);
/// Returns the number of TYPE I elements.
inline int Size() const { return size; }
/** Returns the number of connections in the table. If Finalize() is
not called, it returns the number of possible connections established
by the used constructor. Otherwise, it is exactly the number of
established connections before calling Finalize(). */
inline int Size_of_connections() const { HostReadI(); return I[size]; }
/// @brief Returns the number of connections in the table.
///
/// If Finalize() is not called, it returns the number of possible
/// connections established by the used constructor. Otherwise, it is exactly
/// the number of established connections after calling Finalize(). */
inline int Size_of_connections() const { return J.Size(); }
/** Returns index of the connection between element i of TYPE I and
element j of TYPE II. If there is no connection between element i
and element j established in the table, then the return value is -1. */
/// @brief Returns index of the connection between element i of TYPE I and
/// element j of TYPE II.
///
/// If there is no connection between element i and element j established in
/// the table, then the return value is -1.
int operator() (int i, int j) const;
/// Return row i in array row (the Table must be finalized)
void GetRow(int i, Array<int> &row) const;
int RowSize(int i) const { return I[i+1]-I[i]; }
int RowSize(int i) const { return I[i+1] - I[i]; }
const int *GetRow(int i) const { return J+I[i]; }
int *GetRow(int i) { return J+I[i]; }
const int *GetRow(int i) const { return J.GetMemory() + I[i]; }
int *GetRow(int i) { return J.GetMemory() + I[i]; }
int *GetI() { return I; }
int *GetJ() { return J; }
const int *GetI() const { return I; }
const int *GetJ() const { return J; }
int *GetI() { return I.GetData(); }
int *GetJ() { return J.GetData(); }
const int *GetI() const { return I.GetData(); }
const int *GetJ() const { return J.GetData(); }
Memory<int> &GetIMemory() { return I; }
Memory<int> &GetJMemory() { return J; }
const Memory<int> &GetIMemory() const { return I; }
const Memory<int> &GetJMemory() const { return J; }
Memory<int> &GetIMemory() { return I.GetMemory(); }
Memory<int> &GetJMemory() { return J.GetMemory(); }
const Memory<int> &GetIMemory() const { return I.GetMemory(); }
const Memory<int> &GetJMemory() const { return J.GetMemory(); }
const int *ReadI(bool on_dev = true) const
{ return mfem::Read(I, I.Capacity(), on_dev); }
int *WriteI(bool on_dev = true)
{ return mfem::Write(I, I.Capacity(), on_dev); }
int *ReadWriteI(bool on_dev = true)
{ return mfem::ReadWrite(I, I.Capacity(), on_dev); }
const int *HostReadI() const
{ return mfem::Read(I, I.Capacity(), false); }
int *HostWriteI()
{ return mfem::Write(I, I.Capacity(), false); }
int *HostReadWriteI()
{ return mfem::ReadWrite(I, I.Capacity(), false); }
const int *ReadI(bool on_dev = true) const { return I.Read(on_dev); }
int *WriteI(bool on_dev = true) { return I.Write(on_dev); }
int *ReadWriteI(bool on_dev = true) { return I.ReadWrite(on_dev); }
const int *HostReadI() const { return I.HostRead(); }
int *HostWriteI() { return I.HostWrite(); }
int *HostReadWriteI() { return I.HostReadWrite(); }
const int *ReadJ(bool on_dev = true) const
{ return mfem::Read(J, J.Capacity(), on_dev); }
int *WriteJ(bool on_dev = true)
{ return mfem::Write(J, J.Capacity(), on_dev); }
int *ReadWriteJ(bool on_dev = true)
{ return mfem::ReadWrite(J, J.Capacity(), on_dev); }
const int *HostReadJ() const
{ return mfem::Read(J, J.Capacity(), false); }
int *HostWriteJ()
{ return mfem::Write(J, J.Capacity(), false); }
int *HostReadWriteJ()
{ return mfem::ReadWrite(J, J.Capacity(), false); }
const int *ReadJ(bool on_dev = true) const { return J.Read(on_dev); }
int *WriteJ(bool on_dev = true) { return J.Write(on_dev); }
int *ReadWriteJ(bool on_dev = true) { return J.ReadWrite(on_dev); }
const int *HostReadJ() const { return J.HostRead(); }
int *HostWriteJ() { return J.HostWrite(); }
int *ReadWriteJ() { return J.HostReadWrite(); }
/// @brief Sort the column (TYPE II) indices in each row.
/// Sort the column (TYPE II) indices in each row.
void SortRows();
/// Replace the #I and #J arrays with the given @a newI and @a newJ arrays.
/** If @a newsize < 0, then the size of the Table is not modified. */
void SetIJ(int *newI, int *newJ, int newsize = -1);
/** Establish connection between element i and element j in the table.
The return value is the index of the connection. It returns -1 if it
fails to establish the connection. Possibilities are there is not
enough memory on row i to establish connection to j, an attempt to
establish new connection after calling Finalize(). */
/// Establish connection between element i and element j in the table.
/** The return value is the index of the connection. It returns -1 if it
fails to establish the connection. Possibilities are there is not enough
memory on row i to establish connection to j, an attempt to establish new
connection after calling Finalize(). */
int Push( int i, int j );
/** Finalize the table initialization. The function may be called
only once, after the table has been initialized, in order to compress
array J (by getting rid of -1's in array J). Calling this function
will "freeze" the table and function Push will work no more.
Note: The table is functional even without calling Finalize(). */
/// Finalize the table initialization.
/** The function may be called only once, after the table has been
initialized, in order to compress array J (by getting rid of -1's in
array J). Calling this function will "freeze" the table and function Push
will work no more. Note: The table is functional even without calling
Finalize(). */
void Finalize();
/** Create the table from a list of connections {(from, to)}, where 'from'
is a TYPE I index and 'to' is a TYPE II index. The list is assumed to be
sorted and free of duplicities, i.e., you need to call Array::Sort and
Array::Unique before calling this method. */
/// @brief Create the table from a list of connections {(from, to)}, where
/// 'from' is a TYPE I index and 'to' is a TYPE II index.
///
/// The list is assumed to be sorted and free of duplicities, i.e., you need
/// to call Array::Sort and Array::Unique before calling this method. */
void MakeFromList(int nrows, const Array<Connection> &list);
/// Returns the number of TYPE II elements (after Finalize() is called).
int Width() const;
/// Call this if data has been stolen.
void LoseData() { size = -1; I.Reset(); J.Reset(); }
/// Releases ownership of and null-ifies the data.
void LoseData() { size = -1; I.LoseData(); J.LoseData(); }
/// Prints the table to stream out.
/// Prints the table to the stream @a out.
void Print(std::ostream & out = mfem::out, int width = 4) const;
void PrintMatlab(std::ostream & out) const;
@@ -200,17 +196,8 @@ public:
void Clear();
std::size_t MemoryUsage() const;
/// Destroys Table.
~Table();
};
/// Specialization of the template function Swap<> for class Table
template <> inline void Swap<Table>(Table &a, Table &b)
{
a.Swap(b);
}
/// Transpose a Table
void Transpose (const Table &A, Table &At, int ncols_A_ = -1);
Table * Transpose (const Table &A);
+4
View File
@@ -29,6 +29,8 @@ list(APPEND SRCS
mma.cpp
ode.cpp
operator.cpp
ordering.cpp
particlevector.cpp
solvers.cpp
sparsemat.cpp
sparsesmoothers.cpp
@@ -63,6 +65,8 @@ list(APPEND HDRS
mma.hpp
ode.hpp
operator.hpp
ordering.hpp
particlevector.hpp
solvers.hpp
sparsemat.hpp
sparsesmoothers.hpp
+6 -51
View File
@@ -41,23 +41,12 @@ using namespace std;
DenseMatrix::DenseMatrix() : Matrix(0) { }
DenseMatrix::DenseMatrix(const DenseMatrix &m) : Matrix(m.height, m.width)
{
const int hw = height * width;
if (hw > 0)
{
MFEM_ASSERT(m.data, "invalid source matrix");
data.New(hw);
std::memcpy(data, m.data, sizeof(real_t)*hw);
}
}
DenseMatrix::DenseMatrix(int s) : Matrix(s)
{
MFEM_ASSERT(s >= 0, "invalid DenseMatrix size: " << s);
if (s > 0)
{
data.New(s*s);
data.SetSize(s*s);
*this = 0.0; // init with zeroes
}
}
@@ -69,7 +58,7 @@ DenseMatrix::DenseMatrix(int m, int n) : Matrix(m, n)
const int capacity = m*n;
if (capacity > 0)
{
data.New(capacity);
data.SetSize(capacity);
*this = 0.0; // init with zeroes
}
}
@@ -81,7 +70,7 @@ DenseMatrix::DenseMatrix(const DenseMatrix &mat, char ch)
const int capacity = height*width;
if (capacity > 0)
{
data.New(capacity);
data.SetSize(capacity);
for (int i = 0; i < height; i++)
{
@@ -103,13 +92,7 @@ void DenseMatrix::SetSize(int h, int w)
}
height = h;
width = w;
const int hw = h*w;
if (hw > data.Capacity())
{
data.Delete();
data.New(hw);
*this = 0.0; // init with zeroes
}
data.SetSize(h*w, 0.0);
}
real_t &DenseMatrix::Elem(int i, int j)
@@ -643,19 +626,6 @@ DenseMatrix &DenseMatrix::operator=(const real_t *d)
return *this;
}
DenseMatrix &DenseMatrix::operator=(const DenseMatrix &m)
{
SetSize(m.height, m.width);
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] = m.data[i];
}
return *this;
}
DenseMatrix &DenseMatrix::operator+=(const real_t *m)
{
kernels::Add(Height(), Width(), m, (real_t*)data);
@@ -2362,17 +2332,9 @@ void DenseMatrix::TestInversion()
void DenseMatrix::Swap(DenseMatrix &other)
{
mfem::Swap(width, other.width);
mfem::Swap(height, other.height);
mfem::Swap(data, other.data);
mfem::Swap(*this, other);
}
DenseMatrix::~DenseMatrix()
{
data.Delete();
}
void Add(const DenseMatrix &A, const DenseMatrix &B,
real_t alpha, DenseMatrix &C)
@@ -4328,7 +4290,7 @@ const
{
int n = SizeI(), ne = SizeK();
const int *I = elem_dof.GetI(), *J = elem_dof.GetJ(), *dofs;
const real_t *d_col = mfem::HostRead(tdata, n*SizeJ()*ne);
const real_t *d_col = tdata.HostRead();
real_t *yp = y.HostReadWrite();
real_t x_col;
const real_t *xp = x.HostRead();
@@ -4388,13 +4350,6 @@ DenseTensor &DenseTensor::operator=(real_t c)
return *this;
}
DenseTensor &DenseTensor::operator=(const DenseTensor &other)
{
DenseTensor new_tensor(other);
Swap(new_tensor);
return *this;
}
void BatchLUFactor(DenseTensor &Mlu, Array<int> &P, const real_t TOL)
{
BatchedLinAlg::LUFactor(Mlu, P);
+69 -119
View File
@@ -26,7 +26,7 @@ class DenseMatrix : public Matrix
friend class DenseMatrixInverse;
private:
Memory<real_t> data;
Array<real_t> data;
void Eigensystem(Vector &ev, DenseMatrix *evect = NULL);
@@ -40,9 +40,6 @@ public:
Sets data = NULL and height = width = 0. */
DenseMatrix();
/// Copy constructor
DenseMatrix(const DenseMatrix &);
/// Creates square matrix of size s.
explicit DenseMatrix(int s);
@@ -58,6 +55,18 @@ public:
DenseMatrix(real_t *d, int h, int w)
: Matrix(h, w) { UseExternalData(d, h, w); }
/// Copy constructor (deep copy).
DenseMatrix(const DenseMatrix &) = default;
/// Move constructor.
DenseMatrix(DenseMatrix &&) = default;
/// Copy assignment (deep copy).
DenseMatrix &operator=(const DenseMatrix &) = default;
/// Move assignment.
DenseMatrix &operator=(DenseMatrix &&) = default;
/// Create a dense matrix using a braced initializer list
/// The inner lists correspond to rows of the matrix
template <int M, int N, typename T = real_t>
@@ -75,11 +84,10 @@ public:
/// Change the data array and the size of the DenseMatrix.
/** The DenseMatrix does not assume ownership of the data array, i.e. it will
not delete the data array @a d. This method should not be used with
DenseMatrix that owns its current data array. */
not delete the data array @a d. */
void UseExternalData(real_t *d, int h, int w)
{
data.Wrap(d, h*w, false);
data.MakeRef(d, h*w);
height = h; width = w;
}
@@ -88,20 +96,20 @@ public:
not delete the new array @a d. This method will delete the current data
array, if owned. */
void Reset(real_t *d, int h, int w)
{ if (OwnsData()) { data.Delete(); } UseExternalData(d, h, w); }
{ UseExternalData(d, h, w); }
/** Clear the data array and the dimensions of the DenseMatrix. This method
should not be used with DenseMatrix that owns its current data array. */
void ClearExternalData() { data.Reset(); height = width = 0; }
void ClearExternalData() { data.LoseData(); height = width = 0; }
/// Delete the matrix data array (if owned) and reset the matrix state.
void Clear()
{ if (OwnsData()) { data.Delete(); } ClearExternalData(); }
{ data.DeleteAll(); height = width = 0; }
/// For backward compatibility define Size to be synonym of Width()
int Size() const { return Width(); }
// Total size = width*height
/// Total size = width*height
int TotalSize() const { return width*height; }
/// Change the size of the DenseMatrix to s x s.
@@ -110,18 +118,19 @@ public:
/// Change the size of the DenseMatrix to h x w.
void SetSize(int h, int w);
/// Returns the matrix data array.
/// Returns the matrix data array. Warning: this method casts away constness.
inline real_t *Data() const
{ return const_cast<real_t*>((const real_t*)data);}
/// Returns the matrix data array.
/// Returns the matrix data array. Warning: this method casts away constness.
inline real_t *GetData() const { return Data(); }
Memory<real_t> &GetMemory() { return data; }
const Memory<real_t> &GetMemory() const { return data; }
Memory<real_t> &GetMemory() { return data.GetMemory(); }
const Memory<real_t> &GetMemory() const { return data.GetMemory(); }
/// Return the DenseMatrix data (host pointer) ownership flag.
inline bool OwnsData() const { return data.OwnsHostPtr(); }
inline bool OwnsData() const { return data.OwnsData(); }
/// Returns reference to a_{ij}.
inline real_t &operator()(int i, int j);
@@ -249,9 +258,6 @@ public:
/// Copy the matrix entries from the given array
DenseMatrix &operator=(const real_t *d);
/// Sets the matrix size and elements equal to those of m
DenseMatrix &operator=(const DenseMatrix &m);
DenseMatrix &operator+=(const real_t *m);
DenseMatrix &operator+=(const DenseMatrix &m);
@@ -477,33 +483,24 @@ public:
std::size_t MemoryUsage() const { return data.Capacity() * sizeof(real_t); }
/// Shortcut for mfem::Read( GetMemory(), TotalSize(), on_dev).
const real_t *Read(bool on_dev = true) const
{ return mfem::Read(data, Height()*Width(), on_dev); }
const real_t *Read(bool on_dev = true) const { return data.Read(on_dev); }
/// Shortcut for mfem::Read(GetMemory(), TotalSize(), false).
const real_t *HostRead() const
{ return mfem::Read(data, Height()*Width(), false); }
const real_t *HostRead() const { return data.HostRead(); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), on_dev).
real_t *Write(bool on_dev = true)
{ return mfem::Write(data, Height()*Width(), on_dev); }
real_t *Write(bool on_dev = true) { return data.Write(on_dev); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), false).
real_t *HostWrite()
{ return mfem::Write(data, Height()*Width(), false); }
real_t *HostWrite() { return data.HostWrite(); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), on_dev).
real_t *ReadWrite(bool on_dev = true)
{ return mfem::ReadWrite(data, Height()*Width(), on_dev); }
real_t *ReadWrite(bool on_dev = true) { return data.ReadWrite(on_dev); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), false).
real_t *HostReadWrite()
{ return mfem::ReadWrite(data, Height()*Width(), false); }
real_t *HostReadWrite() { return data.HostReadWrite(); }
void Swap(DenseMatrix &other);
/// Destroys dense matrix.
virtual ~DenseMatrix();
};
/// C = A + alpha*B
@@ -1114,69 +1111,44 @@ class DenseTensor
{
private:
mutable DenseMatrix Mk;
Memory<real_t> tdata;
int nk;
Array<real_t> tdata;
int ni, nj, nk;
public:
DenseTensor()
{
nk = 0;
}
DenseTensor() : ni(0), nj(0), nk(0) { }
DenseTensor(int i, int j, int k)
: Mk(NULL, i, j)
{
nk = k;
tdata.New(i*j*k);
}
DenseTensor(int i, int j, int k) : tdata(i*j*k), ni(i), nj(j), nk(k) { }
DenseTensor(real_t *d, int i, int j, int k)
: Mk(NULL, i, j)
{
nk = k;
tdata.Wrap(d, i*j*k, false);
}
: tdata(d, i*j*k), ni(i), nj(j), nk(k) { }
DenseTensor(int i, int j, int k, MemoryType mt)
: Mk(NULL, i, j)
{
nk = k;
tdata.New(i*j*k, mt);
}
: tdata(i*j*k, mt), ni(i), nj(j), nk(k) { }
/// Copy constructor: deep copy
DenseTensor(const DenseTensor &other)
: Mk(NULL, other.Mk.height, other.Mk.width), nk(other.nk)
{
const int size = Mk.Height()*Mk.Width()*nk;
if (size > 0)
{
tdata.New(size, other.tdata.GetMemoryType());
tdata.CopyFrom(other.tdata, size);
}
}
int SizeI() const { return Mk.Height(); }
int SizeJ() const { return Mk.Width(); }
int SizeI() const { return ni; }
int SizeJ() const { return nj; }
int SizeK() const { return nk; }
int TotalSize() const { return SizeI()*SizeJ()*SizeK(); }
void SetSize(int i, int j, int k, MemoryType mt_ = MemoryType::PRESERVE)
{
const MemoryType mt = mt_ == MemoryType::PRESERVE ? tdata.GetMemoryType() : mt_;
tdata.Delete();
Mk.UseExternalData(NULL, i, j);
const MemoryType mt = mt_ == MemoryType::PRESERVE ?
tdata.GetMemory().GetMemoryType() : mt_;
ni = i;
nj = j;
nk = k;
tdata.New(i*j*k, mt);
Mk.ClearExternalData();
tdata.SetSize(i*j*k, mt);
}
void UseExternalData(real_t *ext_data, int i, int j, int k)
{
tdata.Delete();
Mk.UseExternalData(NULL, i, j);
ni = i;
nj = j;
nk = k;
tdata.Wrap(ext_data, i*j*k, false);
Mk.ClearExternalData();
tdata.MakeRef(ext_data, i*j*k);
}
/// @brief Reset the DenseTensor to use the given external Memory @a mem and
@@ -1191,25 +1163,16 @@ public:
void NewMemoryAndSize(const Memory<real_t> &mem, int i, int j, int k,
bool own_mem)
{
tdata.Delete();
Mk.UseExternalData(NULL, i, j);
ni = i;
nj = j;
nk = k;
if (own_mem)
{
tdata = mem;
}
else
{
tdata.MakeAlias(mem, 0, i*j*k);
}
Mk.ClearExternalData();
tdata.NewMemoryAndSize(mem, i*j*k, own_mem);
}
/// Sets the tensor elements equal to constant c
DenseTensor &operator=(real_t c);
/// Copy assignment operator (performs a deep copy)
DenseTensor &operator=(const DenseTensor &other);
DenseMatrix &operator()(int k)
{
return operator()(k, Mk);
@@ -1221,16 +1184,13 @@ public:
DenseMatrix &operator()(int k, DenseMatrix& buff)
{
MFEM_ASSERT_INDEX_IN_RANGE(k, 0, SizeK());
buff.UseExternalData(nullptr, SizeI(), SizeJ());
buff.data = Memory<real_t>(GetData(k), SizeI()*SizeJ(), false);
buff.UseExternalData(GetData(k), SizeI(), SizeJ());
return buff;
}
const DenseMatrix &operator()(int k, DenseMatrix& buff) const
{
MFEM_ASSERT_INDEX_IN_RANGE(k, 0, SizeK());
buff.UseExternalData(nullptr, SizeI(), SizeJ());
buff.data = Memory<real_t>(const_cast<real_t*>(GetData(k)), SizeI()*SizeJ(),
false);
buff.UseExternalData(const_cast<real_t*>(GetData(k)), SizeI(), SizeJ());
return buff;
}
@@ -1253,21 +1213,21 @@ public:
real_t *GetData(int k)
{
MFEM_ASSERT_INDEX_IN_RANGE(k, 0, SizeK());
return tdata+k*Mk.Height()*Mk.Width();
return tdata.GetMemory()+k*ni*nj;
}
const real_t *GetData(int k) const
{
MFEM_ASSERT_INDEX_IN_RANGE(k, 0, SizeK());
return tdata+k*Mk.Height()*Mk.Width();
return tdata.GetMemory()+k*ni*nj;
}
real_t *Data() { return tdata; }
real_t *Data() { return tdata.GetData(); }
const real_t *Data() const { return tdata; }
const real_t *Data() const { return tdata.GetData(); }
Memory<real_t> &GetMemory() { return tdata; }
const Memory<real_t> &GetMemory() const { return tdata; }
Memory<real_t> &GetMemory() { return tdata.GetMemory(); }
const Memory<real_t> &GetMemory() const { return tdata.GetMemory(); }
/** Matrix-vector product from unassembled element matrices, assuming both
'x' and 'y' use the same elem_dof table. */
@@ -1276,40 +1236,30 @@ public:
void Clear()
{ UseExternalData(NULL, 0, 0, 0); }
std::size_t MemoryUsage() const { return nk*Mk.MemoryUsage(); }
std::size_t MemoryUsage() const { return tdata.MemoryUsage(); }
/// Shortcut for mfem::Read( GetMemory(), TotalSize(), on_dev).
const real_t *Read(bool on_dev = true) const
{ return mfem::Read(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
const real_t *Read(bool on_dev = true) const { return tdata.Read(on_dev); }
/// Shortcut for mfem::Read(GetMemory(), TotalSize(), false).
const real_t *HostRead() const
{ return mfem::Read(tdata, Mk.Height()*Mk.Width()*nk, false); }
const real_t *HostRead() const { return tdata.HostRead(); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), on_dev).
real_t *Write(bool on_dev = true)
{ return mfem::Write(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
real_t *Write(bool on_dev = true) { return tdata.Write(on_dev); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), false).
real_t *HostWrite()
{ return mfem::Write(tdata, Mk.Height()*Mk.Width()*nk, false); }
real_t *HostWrite() { return tdata.HostWrite(); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), on_dev).
real_t *ReadWrite(bool on_dev = true)
{ return mfem::ReadWrite(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
real_t *ReadWrite(bool on_dev = true) { return tdata.ReadWrite(on_dev); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), false).
real_t *HostReadWrite()
{ return mfem::ReadWrite(tdata, Mk.Height()*Mk.Width()*nk, false); }
real_t *HostReadWrite() { return tdata.HostReadWrite(); }
void Swap(DenseTensor &t)
{
mfem::Swap(tdata, t.tdata);
mfem::Swap(nk, t.nk);
Mk.Swap(t.Mk);
mfem::Swap(*this, t);
}
~DenseTensor() { tdata.Delete(); }
};
/** @brief Compute the LU factorization of a batch of matrices. Calls
+4 -2
View File
@@ -3151,7 +3151,8 @@ void GatherBlockOffsetData(MPI_Comm comm, const int rank, const int nprocs,
{
std::vector<std::vector<int>> all_block_num_loc(numBlocks);
MPI_Allgather(&num_loc, 1, MPI_INT, all_num_loc.data(), 1, MPI_INT, comm);
MPI_Allgather(const_cast<int*>(&num_loc), 1, MPI_INT, all_num_loc.data(), 1,
MPI_INT, comm);
for (int j = 0; j < numBlocks; ++j)
{
@@ -3159,7 +3160,8 @@ void GatherBlockOffsetData(MPI_Comm comm, const int rank, const int nprocs,
blockProcOffsets[j].resize(nprocs);
const int blockNumRows = offsets[j + 1] - offsets[j];
MPI_Allgather(&blockNumRows, 1, MPI_INT, all_block_num_loc[j].data(), 1,
MPI_Allgather(const_cast<int*>(&blockNumRows), 1, MPI_INT,
all_block_num_loc[j].data(), 1,
MPI_INT, comm);
blockProcOffsets[j][0] = 0;
for (int i = 0; i < nprocs - 1; ++i)
+2
View File
@@ -38,6 +38,8 @@
#include "batched/solver.hpp"
#include "tensor.hpp"
#include "filteredsolver.hpp"
#include "ordering.hpp"
#include "particlevector.hpp"
#ifdef MFEM_USE_AMGX
#include "amgxsolver.hpp"
+8 -1
View File
@@ -19,7 +19,8 @@
namespace mfem
{
void Matrix::Print (std::ostream & os, int width_) const
template <class T>
void MatrixMP<T>::Print(std::ostream & os, int width_) const
{
using namespace std;
// output flags = scientific + show sign
@@ -40,4 +41,10 @@ void Matrix::Print (std::ostream & os, int width_) const
os << '\n';
}
template class MatrixMP<float>;
template class MatrixMP<double>;
template class AbstractSparseMatrixMP<float>;
template class AbstractSparseMatrixMP<double>;
}
+41 -25
View File
@@ -21,31 +21,39 @@ namespace mfem
// Abstract data types matrix, inverse matrix
class MatrixInverse;
template <class T>
class MatrixInverseMP;
/// Abstract data type matrix
class Matrix : public Operator
template <class T>
class MatrixMP : public OperatorMP<T>
{
friend class MatrixInverse;
friend class MatrixInverseMP<T>;
protected:
using OperatorBase::height;
using OperatorBase::width;
public:
/// Creates a square matrix of size s.
explicit Matrix(int s) : Operator(s) { }
explicit MatrixMP(int s) : OperatorMP<T>(s) { }
/// Creates a matrix of the given height and width.
explicit Matrix(int h, int w) : Operator(h, w) { }
explicit MatrixMP(int h, int w) : OperatorMP<T>(h, w) { }
/// Returns whether the matrix is a square matrix.
bool IsSquare() const { return (height == width); }
/// Returns reference to a_{ij}.
virtual real_t &Elem(int i, int j) = 0;
virtual T &Elem(int i, int j) = 0;
/// Returns constant reference to a_{ij}.
virtual const real_t &Elem(int i, int j) const = 0;
virtual const T &Elem(int i, int j) const = 0;
/// Returns a pointer to (an approximation) of the matrix inverse.
virtual MatrixInverse *Inverse() const = 0;
virtual MatrixInverseMP<T> *Inverse() const = 0;
/// Finalizes the matrix initialization.
virtual void Finalize(int) { }
@@ -54,30 +62,35 @@ public:
virtual void Print(std::ostream & os = mfem::out, int width_ = 4) const;
/// Destroys matrix.
virtual ~Matrix() { }
virtual ~MatrixMP() { }
};
using Matrix = MatrixMP<real_t>;
/// Abstract data type for matrix inverse
class MatrixInverse : public Solver
template <class T>
class MatrixInverseMP : public SolverMP<T>
{
public:
MatrixInverse() { }
MatrixInverseMP() { }
/// Creates approximation of the inverse of square matrix
MatrixInverse(const Matrix &mat)
: Solver(mat.height, mat.width) { }
MatrixInverseMP(const MatrixMP<T> &mat)
: SolverMP<T>(mat.height, mat.width) { }
};
using MatrixInverse = MatrixInverseMP<real_t>;
/// Abstract data type for sparse matrices
class AbstractSparseMatrix : public Matrix
template <class T>
class AbstractSparseMatrixMP : public MatrixMP<T>
{
public:
/// Creates a square matrix of the given size.
explicit AbstractSparseMatrix(int s = 0) : Matrix(s) { }
explicit AbstractSparseMatrixMP(int s = 0) : MatrixMP<T>(s) { }
/// Creates a matrix of the given height and width.
explicit AbstractSparseMatrix(int h, int w) : Matrix(h, w) { }
explicit AbstractSparseMatrixMP(int h, int w) : MatrixMP<T>(h, w) { }
/// Returns the number of non-zeros in a matrix
virtual int NumNonZeroElems() const = 0;
@@ -86,30 +99,33 @@ public:
/** Returns:
- 0 if @a cols and @a srow are copies of the values in the matrix.
- 1 if @a cols and @a srow are views of the values in the matrix. */
virtual int GetRow(const int row, Array<int> &cols, Vector &srow) const = 0;
virtual int GetRow(const int row, Array<int> &cols,
VectorMP<T> &srow) const = 0;
/** @brief If the matrix is square, this method will place 1 on the diagonal
(i,i) if row i has "almost" zero l1-norm.
If entry (i,i) does not belong to the sparsity pattern of A, then an
error will occur. */
virtual void EliminateZeroRows(const real_t threshold = 1e-12) = 0;
virtual void EliminateZeroRows(const T threshold = 1e-12) = 0;
/// Matrix-Vector Multiplication y = A*x
void Mult(const Vector &x, Vector &y) const override = 0;
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override = 0;
/// Matrix-Vector Multiplication y = y + val*A*x
void AddMult(const Vector &x, Vector &y,
const real_t val = 1.) const override = 0;
void AddMult(const VectorMP<T> &x, VectorMP<T> &y,
const T val = 1.) const override = 0;
/// MatrixTranspose-Vector Multiplication y = A'*x
void MultTranspose(const Vector &x, Vector &y) const override = 0;
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override = 0;
/// MatrixTranspose-Vector Multiplication y = y + val*A'*x
void AddMultTranspose(const Vector &x, Vector &y,
const real_t val = 1.) const override = 0;
void AddMultTranspose(const VectorMP<T> &x, VectorMP<T> &y,
const T val = 1.) const override = 0;
/// Destroys AbstractSparseMatrix.
virtual ~AbstractSparseMatrix() { }
virtual ~AbstractSparseMatrixMP() { }
};
using AbstractSparseMatrix = AbstractSparseMatrixMP<real_t>;
}
#endif
+5 -1
View File
@@ -23,7 +23,11 @@
namespace mfem
{
// forward declaration
class Vector;
template <class T>
class VectorMP;
using Vector = VectorMP<real_t>;
/** \brief MMA (Method of Moving Asymptotes) solves a nonlinear optimization
* problem involving an objective function, inequality constraints,
+3 -3
View File
@@ -204,9 +204,9 @@ void MUMPSSolver::SetOperator(const Operator &op)
delete id;
}
#ifdef MFEM_USE_SINGLE
id = new SMUMPS_STRUC_C;
id = new SMUMPS_STRUC_C();
#else
id = new DMUMPS_STRUC_C;
id = new DMUMPS_STRUC_C();
#endif
id->sym = mat_type;
@@ -341,7 +341,7 @@ void MUMPSSolver::InitRhsSol(int nrhs) const
#else
if (myid == 0)
{
delete rhs_glob;
delete [] rhs_glob;
rhs_glob = new real_t[nrhs * id->lrhs];
id->rhs = rhs_glob;
}
+229 -1
View File
@@ -28,8 +28,14 @@ std::string ODESolver::ImplicitTypes =
" GA : 40 -- 50 - Generalized-alpha,\n\t"
" AM : 51 - AM1, 52 - AM2, 53 - AM3, 54 - AM4\n";
std::string ODESolver::IMEXTypes =
"\n\tIMEX solver: \n\t"
" (L-Stab): 61 - Forward Backward Euler, 62 - IMEXRK2(2,2,2),\n\t"
" 63 - IMEXRK2(2,3,2), 64 - IMEX_DIRK_RK3\n";
std::string ODESolver::Types = ODESolver::ExplicitTypes +
ODESolver::ImplicitTypes;
ODESolver::ImplicitTypes +
ODESolver::IMEXTypes;
std::unique_ptr<ODESolver> ODESolver::Select(int ode_solver_type)
{
@@ -106,6 +112,20 @@ std::unique_ptr<ODESolver> ODESolver::SelectImplicit(int ode_solver_type)
}
}
std::unique_ptr<ODESolver> ODESolver::SelectIMEX(const int ode_solver_type)
{
using ode_ptr = std::unique_ptr<ODESolver>;
switch (ode_solver_type)
{
// L-stable IMEX methods
case 61: return ode_ptr(new IMEXExpImplEuler);
case 62: return ode_ptr(new IMEXRK2);
case 63: return ode_ptr(new IMEXRK2_3StageExplicit);
case 64: return ode_ptr(new IMEX_DIRK_RK3);
default: MFEM_ABORT("Unknown ODE solver type: " << ode_solver_type );
}
}
void ODEStateDataVector::SetSize( int vsize, MemoryType m_t)
{
@@ -1277,4 +1297,212 @@ void GeneralizedAlpha2Solver::Step(Vector &x, Vector &dxdt,
t += dt;
}
void IMEXExpImplEuler::Init(TimeDependentOperator &f_)
{
ODESolver::Init(f_);
int n = f->Width();
k1.SetSize(n, mem_type);
k2.SetSize(n, mem_type);
}
void IMEXExpImplEuler::Step(Vector &x, real_t &t, real_t &dt)
{
f->SetTime(t);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(x, k1);
f->SetTime(t+dt);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt, x, k2);
f->SetTime(t);
x.Add(dt, k1);
x.Add(dt, k2);
t += dt;
}
void IMEXRK2::Init(TimeDependentOperator &f_)
{
ODESolver::Init(f_);
int n = f->Width();
k1_exp.SetSize(n, mem_type);
k2_exp.SetSize(n, mem_type);
k_imp.SetSize(n, mem_type);
y.SetSize(n, mem_type);
}
void IMEXRK2::Step(Vector &x, real_t &t, real_t &dt)
{
double gamma = 1 - sqrt(2)/2;
double delta = 1 - 1/(2*gamma);
f->SetTime(t);
//K1 exp is just f_1(t, x)
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(x, k1_exp);
//K2 exp is f_1(t + gamma dt, x + dt gamma K1)
f->SetTime(t + gamma*dt);
add(x, dt*gamma, k1_exp, y);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k2_exp);
//K2_imp = f_2(t + gamma dt, x + dt gamma K2_imp)
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, x, k_imp);
//reuse k_imp to avoid extra vector
//K3_imp = f_2(t+dt,x + dt(1-gamma)K2_imp + dt gamma K3_imp)
f -> SetTime(t + dt);
//add(x, dt*(1-gamma), k2_imp, z);
//optimization to avoid extra vector
x.Add(dt*(1-gamma), k_imp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
//f->ImplicitSolve(dt*gamma, z, k3_imp);
//reuse k_imp to avoid extra vector
f->ImplicitSolve(dt*gamma, x, k_imp);
//add it all up
x.Add(dt*delta, k1_exp);
x.Add(dt*(1-delta), k2_exp);
//x.Add(dt*(1-gamma), k2_imp); it is already added to x above
x.Add(dt*gamma, k_imp);
t += dt;
}
void IMEXRK2_3StageExplicit::Init(TimeDependentOperator &f_)
{
ODESolver::Init(f_);
int n = f->Width();
k1_exp.SetSize(n, mem_type);
k2_exp.SetSize(n, mem_type);
k3_exp.SetSize(n, mem_type);
k_imp.SetSize(n, mem_type);
y.SetSize(n, mem_type);
}
void IMEXRK2_3StageExplicit::Step(Vector &x, real_t &t, real_t &dt)
{
double gamma = 1 - sqrt(2)/2;
double delta = -2*sqrt(2)/3;
f->SetTime(t);
//K1 exp is just f_1(t, x)
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(x, k1_exp);
//K2 exp is f_1(t + gamma dt, x + dt gamma K1)
f->SetTime(t + gamma*dt);
add(x, dt*gamma, k1_exp, y);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k2_exp);
//K3 Exp is f_1(t + dt, x + dt delta K1_exp + dt (1-delta) K2_exp)
f->SetTime(t + dt);
add(x, dt*delta, k1_exp, y);
//add(y, dt*(1-delta), k2_exp, w);
//optimization to avoid extra vector
y.Add(dt*(1-delta), k2_exp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k3_exp);
//K2_imp = f_2(t + gamma dt, x + dt gamma K2_imp)
f->SetTime(t + gamma*dt);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, x, k_imp);
//K3_imp = f_2(t+dt,x + dt(1-gamma)K2_imp + dt gamma K3_imp)
f -> SetTime(t + dt);
//add(x, dt*(1-gamma), k2_imp, z);
x.Add(dt*(1-gamma), k_imp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, x, k_imp);
//add it all up
x.Add(dt*delta, k2_exp);
x.Add(dt*(1-delta), k3_exp);
//x.Add(dt*(1-gamma), k2_imp); // it is already added to x above
x.Add(dt*gamma, k_imp);
t += dt;
}
void IMEX_DIRK_RK3::Init(TimeDependentOperator &f_)
{
ODESolver::Init(f_);
int n = f->Width();
k1_exp.SetSize(n, mem_type);
k2_exp.SetSize(n, mem_type);
k3_exp.SetSize(n, mem_type);
k4_exp.SetSize(n, mem_type);
k2_imp.SetSize(n, mem_type);
k3_imp.SetSize(n, mem_type);
y.SetSize(n, mem_type);
}
void IMEX_DIRK_RK3::Step(Vector &x, real_t &t, real_t &dt)
{
double gamma = 0.4358665215;
double b1 = 1.208496649;
double b2 = -0.644363171;
double a_31 = 0.3212788860;
double a_32 = 0.3966543747;
double a_41 = -0.105858296;
double a_42 = 0.5529291479;
double a_43 = 0.5529291479;
//K1_exp
f->SetTime(t);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(x, k1_exp);
//K2_imp, K2_exp
f->SetTime(t + gamma*dt);
add(x, dt*gamma, k1_exp, y);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k2_exp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, x, k2_imp);
//K3_imp, K3_exp
f->SetTime(t + (1+gamma)/2*dt);
add(x, dt*a_31, k1_exp, y);
//add(y, dt*a_32, k2_exp, w);
//optimization to avoid extra vector
y.Add(dt*a_32, k2_exp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k3_exp);
add(x, dt*(1-gamma)/2, k2_imp, y);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, y, k3_imp);
//K4_imp, K4_exp
f->SetTime(t+dt);
add(x, dt*a_41, k1_exp, y);
//add(y, dt*a_42, k2_exp, v);
y.Add(dt*a_42, k2_exp);
//add(w, dt*a_43, k3_exp, v);
y.Add(dt*a_43, k3_exp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_1);
f->Mult(y, k4_exp);
//add(x, dt*b1, k2_imp, z);
//add(z, dt*b2, k3_imp, u);
//optimization to avoid extra vector
x.Add(dt*b1, k2_imp);
x.Add(dt*b2, k3_imp);
f->SetEvalMode(TimeDependentOperator::ADDITIVE_TERM_2);
f->ImplicitSolve(dt*gamma, x, k3_imp);
//add it all together
x.Add(dt*b1, k2_exp);
x.Add(dt*b2, k3_exp);
x.Add(dt*gamma, k4_exp);
//x.Add(dt*b1, k2_imp); //already added above
//x.Add(dt*b2, k3_imp); //already added above
x.Add(dt*gamma, k3_imp);
t += dt;
}
}
+83
View File
@@ -106,6 +106,17 @@ public:
/// Abstract class for solving systems of ODEs: dx/dt = f(x,t)
/** For systems of split ODEs:
$$ M dx/dt = f_1(x,t) + f_2(x,t) $$
where $ M^{-1} f_1 $ and $ M^{-1} f_2 $ are treated differently (e.g.,
explicitly and implicitly), the solver class expects a
TimeDependentOperator with split functionality. Setting
TimeDependentOperator::EvalMode = TimeDependentOperator::ADDITIVE_TERM_1
and calling TimeDependentOperator::Mult() should return
$ k_1=M^{-1} f_1(x,t) $. Setting TimeDependentOperator::EvalMode =
TimeDependentOperator::ADDITIVE_TERM_2 and calling
TimeDependentOperator::ImplicitSolve() should solve
$ M k_2 = f_2(x+\gamma k_2,t) $. */
class ODESolver
{
protected:
@@ -184,6 +195,7 @@ public:
// Help info for ODESolver options
static MFEM_EXPORT std::string ExplicitTypes;
static MFEM_EXPORT std::string ImplicitTypes;
static MFEM_EXPORT std::string IMEXTypes;
static MFEM_EXPORT std::string Types;
/// Function for selecting the desired ODESolver (Explicit and Implicit)
@@ -203,6 +215,13 @@ public:
static MFEM_EXPORT std::unique_ptr<ODESolver> SelectImplicit(
const int ode_solver_type);
/// Function for selecting the desired IMEX ODESolver
/// Returns an ODESolver pointer based on an type
/// Caller gets ownership of the object and is responsible for its deletion
static MFEM_EXPORT std::unique_ptr<ODESolver> SelectIMEX(
const int ode_solver_type);
virtual ~ODESolver() { }
};
@@ -931,6 +950,70 @@ public:
};
/// Forward-backward Euler method
class IMEXExpImplEuler : public ODESolver
{
private:
Vector k1; Vector k2;
public:
void Init(TimeDependentOperator &f_) override;
void Step(Vector &x, real_t &t, real_t &dt) override;
};
/// Second order, two-stage implicit-explicit (IMEX) Runge-Kutta (RK) method
/** L-stable IMEX RK2 method adopted from "On the Stability of IMEX Upwind gSBP
Schemes for 1D Linear AdvectionDifusion Equations" by Sigrun Ortleb. Same
as (2,2,2) from "Implicit-explicit Runge-Kutta methods for time-dependent
partial differential equations" by Ascher, Ruuth and Spiteri, Applied
Numerical Mathematics (1997). */
class IMEXRK2 : public ODESolver
{
private:
Vector k1_exp; Vector k2_exp; Vector k_imp;
//helper vector
Vector y;
public:
void Init(TimeDependentOperator &f_) override;
void Step(Vector &x, real_t &t, real_t &dt) override;
};
/// Second order, 2/3-stage implicit-explicit (IMEX) Runge-Kutta (RK) method
/** L-stable method (2,3,2) from "Implicit-explicit Runge-Kutta methods for
time-dependent partial differential equations" by Ascher, Ruuth and
Spiteri, Applied Numerical Mathematics (1997). */
class IMEXRK2_3StageExplicit : public ODESolver
{
private:
Vector k1_exp; Vector k2_exp; Vector k3_exp;
Vector k_imp;
//helper vectors
Vector y;
public:
void Init(TimeDependentOperator &f_) override;
void Step(Vector &x, real_t &t, real_t &dt) override;
};
/// Third order, 3/4-stage implicit-explicit (IMEX) Runge-Kutta (RK) method
/** L-stable method (3,4,3) from "Implicit-explicit Runge-Kutta methods for
time-dependent partial differential equations" by Ascher, Ruuth and
Spiteri, Applied Numerical Mathematics (1997). */
class IMEX_DIRK_RK3 : public ODESolver
{
private:
Vector k1_exp; Vector k2_exp; Vector k3_exp; Vector k4_exp;
Vector k2_imp; Vector k3_imp;
//helper vectors
Vector y;
public:
void Init(TimeDependentOperator &f_) override;
void Step(Vector &x, real_t &t, real_t &dt) override;
};
}
#endif
+183 -111
View File
@@ -19,10 +19,12 @@
namespace mfem
{
void Operator::InitTVectors(const Operator *Po, const Operator *Ri,
const Operator *Pi,
Vector &x, Vector &b,
Vector &X, Vector &B) const
template <class T>
void OperatorMP<T>::InitTVectors(const OperatorMP<T> *Po,
const OperatorMP<T> *Ri,
const OperatorMP<T> *Pi,
VectorMP<T> &x, VectorMP<T> &b,
VectorMP<T> &X, VectorMP<T> &B) const
{
if (!IsIdentityProlongation(Po))
{
@@ -48,23 +50,27 @@ void Operator::InitTVectors(const Operator *Po, const Operator *Ri,
}
}
void Operator::AddMult(const Vector &x, Vector &y, const real_t a) const
template <class T>
void OperatorMP<T>::AddMult(const VectorMP<T> &x, VectorMP<T> &y,
const T a) const
{
mfem::Vector z(y.Size());
mfem::VectorMP<T> z(y.Size());
Mult(x, z);
y.Add(a, z);
}
void Operator::AddMultTranspose(const Vector &x, Vector &y,
const real_t a) const
template <class T>
void OperatorMP<T>::AddMultTranspose(const VectorMP<T> &x, VectorMP<T> &y,
const T a) const
{
mfem::Vector z(y.Size());
mfem::VectorMP<T> z(y.Size());
MultTranspose(x, z);
y.Add(a, z);
}
void Operator::ArrayMult(const Array<const Vector *> &X,
Array<Vector *> &Y) const
template <class T>
void OperatorMP<T>::ArrayMult(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y) const
{
MFEM_ASSERT(X.Size() == Y.Size(),
"Number of columns mismatch in Operator::Mult!");
@@ -75,8 +81,9 @@ void Operator::ArrayMult(const Array<const Vector *> &X,
}
}
void Operator::ArrayMultTranspose(const Array<const Vector *> &X,
Array<Vector *> &Y) const
template <class T>
void OperatorMP<T>::ArrayMultTranspose(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y) const
{
MFEM_ASSERT(X.Size() == Y.Size(),
"Number of columns mismatch in Operator::MultTranspose!");
@@ -87,8 +94,10 @@ void Operator::ArrayMultTranspose(const Array<const Vector *> &X,
}
}
void Operator::ArrayAddMult(const Array<const Vector *> &X, Array<Vector *> &Y,
const real_t a) const
template <class T>
void OperatorMP<T>::ArrayAddMult(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y,
const T a) const
{
MFEM_ASSERT(X.Size() == Y.Size(),
"Number of columns mismatch in Operator::AddMult!");
@@ -99,8 +108,9 @@ void Operator::ArrayAddMult(const Array<const Vector *> &X, Array<Vector *> &Y,
}
}
void Operator::ArrayAddMultTranspose(const Array<const Vector *> &X,
Array<Vector *> &Y, const real_t a) const
template <class T>
void OperatorMP<T>::ArrayAddMultTranspose(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y, const T a) const
{
MFEM_ASSERT(X.Size() == Y.Size(),
"Number of columns mismatch in Operator::AddMultTranspose!");
@@ -111,44 +121,48 @@ void Operator::ArrayAddMultTranspose(const Array<const Vector *> &X,
}
}
void Operator::FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
Operator* &Aout, Vector &X, Vector &B,
int copy_interior)
template <class T>
void OperatorMP<T>::FormLinearSystem(const Array<int> &ess_tdof_list,
VectorMP<T> &x, VectorMP<T> &b,
OperatorMP<T>* &Aout, VectorMP<T> &X, VectorMP<T> &B,
int copy_interior)
{
const Operator *P = this->GetProlongation();
const Operator *R = this->GetRestriction();
const OperatorMP<T> *P = this->GetProlongation();
const OperatorMP<T> *R = this->GetRestriction();
InitTVectors(P, R, P, x, b, X, B);
if (!copy_interior) { X.SetSubVectorComplement(ess_tdof_list, 0.0); }
ConstrainedOperator *constrainedA;
ConstrainedOperatorMP<T> *constrainedA;
FormConstrainedSystemOperator(ess_tdof_list, constrainedA);
constrainedA->EliminateRHS(X, B);
Aout = constrainedA;
}
void Operator::FormRectangularLinearSystem(
template <class T>
void OperatorMP<T>::FormRectangularLinearSystem(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list, Vector &x, Vector &b,
Operator* &Aout, Vector &X, Vector &B)
const Array<int> &test_tdof_list, VectorMP<T> &x, VectorMP<T> &b,
OperatorMP<T>* &Aout, VectorMP<T> &X, VectorMP<T> &B)
{
const Operator *Pi = this->GetProlongation();
const Operator *Po = this->GetOutputProlongation();
const Operator *Ri = this->GetRestriction();
const OperatorMP<T> *Pi = this->GetProlongation();
const OperatorMP<T> *Po = this->GetOutputProlongation();
const OperatorMP<T> *Ri = this->GetRestriction();
InitTVectors(Po, Ri, Pi, x, b, X, B);
RectangularConstrainedOperator *constrainedA;
RectangularConstrainedOperatorMP<T> *constrainedA;
FormRectangularConstrainedSystemOperator(trial_tdof_list, test_tdof_list,
constrainedA);
constrainedA->EliminateRHS(X, B);
Aout = constrainedA;
}
void Operator::RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x)
template <class T>
void OperatorMP<T>::RecoverFEMSolution(const VectorMP<T> &X,
const VectorMP<T> &b, VectorMP<T> &x)
{
// Same for Rectangular and Square operators
const Operator *P = this->GetProlongation();
const OperatorMP<T> *P = this->GetProlongation();
if (!IsIdentityProlongation(P))
{
// Apply conforming prolongation
@@ -165,26 +179,28 @@ void Operator::RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x)
}
}
Operator * Operator::SetupRAP(const Operator *Pi, const Operator *Po)
template <class T>
OperatorMP<T> * OperatorMP<T>::SetupRAP(const OperatorMP<T> *Pi,
const OperatorMP<T> *Po)
{
Operator *rap;
OperatorMP<T> *rap;
if (!IsIdentityProlongation(Pi))
{
if (!IsIdentityProlongation(Po))
{
rap = new RAPOperator(*Po, *this, *Pi);
rap = new RAPOperatorMP<T>(*Po, *this, *Pi);
}
else
{
rap = new ProductOperator(this, Pi, false,false);
rap = new ProductOperatorMP<T>(this, Pi, false, false);
}
}
else
{
if (!IsIdentityProlongation(Po))
{
TransposeOperator * PoT = new TransposeOperator(Po);
rap = new ProductOperator(PoT, this, true,false);
TransposeOperatorMP<T> * PoT = new TransposeOperatorMP<T>(Po);
rap = new ProductOperatorMP<T>(PoT, this, true, false);
}
else
{
@@ -194,67 +210,74 @@ Operator * Operator::SetupRAP(const Operator *Pi, const Operator *Po)
return rap;
}
void Operator::FormConstrainedSystemOperator(
const Array<int> &ess_tdof_list, ConstrainedOperator* &Aout)
template <class T>
void OperatorMP<T>::FormConstrainedSystemOperator(
const Array<int> &ess_tdof_list, ConstrainedOperatorMP<T>* &Aout)
{
const Operator *P = this->GetProlongation();
Operator *rap = SetupRAP(P, P);
const OperatorMP<T> *P = this->GetProlongation();
OperatorMP<T> *rap = SetupRAP(P, P);
// Impose the boundary conditions through a ConstrainedOperator, which owns
// the rap operator when P and R are non-trivial
ConstrainedOperator *A = new ConstrainedOperator(rap, ess_tdof_list,
rap != this);
ConstrainedOperatorMP<T> *A = new ConstrainedOperatorMP<T>(rap, ess_tdof_list,
rap != this);
Aout = A;
}
void Operator::FormRectangularConstrainedSystemOperator(
template <class T>
void OperatorMP<T>::FormRectangularConstrainedSystemOperator(
const Array<int> &trial_tdof_list, const Array<int> &test_tdof_list,
RectangularConstrainedOperator* &Aout)
RectangularConstrainedOperatorMP<T>* &Aout)
{
const Operator *Pi = this->GetProlongation();
const Operator *Po = this->GetOutputProlongation();
Operator *rap = SetupRAP(Pi, Po);
const OperatorMP<T> *Pi = this->GetProlongation();
const OperatorMP<T> *Po = this->GetOutputProlongation();
OperatorMP<T> *rap = SetupRAP(Pi, Po);
// Impose the boundary conditions through a RectangularConstrainedOperator,
// which owns the rap operator when P and R are non-trivial
RectangularConstrainedOperator *A
= new RectangularConstrainedOperator(rap,
trial_tdof_list, test_tdof_list,
rap != this);
RectangularConstrainedOperatorMP<T> *A
= new RectangularConstrainedOperatorMP<T>(rap,
trial_tdof_list, test_tdof_list,
rap != this);
Aout = A;
}
void Operator::FormSystemOperator(const Array<int> &ess_tdof_list,
Operator* &Aout)
template <class T>
void OperatorMP<T>::FormSystemOperator(const Array<int> &ess_tdof_list,
OperatorMP<T>* &Aout)
{
ConstrainedOperator *A;
ConstrainedOperatorMP<T> *A;
FormConstrainedSystemOperator(ess_tdof_list, A);
Aout = A;
}
void Operator::FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Operator* &Aout)
template <class T>
void OperatorMP<T>::FormRectangularSystemOperator(const Array<int>
&trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorMP<T>* &Aout)
{
RectangularConstrainedOperator *A;
RectangularConstrainedOperatorMP<T> *A;
FormRectangularConstrainedSystemOperator(trial_tdof_list, test_tdof_list, A);
Aout = A;
}
void Operator::FormDiscreteOperator(Operator* &Aout)
template <class T>
void OperatorMP<T>::FormDiscreteOperator(OperatorMP<T>* &Aout)
{
const Operator *Pin = this->GetProlongation();
const Operator *Rout = this->GetOutputRestriction();
Aout = new TripleProductOperator(Rout, this, Pin,false, false, false);
const OperatorMP<T> *Pin = this->GetProlongation();
const OperatorMP<T> *Rout = this->GetOutputRestriction();
Aout = new TripleProductOperatorMP<T>(Rout, this, Pin, false, false, false);
}
void Operator::PrintMatlab(std::ostream & os, int n, int m) const
template <class T>
void OperatorMP<T>::PrintMatlab(std::ostream & os, int n, int m) const
{
using namespace std;
if (n == 0) { n = width; }
if (m == 0) { m = height; }
Vector x(n), y(m);
VectorMP<T> x(n), y(m);
x = 0.0;
os << setiosflags(ios::scientific | ios::showpos);
@@ -273,7 +296,8 @@ void Operator::PrintMatlab(std::ostream & os, int n, int m) const
}
}
void Operator::PrintMatlab(std::ostream &os) const
template <class T>
void OperatorMP<T>::PrintMatlab(std::ostream &os) const
{
PrintMatlab(os, width, height);
}
@@ -404,9 +428,11 @@ SumOperator::~SumOperator()
if (ownB) { delete B; }
}
ProductOperator::ProductOperator(const Operator *A, const Operator *B,
bool ownA, bool ownB)
: Operator(A->Height(), B->Width()),
template <class T>
ProductOperatorMP<T>::ProductOperatorMP(const OperatorMP<T> *A,
const OperatorMP<T> *B,
bool ownA, bool ownB)
: OperatorMP<T>(A->Height(), B->Width()),
A(A), B(B), ownA(ownA), ownB(ownB), z(A->Width())
{
MFEM_VERIFY(A->Width() == B->Height(),
@@ -423,16 +449,18 @@ ProductOperator::ProductOperator(const Operator *A, const Operator *B,
}
}
ProductOperator::~ProductOperator()
template <class T>
ProductOperatorMP<T>::~ProductOperatorMP()
{
if (ownA) { delete A; }
if (ownB) { delete B; }
}
RAPOperator::RAPOperator(const Operator &Rt_, const Operator &A_,
const Operator &P_)
: Operator(Rt_.Width(), P_.Width()), Rt(Rt_), A(A_), P(P_)
template <class T>
RAPOperatorMP<T>::RAPOperatorMP(const OperatorMP<T> &Rt_,
const OperatorMP<T> &A_,
const OperatorMP<T> &P_)
: OperatorMP<T>(Rt_.Width(), P_.Width()), Rt(Rt_), A(A_), P(P_)
{
MFEM_VERIFY(Rt.Height() == A.Height(),
"incompatible Operators: Rt.Height() = " << Rt.Height()
@@ -463,11 +491,11 @@ RAPOperator::RAPOperator(const Operator &Rt_, const Operator &A_,
APx.SetSize(A.Height(), mem_type);
}
TripleProductOperator::TripleProductOperator(
const Operator *A, const Operator *B, const Operator *C,
template <class T>
TripleProductOperatorMP<T>::TripleProductOperatorMP(
const OperatorMP<T> *A, const OperatorMP<T> *B, const OperatorMP<T> *C,
bool ownA, bool ownB, bool ownC)
: Operator(A->Height(), C->Width())
: OperatorMP<T>(A->Height(), C->Width())
, A(A), B(B), C(C)
, ownA(ownA), ownB(ownB), ownC(ownC)
{
@@ -500,18 +528,20 @@ TripleProductOperator::TripleProductOperator(
t2.SetSize(B->Height(), mem_type);
}
TripleProductOperator::~TripleProductOperator()
template <class T>
TripleProductOperatorMP<T>::~TripleProductOperatorMP()
{
if (ownA) { delete A; }
if (ownB) { delete B; }
if (ownC) { delete C; }
}
ConstrainedOperator::ConstrainedOperator(Operator *A, const Array<int> &list,
bool own_A_,
DiagonalPolicy diag_policy_)
: Operator(A->Height(), A->Width()), A(A), own_A(own_A_),
template <class T>
ConstrainedOperatorMP<T>::ConstrainedOperatorMP(OperatorMP<T> *A,
const Array<int> &list,
bool own_A_,
DiagonalPolicy diag_policy_)
: OperatorMP<T>(A->Height(), A->Width()), A(A), own_A(own_A_),
diag_policy(diag_policy_)
{
// 'mem_class' should work with A->Mult() and mfem::forall():
@@ -521,11 +551,12 @@ ConstrainedOperator::ConstrainedOperator(Operator *A, const Array<int> &list,
constraint_list.MakeRef(list);
// typically z and w are large vectors, so use the device (GPU) to perform
// operations on them
z.SetSize(height, mem_type); z.UseDevice(true);
w.SetSize(height, mem_type); w.UseDevice(true);
z.SetSize(this->height, mem_type); z.UseDevice(true);
w.SetSize(this->height, mem_type); w.UseDevice(true);
}
void ConstrainedOperator::AssembleDiagonal(Vector &diag) const
template <class T>
void ConstrainedOperatorMP<T>::AssembleDiagonal(VectorMP<T> &diag) const
{
A->AssembleDiagonal(diag);
@@ -556,7 +587,9 @@ void ConstrainedOperator::AssembleDiagonal(Vector &diag) const
}
}
void ConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
template <class T>
void ConstrainedOperatorMP<T>::EliminateRHS(const VectorMP<T> &x,
VectorMP<T> &b) const
{
w = 0.0;
const int csz = constraint_list.Size();
@@ -583,8 +616,10 @@ void ConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
});
}
void ConstrainedOperator::ConstrainedMult(const Vector &x, Vector &y,
const bool transpose) const
template <class T>
void ConstrainedOperatorMP<T>::ConstrainedMult(const VectorMP<T> &x,
VectorMP<T> &y,
const bool transpose) const
{
const int csz = constraint_list.Size();
if (csz == 0)
@@ -645,8 +680,10 @@ void ConstrainedOperator::ConstrainedMult(const Vector &x, Vector &y,
}
}
void ConstrainedOperator::ConstrainedAbsMult(const Vector &x, Vector &y,
const bool transpose) const
template <class T>
void ConstrainedOperatorMP<T>::ConstrainedAbsMult(const VectorMP<T> &x,
VectorMP<T> &y,
const bool transpose) const
{
const int csz = constraint_list.Size();
if (csz == 0)
@@ -707,43 +744,52 @@ void ConstrainedOperator::ConstrainedAbsMult(const Vector &x, Vector &y,
}
}
void ConstrainedOperator::Mult(const Vector &x, Vector &y) const
template <class T>
void ConstrainedOperatorMP<T>::Mult(const VectorMP<T> &x, VectorMP<T> &y) const
{
constexpr bool transpose = false;
ConstrainedMult(x, y, transpose);
}
void ConstrainedOperator::AbsMult(const Vector &x, Vector &y) const
template <class T>
void ConstrainedOperatorMP<T>::AbsMult(const VectorMP<T> &x,
VectorMP<T> &y) const
{
constexpr bool transpose = false;
ConstrainedAbsMult(x, y, transpose);
}
void ConstrainedOperator::MultTranspose(const Vector &x, Vector &y) const
template <class T>
void ConstrainedOperatorMP<T>::MultTranspose(const VectorMP<T> &x,
VectorMP<T> &y) const
{
constexpr bool transpose = true;
ConstrainedMult(x, y, transpose);
}
void ConstrainedOperator::AbsMultTranspose(const Vector &x, Vector &y) const
template <class T>
void ConstrainedOperatorMP<T>::AbsMultTranspose(const VectorMP<T> &x,
VectorMP<T> &y) const
{
constexpr bool transpose = true;
ConstrainedAbsMult(x, y, transpose);
}
void ConstrainedOperator::AddMult(const Vector &x, Vector &y,
const real_t a) const
template <class T>
void ConstrainedOperatorMP<T>::AddMult(const VectorMP<T> &x, VectorMP<T> &y,
const T a) const
{
Mult(x, w);
y.Add(a, w);
}
RectangularConstrainedOperator::RectangularConstrainedOperator(
Operator *A,
template <class T>
RectangularConstrainedOperatorMP<T>::RectangularConstrainedOperatorMP(
OperatorMP<T> *A,
const Array<int> &trial_list,
const Array<int> &test_list,
bool own_A_)
: Operator(A->Height(), A->Width()), A(A), own_A(own_A_)
: OperatorMP<T>(A->Height(), A->Width()), A(A), own_A(own_A_)
{
// 'mem_class' should work with A->Mult() and mfem::forall():
mem_class = A->GetMemoryClass()*Device::GetMemoryClass();
@@ -753,12 +799,13 @@ RectangularConstrainedOperator::RectangularConstrainedOperator(
trial_constraints.MakeRef(trial_list);
test_constraints.MakeRef(test_list);
// typically z and w are large vectors, so store them on the device
z.SetSize(height, mem_type); z.UseDevice(true);
w.SetSize(width, mem_type); w.UseDevice(true);
z.SetSize(this->height, mem_type); z.UseDevice(true);
w.SetSize(this->width, mem_type); w.UseDevice(true);
}
void RectangularConstrainedOperator::EliminateRHS(const Vector &x,
Vector &b) const
template <class T>
void RectangularConstrainedOperatorMP<T>::EliminateRHS(const VectorMP<T> &x,
VectorMP<T> &b) const
{
w = 0.0;
const int trial_csz = trial_constraints.Size();
@@ -783,7 +830,9 @@ void RectangularConstrainedOperator::EliminateRHS(const Vector &x,
});
}
void RectangularConstrainedOperator::Mult(const Vector &x, Vector &y) const
template <class T>
void RectangularConstrainedOperatorMP<T>::Mult(const VectorMP<T> &x,
VectorMP<T> &y) const
{
const int trial_csz = trial_constraints.Size();
const int test_csz = test_constraints.Size();
@@ -817,8 +866,9 @@ void RectangularConstrainedOperator::Mult(const Vector &x, Vector &y) const
}
}
void RectangularConstrainedOperator::MultTranspose(const Vector &x,
Vector &y) const
template <class T>
void RectangularConstrainedOperatorMP<T>::MultTranspose(const VectorMP<T> &x,
VectorMP<T> &y) const
{
const int trial_csz = trial_constraints.Size();
const int test_csz = test_constraints.Size();
@@ -852,7 +902,9 @@ void RectangularConstrainedOperator::MultTranspose(const Vector &x,
}
}
real_t InnerProductOperator::Dot(const Vector &x, const Vector &y) const
template <class T>
T InnerProductOperatorMP<T>::Dot(const VectorMP<T> &x,
const VectorMP<T> &y) const
{
#ifndef MFEM_USE_MPI
return (x * y);
@@ -927,4 +979,24 @@ real_t PowerMethod::EstimateLargestEigenvalue(Operator& opr, Vector& v0,
return eigenvalue;
}
template class OperatorMP<float>;
template class OperatorMP<double>;
template class ConstrainedOperatorMP<float>;
template class ConstrainedOperatorMP<double>;
template class RectangularConstrainedOperatorMP<float>;
template class RectangularConstrainedOperatorMP<double>;
template class RAPOperatorMP<float>;
template class RAPOperatorMP<double>;
template class ProductOperatorMP<float>;
template class ProductOperatorMP<double>;
template class TripleProductOperatorMP<float>;
template class TripleProductOperatorMP<double>;
template class InnerProductOperatorMP<float>;
template class InnerProductOperatorMP<double>;
}
+208 -160
View File
@@ -17,32 +17,44 @@
namespace mfem
{
class ConstrainedOperator;
class RectangularConstrainedOperator;
template <class T>
class ConstrainedOperatorMP;
/// Abstract operator
class Operator
template <class T>
class RectangularConstrainedOperatorMP;
class OperatorBase
{
protected:
int height; ///< Dimension of the output / number of rows in the matrix.
int width; ///< Dimension of the input / number of columns in the matrix.
/// see FormSystemOperator()
/** @note Uses DiagonalPolicy::DIAG_ONE. */
void FormConstrainedSystemOperator(
const Array<int> &ess_tdof_list, ConstrainedOperator* &Aout);
/// see FormRectangularSystemOperator()
void FormRectangularConstrainedSystemOperator(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
RectangularConstrainedOperator* &Aout);
/** @brief Returns RAP Operator of this, using input/output Prolongation matrices
@a Pi corresponds to "P", @a Po corresponds to "Rt" */
Operator *SetupRAP(const Operator *Pi, const Operator *Po);
public:
/// Get the height (size of output) of the Operator. Synonym with NumRows().
inline int Height() const { return height; }
/// Get the width (size of input) of the Operator. Synonym with NumCols().
inline int Width() const { return width; }
enum Type
{
ANY_TYPE, ///< ID for the base class Operator, i.e. any type.
MFEM_SPARSEMAT, ///< ID for class SparseMatrix.
Hypre_ParCSR, ///< ID for class HypreParMatrix.
PETSC_MATAIJ, ///< ID for class PetscParMatrix, MATAIJ format.
PETSC_MATIS, ///< ID for class PetscParMatrix, MATIS format.
PETSC_MATSHELL, ///< ID for class PetscParMatrix, MATSHELL format.
PETSC_MATNEST, ///< ID for class PetscParMatrix, MATNEST format.
PETSC_MATHYPRE, ///< ID for class PetscParMatrix, MATHYPRE format.
PETSC_MATGENERIC, ///< ID for class PetscParMatrix, unspecified format.
Complex_Operator, ///< ID for class ComplexOperator.
MFEM_ComplexSparseMat, ///< ID for class ComplexSparseMatrix.
Complex_Hypre_ParCSR, ///< ID for class ComplexHypreParMatrix.
Complex_DenseMat, ///< ID for class ComplexDenseMatrix
MFEM_Block_Matrix, ///< ID for class BlockMatrix.
MFEM_Block_Operator ///< ID for the base class BlockOperator.
};
/// Defines operator diagonal policy upon elimination of rows and/or columns.
enum DiagonalPolicy
{
@@ -50,26 +62,45 @@ public:
DIAG_ONE, ///< Set the diagonal value to one
DIAG_KEEP ///< Keep the diagonal value
};
};
/// Abstract operator
template <class T>
class OperatorMP : public OperatorBase
{
protected:
/// see FormSystemOperator()
/** @note Uses DiagonalPolicy::DIAG_ONE. */
void FormConstrainedSystemOperator(
const Array<int> &ess_tdof_list, ConstrainedOperatorMP<T>* &Aout);
/// see FormRectangularSystemOperator()
void FormRectangularConstrainedSystemOperator(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
RectangularConstrainedOperatorMP<T>* &Aout);
/** @brief Returns RAP Operator of this, using input/output Prolongation matrices
@a Pi corresponds to "P", @a Po corresponds to "Rt" */
OperatorMP *SetupRAP(const OperatorMP<T> *Pi, const OperatorMP<T> *Po);
public:
/// Initializes memory for true vectors of linear system
void InitTVectors(const Operator *Po, const Operator *Ri, const Operator *Pi,
Vector &x, Vector &b, Vector &X, Vector &B) const;
void InitTVectors(const OperatorMP<T> *Po, const OperatorMP<T> *Ri,
const OperatorMP<T> *Pi,
VectorMP<T> &x, VectorMP<T> &b, VectorMP<T> &X, VectorMP<T> &B) const;
/// Construct a square Operator with given size s (default 0).
explicit Operator(int s = 0) { height = width = s; }
explicit OperatorMP(int s = 0) { height = width = s; }
/** @brief Construct an Operator with the given height (output size) and
width (input size). */
Operator(int h, int w) { height = h; width = w; }
OperatorMP(int h, int w) { height = h; width = w; }
/// Get the height (size of output) of the Operator. Synonym with NumRows().
inline int Height() const { return height; }
/** @brief Get the number of rows (size of output) of the Operator. Synonym
with Height(). */
inline int NumRows() const { return height; }
/// Get the width (size of input) of the Operator. Synonym with NumCols().
inline int Width() const { return width; }
/** @brief Get the number of columns (size of input) of the Operator. Synonym
with Width(). */
inline int NumCols() const { return width; }
@@ -86,61 +117,63 @@ public:
virtual MemoryClass GetMemoryClass() const { return MemoryClass::HOST; }
/// Operator application: `y=A(x)`.
virtual void Mult(const Vector &x, Vector &y) const = 0;
virtual void Mult(const VectorMP<T> &x, VectorMP<T> &y) const = 0;
/** @brief Action of the absolute-value operator: `y=|A|(x)`. The default
behavior in class Operator is to generate an error. If the Operator is a
composition of several operators, the composition unfold into a product
of absolute-value operators too. */
virtual void AbsMult(const Vector &x, Vector &y) const
virtual void AbsMult(const VectorMP<T> &x, VectorMP<T> &y) const
{ MFEM_ABORT("Operator::AbsMult() is not overridden!"); }
/** @brief Action of the transpose operator: `y=A^t(x)`. The default behavior
in class Operator is to generate an error. */
virtual void MultTranspose(const Vector &x, Vector &y) const
virtual void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const
{ MFEM_ABORT("Operator::MultTranspose() is not overridden!"); }
/** @brief Action of the transpose absolute-value operator: `y=|A|^t(x)`.
The default behavior in class Operator is to generate an error. */
virtual void AbsMultTranspose(const Vector &x, Vector &y) const
virtual void AbsMultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const
{ MFEM_ABORT("Operator::AbsMultTranspose() is not overridden!"); }
/// Operator application: `y+=A(x)` (default) or `y+=a*A(x)`.
virtual void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const;
virtual void AddMult(const VectorMP<T> &x, VectorMP<T> &y,
const T a = 1.0) const;
/// Operator transpose application: `y+=A^t(x)` (default) or `y+=a*A^t(x)`.
virtual void AddMultTranspose(const Vector &x, Vector &y,
const real_t a = 1.0) const;
virtual void AddMultTranspose(const VectorMP<T> &x, VectorMP<T> &y,
const T a = 1.0) const;
/// Operator application on a matrix: `Y=A(X)`.
virtual void ArrayMult(const Array<const Vector *> &X,
Array<Vector *> &Y) const;
virtual void ArrayMult(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y) const;
/// Action of the transpose operator on a matrix: `Y=A^t(X)`.
virtual void ArrayMultTranspose(const Array<const Vector *> &X,
Array<Vector *> &Y) const;
virtual void ArrayMultTranspose(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y) const;
/// Operator application on a matrix: `Y+=A(X)` (default) or `Y+=a*A(X)`.
virtual void ArrayAddMult(const Array<const Vector *> &X, Array<Vector *> &Y,
const real_t a = 1.0) const;
virtual void ArrayAddMult(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y,
const T a = 1.0) const;
/** @brief Operator transpose application on a matrix: `Y+=A^t(X)` (default)
or `Y+=a*A^t(X)`. */
virtual void ArrayAddMultTranspose(const Array<const Vector *> &X,
Array<Vector *> &Y, const real_t a = 1.0) const;
virtual void ArrayAddMultTranspose(const Array<const VectorMP<T> *> &X,
Array<VectorMP<T> *> &Y, const T a = 1.0) const;
/** @brief Evaluate the gradient operator at the point @a x. The default
behavior in class Operator is to generate an error. */
virtual Operator &GetGradient(const Vector &x) const
virtual OperatorMP<T> &GetGradient(const VectorMP<T> &x) const
{
MFEM_ABORT("Operator::GetGradient() is not overridden!");
return const_cast<Operator &>(*this);
return const_cast<OperatorMP<T> &>(*this);
}
/** @brief Computes the diagonal entries into @a diag. Typically, this
operation only makes sense for linear Operator%s. In some cases, only an
approximation of the diagonal is computed. */
virtual void AssembleDiagonal(Vector &diag) const
virtual void AssembleDiagonal(VectorMP<T> &diag) const
{
MFEM_CONTRACT_VAR(diag);
MFEM_ABORT("Not relevant or not implemented for this Operator.");
@@ -148,15 +181,15 @@ public:
/** @brief Prolongation operator from linear algebra (linear system) vectors,
to input vectors for the operator. `NULL` means identity. */
virtual const Operator *GetProlongation() const { return NULL; }
virtual const OperatorMP<T> *GetProlongation() const { return NULL; }
/** @brief Restriction operator from input vectors for the operator to linear
algebra (linear system) vectors. `NULL` means identity. */
virtual const Operator *GetRestriction() const { return NULL; }
virtual const OperatorMP<T> *GetRestriction() const { return NULL; }
/** @brief Prolongation operator from linear algebra (linear system) vectors,
to output vectors for the operator. `NULL` means identity. */
virtual const Operator *GetOutputProlongation() const
virtual const OperatorMP<T> *GetOutputProlongation() const
{
return GetProlongation(); // Assume square unless specialized
}
@@ -165,11 +198,11 @@ public:
form to facilitate matrix-free RAP-type operators.
`NULL` means identity. */
virtual const Operator *GetOutputRestrictionTranspose() const { return NULL; }
virtual const OperatorMP<T> *GetOutputRestrictionTranspose() const { return NULL; }
/** @brief Restriction operator from output vectors for the operator to linear
algebra (linear system) vectors. `NULL` means identity. */
virtual const Operator *GetOutputRestriction() const
virtual const OperatorMP<T> *GetOutputRestriction() const
{
return GetRestriction(); // Assume square unless specialized
}
@@ -205,8 +238,8 @@ public:
@note If there are no transformations, @a X simply reuses the data of @a
x. */
void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
Operator* &A, Vector &X, Vector &B,
VectorMP<T> &x, VectorMP<T> &b,
OperatorMP<T>* &A, VectorMP<T> &X, VectorMP<T> &B,
int copy_interior = 0);
/** @brief Form a column-constrained linear system using a matrix-free approach.
@@ -237,8 +270,8 @@ public:
x. */
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
Operator* &A, Vector &X, Vector &B);
VectorMP<T> &x, VectorMP<T> &b,
OperatorMP<T>* &A, VectorMP<T> &X, VectorMP<T> &B);
/** @brief Reconstruct a solution vector @a x (e.g. a GridFunction) from the
solution @a X of a constrained linear system obtained from
@@ -249,7 +282,8 @@ public:
@a x, for this Operator (presumably a finite element grid function). This
method has identical signature to the analogous method for bilinear
forms, though currently @a b is not used in the implementation. */
virtual void RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x);
virtual void RecoverFEMSolution(const VectorMP<T> &X, const VectorMP<T> &b,
VectorMP<T> &x);
/** @brief Return in @a A a parallel (on truedofs) version of this square
operator.
@@ -257,7 +291,7 @@ public:
This returns the same operator as FormLinearSystem(), but does without
the transformations of the right-hand side and initial guess. */
void FormSystemOperator(const Array<int> &ess_tdof_list,
Operator* &A);
OperatorMP<T>* &A);
/** @brief Return in @a A a parallel (on truedofs) version of this
rectangular operator (including constraints).
@@ -266,7 +300,7 @@ public:
without the transformations of the right-hand side. */
void FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Operator* &A);
OperatorMP<T>* &A);
/** @brief Return in @a A a parallel (on truedofs) version of this
rectangular operator.
@@ -279,7 +313,7 @@ public:
Operator maps between. These are e.g. available through the (parallel)
finite element space of any (parallel) bilinear form operator. We have:
`A(X)=[Rout (*this) Pin](X)`. */
void FormDiscreteOperator(Operator* &A);
void FormDiscreteOperator(OperatorMP<T>* &A);
/// Prints operator with input size n and output size m in Matlab format.
void PrintMatlab(std::ostream & out, int n, int m = 0) const;
@@ -288,28 +322,7 @@ public:
virtual void PrintMatlab(std::ostream & out) const;
/// Virtual destructor.
virtual ~Operator() { }
/// Enumeration defining IDs for some classes derived from Operator.
/** This enumeration is primarily used with class OperatorHandle. */
enum Type
{
ANY_TYPE, ///< ID for the base class Operator, i.e. any type.
MFEM_SPARSEMAT, ///< ID for class SparseMatrix.
Hypre_ParCSR, ///< ID for class HypreParMatrix.
PETSC_MATAIJ, ///< ID for class PetscParMatrix, MATAIJ format.
PETSC_MATIS, ///< ID for class PetscParMatrix, MATIS format.
PETSC_MATSHELL, ///< ID for class PetscParMatrix, MATSHELL format.
PETSC_MATNEST, ///< ID for class PetscParMatrix, MATNEST format.
PETSC_MATHYPRE, ///< ID for class PetscParMatrix, MATHYPRE format.
PETSC_MATGENERIC, ///< ID for class PetscParMatrix, unspecified format.
Complex_Operator, ///< ID for class ComplexOperator.
MFEM_ComplexSparseMat, ///< ID for class ComplexSparseMatrix.
Complex_Hypre_ParCSR, ///< ID for class ComplexHypreParMatrix.
Complex_DenseMat, ///< ID for class ComplexDenseMatrix
MFEM_Block_Matrix, ///< ID for class BlockMatrix.
MFEM_Block_Operator ///< ID for the base class BlockOperator.
};
virtual ~OperatorMP() { }
/// Return the type ID of the Operator class.
/** This method is intentionally non-virtual, so that it returns the ID of
@@ -319,6 +332,7 @@ public:
Type GetType() const { return ANY_TYPE; }
};
using Operator = OperatorMP<real_t>;
/// Base abstract class for first order time dependent operators.
/** Operator of the form: (u,t) -> k(u,t), where k generally solves the
@@ -788,7 +802,8 @@ public:
/// Base class for solvers
class Solver : public Operator
template <class T>
class SolverMP : public OperatorMP<T>
{
public:
/// If true, use the second argument of Mult() as an initial guess.
@@ -798,37 +813,42 @@ public:
@warning Use a Boolean expression for the second parameter (not an int)
to distinguish this call from the general rectangular constructor. */
explicit Solver(int s = 0, bool iter_mode = false)
: Operator(s) { iterative_mode = iter_mode; }
explicit SolverMP(int s = 0, bool iter_mode = false)
: OperatorMP<T>(s) { iterative_mode = iter_mode; }
/// Initialize a Solver with height @a h and width @a w.
Solver(int h, int w, bool iter_mode = false)
: Operator(h, w) { iterative_mode = iter_mode; }
SolverMP(int h, int w, bool iter_mode = false)
: OperatorMP<T>(h, w) { iterative_mode = iter_mode; }
/// Set/update the solver for the given operator.
virtual void SetOperator(const Operator &op) = 0;
virtual void SetOperator(const OperatorMP<T> &op) = 0;
};
using Solver = SolverMP<real_t>;
/// Identity Operator I: x -> x.
class IdentityOperator : public Operator
template <class T>
class IdentityOperatorMP : public OperatorMP<T>
{
public:
/// Create an identity operator of size @a n.
explicit IdentityOperator(int n) : Operator(n) { }
explicit IdentityOperatorMP(int n) : OperatorMP<T>(n) { }
/// Operator application
void Mult(const Vector &x, Vector &y) const override { y = x; }
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override { y = x; }
/// Application of the transpose
void MultTranspose(const Vector &x, Vector &y) const override { y = x; }
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override { y = x; }
};
using IdentityOperator = IdentityOperatorMP<real_t>;
/// Returns true if P is the identity prolongation, i.e. if it is either NULL or
/// an IdentityOperator.
inline bool IsIdentityProlongation(const Operator *P)
template <class T>
inline bool IsIdentityProlongation(const OperatorMP<T> *P)
{
return !P || dynamic_cast<const IdentityOperator*>(P);
return !P || dynamic_cast<const IdentityOperatorMP<T>*>(P);
}
/// Scaled Operator B: x -> a A(x).
@@ -855,29 +875,32 @@ public:
/** @brief The transpose of a given operator. Switches the roles of the methods
Mult() and MultTranspose(). */
class TransposeOperator : public Operator
template <class T>
class TransposeOperatorMP : public OperatorMP<T>
{
private:
const Operator &A;
const OperatorMP<T> &A;
public:
/// Construct the transpose of a given operator @a *a.
TransposeOperator(const Operator *a)
: Operator(a->Width(), a->Height()), A(*a) { }
TransposeOperatorMP(const OperatorMP<T> *a)
: OperatorMP<T>(a->Width(), a->Height()), A(*a) { }
/// Construct the transpose of a given operator @a a.
TransposeOperator(const Operator &a)
: Operator(a.Width(), a.Height()), A(a) { }
TransposeOperatorMP(const OperatorMP<T> &a)
: OperatorMP<T>(a.Width(), a.Height()), A(a) { }
/// Operator application. Apply the transpose of the original Operator.
void Mult(const Vector &x, Vector &y) const override
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override
{ A.MultTranspose(x, y); }
/// Application of the transpose. Apply the original Operator.
void MultTranspose(const Vector &x, Vector &y) const override
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override
{ A.Mult(x, y); }
};
using TransposeOperator = TransposeOperatorMP<real_t>;
/// General linear combination operator: x -> a A(x) + b B(x).
class SumOperator : public Operator
{
@@ -902,48 +925,53 @@ public:
};
/// General product operator: x -> (A*B)(x) = A(B(x)).
class ProductOperator : public Operator
template <class T>
class ProductOperatorMP : public OperatorMP<T>
{
const Operator *A, *B;
const OperatorMP<T> *A, *B;
bool ownA, ownB;
mutable Vector z;
mutable VectorMP<T> z;
public:
ProductOperator(const Operator *A, const Operator *B, bool ownA, bool ownB);
ProductOperatorMP(const OperatorMP<T> *A, const OperatorMP<T> *B, bool ownA,
bool ownB);
void Mult(const Vector &x, Vector &y) const override
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override
{ B->Mult(x, z); A->Mult(z, y); }
void MultTranspose(const Vector &x, Vector &y) const override
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override
{ A->MultTranspose(x, z); B->MultTranspose(z, y); }
virtual ~ProductOperator();
virtual ~ProductOperatorMP<T>();
};
using ProductOperator = ProductOperatorMP<real_t>;
/// The operator x -> R*A*P*x constructed through the actions of R^T, A and P
class RAPOperator : public Operator
template <class T>
class RAPOperatorMP : public OperatorMP<T>
{
private:
const Operator & Rt;
const Operator & A;
const Operator & P;
mutable Vector Px;
mutable Vector APx;
const OperatorMP<T> & Rt;
const OperatorMP<T> & A;
const OperatorMP<T> & P;
mutable VectorMP<T> Px;
mutable VectorMP<T> APx;
MemoryClass mem_class;
public:
/// Construct the RAP operator given R^T, A and P.
RAPOperator(const Operator &Rt_, const Operator &A_, const Operator &P_);
RAPOperatorMP<T>(const OperatorMP<T> &Rt_, const OperatorMP<T> &A_,
const OperatorMP<T> &P_);
MemoryClass GetMemoryClass() const override { return mem_class; }
/// Operator application.
void Mult(const Vector & x, Vector & y) const override
void Mult(const VectorMP<T> & x, VectorMP<T> & y) const override
{ P.Mult(x, Px); A.Mult(Px, APx); Rt.MultTranspose(APx, y); }
/// Operator-wise absolute-value application.
void AbsMult(const Vector & x, Vector & y) const override
void AbsMult(const VectorMP<T> & x, VectorMP<T> & y) const override
{ P.AbsMult(x, Px); A.AbsMult(Px, APx); Rt.AbsMultTranspose(APx, y); }
/// Approximate diagonal of the RAP Operator.
@@ -953,7 +981,7 @@ public:
When P is the FE space prolongation operator on a mesh without hanging
nodes and Rt = P, the returned diagonal is exact, as long as the diagonal
of A is also exact. */
void AssembleDiagonal(Vector &diag) const override
void AssembleDiagonal(VectorMP<T> &diag) const override
{
A.AssembleDiagonal(APx);
P.MultTranspose(APx, diag);
@@ -964,11 +992,11 @@ public:
}
/// Application of the transpose.
void MultTranspose(const Vector & x, Vector & y) const override
void MultTranspose(const VectorMP<T> & x, VectorMP<T> & y) const override
{ Rt.Mult(x, APx); A.MultTranspose(APx, Px); P.MultTranspose(Px, y); }
/// Operator-wise absolute-value application of the transpose
void AbsMultTranspose(const Vector & x, Vector & y) const override
void AbsMultTranspose(const VectorMP<T> & x, VectorMP<T> & y) const override
{
Rt.AbsMult(x, APx);
A.AbsMultTranspose(APx, Px);
@@ -976,32 +1004,35 @@ public:
}
};
using RAPOperator = RAPOperatorMP<real_t>;
/// General triple product operator x -> A*B*C*x, with ownership of the factors.
class TripleProductOperator : public Operator
template <class T>
class TripleProductOperatorMP : public OperatorMP<T>
{
const Operator *A;
const Operator *B;
const Operator *C;
const OperatorMP<T> *A;
const OperatorMP<T> *B;
const OperatorMP<T> *C;
bool ownA, ownB, ownC;
mutable Vector t1, t2;
mutable VectorMP<T> t1, t2;
MemoryClass mem_class;
public:
TripleProductOperator(const Operator *A, const Operator *B,
const Operator *C, bool ownA, bool ownB, bool ownC);
TripleProductOperatorMP(const OperatorMP<T> *A, const OperatorMP<T> *B,
const OperatorMP<T> *C, bool ownA, bool ownB, bool ownC);
MemoryClass GetMemoryClass() const override { return mem_class; }
void Mult(const Vector &x, Vector &y) const override
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override
{ C->Mult(x, t1); B->Mult(t1, t2); A->Mult(t2, y); }
void MultTranspose(const Vector &x, Vector &y) const override
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override
{ A->MultTranspose(x, t2); B->MultTranspose(t2, t1); C->MultTranspose(t1, y); }
virtual ~TripleProductOperator();
virtual ~TripleProductOperatorMP<T>();
};
using TripleProductOperator = TripleProductOperatorMP<real_t>;
/** @brief Square Operator for imposing essential boundary conditions using only
the action, Mult(), of a given unconstrained Operator.
@@ -1012,13 +1043,19 @@ public:
Do not confuse with ConstrainedSolver, which despite the name has very
different functionality. */
class ConstrainedOperator : public Operator
template <class T>
class ConstrainedOperatorMP : public OperatorMP<T>
{
using DiagonalPolicy = OperatorBase::DiagonalPolicy;
using OperatorBase::DIAG_ONE;
using OperatorBase::DIAG_KEEP;
using OperatorBase::DIAG_ZERO;
protected:
Array<int> constraint_list; ///< List of constrained indices/dofs.
Operator *A; ///< The unconstrained Operator.
OperatorMP<T> *A; ///< The unconstrained Operator.
bool own_A; ///< Ownership flag for A.
mutable Vector z, w; ///< Auxiliary vectors.
mutable VectorMP<T> z, w; ///< Auxiliary vectors.
MemoryClass mem_class;
DiagonalPolicy diag_policy; ///< Diagonal policy for constrained dofs
@@ -1031,8 +1068,9 @@ public:
ownership flag @a own_A is true, the operator @a *A will be destroyed
when this object is destroyed. The @a diag_policy determines how the
operator sets entries corresponding to essential dofs. */
ConstrainedOperator(Operator *A, const Array<int> &list, bool own_A = false,
DiagonalPolicy diag_policy = DIAG_ONE);
ConstrainedOperatorMP(OperatorMP<T> *A, const Array<int> &list,
bool own_A = false,
DiagonalPolicy diag_policy = DIAG_ONE);
/// Returns the type of memory in which the solution and temporaries are stored.
MemoryClass GetMemoryClass() const override { return mem_class; }
@@ -1042,7 +1080,7 @@ public:
{ diag_policy = diag_policy_; }
/// Diagonal of A, modified according to the used DiagonalPolicy.
void AssembleDiagonal(Vector &diag) const override;
void AssembleDiagonal(VectorMP<T> &diag) const override;
/** @brief Eliminate "essential boundary condition" values specified in @a x
from the given right-hand side @a b.
@@ -1055,7 +1093,7 @@ public:
the vectors, and "_i" -- the rest of the entries.
@note This method is consistent with `DiagonalPolicy::DIAG_ONE`. */
void EliminateRHS(const Vector &x, Vector &b) const;
void EliminateRHS(const VectorMP<T> &x, VectorMP<T> &b) const;
/** @brief Constrained operator action.
@@ -1065,29 +1103,33 @@ public:
where the "_b" subscripts denote the essential (boundary) indices/dofs of
the vectors, and "_i" -- the rest of the entries. */
void Mult(const Vector &x, Vector &y) const override;
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override;
void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const override;
void AddMult(const VectorMP<T> &x, VectorMP<T> &y,
const T a = 1.0) const override;
void AbsMult(const Vector &x, Vector &y) const override;
void AbsMult(const VectorMP<T> &x, VectorMP<T> &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override;
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override;
void AbsMultTranspose(const Vector &x, Vector &y) const override;
void AbsMultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override;
/** @brief Implementation of Mult or MultTranspose.
TODO - Generalize to allow constraining rows and columns differently. */
void ConstrainedMult(const Vector &x, Vector &y, const bool transpose) const;
* TODO - Generalize to allow constraining rows and columns differently. */
void ConstrainedMult(const VectorMP<T> &x, VectorMP<T> &y,
const bool transpose) const;
/** @brief Implementation of AbsMult or AbsMultTranspose.
TODO - Generalize to allow constraining rows and columns differently. */
void ConstrainedAbsMult(const Vector &x, Vector &y,
void ConstrainedAbsMult(const VectorMP<T> &x, VectorMP<T> &y,
const bool transpose) const;
/// Destructor: destroys the unconstrained Operator, if owned.
~ConstrainedOperator() override { if (own_A) { delete A; } }
~ConstrainedOperatorMP<T>() override { if (own_A) { delete A; } }
};
using ConstrainedOperator = ConstrainedOperatorMP<real_t>;
/** @brief Rectangular Operator for imposing essential boundary conditions on
the input space using only the action, Mult(), of a given unconstrained
Operator.
@@ -1095,13 +1137,14 @@ public:
Rectangular operator constrained by fixing certain entries in the solution
to given "essential boundary condition" values. This class is used by the
general matrix-free formulation of Operator::FormRectangularLinearSystem. */
class RectangularConstrainedOperator : public Operator
template <class T>
class RectangularConstrainedOperatorMP : public OperatorMP<T>
{
protected:
Array<int> trial_constraints, test_constraints;
Operator *A;
OperatorMP<T> *A;
bool own_A;
mutable Vector z, w;
mutable VectorMP<T> z, w;
MemoryClass mem_class;
public:
@@ -1112,8 +1155,8 @@ public:
constrain, i.e. each entry @a trial_list[i] represents an essential trial
dof. If the ownership flag @a own_A is true, the operator @a *A will be
destroyed when this object is destroyed. */
RectangularConstrainedOperator(Operator *A, const Array<int> &trial_list,
const Array<int> &test_list, bool own_A = false);
RectangularConstrainedOperatorMP(OperatorMP<T> *A, const Array<int> &trial_list,
const Array<int> &test_list, bool own_A = false);
/// Returns the type of memory in which the solution and temporaries are stored.
MemoryClass GetMemoryClass() const override { return mem_class; }
/** @brief Eliminate columns corresponding to "essential boundary condition"
@@ -1126,7 +1169,7 @@ public:
where the "_b" subscripts denote the essential (boundary) indices and the
"_j" subscript denotes the essential test indices */
void EliminateRHS(const Vector &x, Vector &b) const;
void EliminateRHS(const VectorMP<T> &x, VectorMP<T> &b) const;
/** @brief Rectangular-constrained operator action.
Performs the following steps:
@@ -1136,16 +1179,19 @@ public:
where the "_i" subscripts denote all the nonessential (boundary) trial
indices and the "_j" subscript denotes the essential test indices */
void Mult(const Vector &x, Vector &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override;
virtual ~RectangularConstrainedOperator() { if (own_A) { delete A; } }
void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override;
void MultTranspose(const VectorMP<T> &x, VectorMP<T> &y) const override;
virtual ~RectangularConstrainedOperatorMP<T>() { if (own_A) { delete A; } }
};
using RectangularConstrainedOperator = RectangularConstrainedOperatorMP<real_t>;
/** @brief Abstract class for defining inner products. The method Eval()
must be implemented in derived classes to compute the inner product
of two vectors according to a specific inner product definition.
*/
class InnerProductOperator : public Operator
template <class T>
class InnerProductOperatorMP : public OperatorMP<T>
{
#ifdef MFEM_USE_MPI
private:
@@ -1153,16 +1199,16 @@ private:
int dot_prod_type = 0; // 0: local, 1: global
public:
InnerProductOperator(MPI_Comm comm_) : Operator(1)
InnerProductOperatorMP(MPI_Comm comm_) : OperatorMP<T>(1)
{ comm = comm_; dot_prod_type = 1; }
#endif
protected:
/// @brief Standard global/local $\ell_2$ inner product.
virtual real_t Dot(const Vector &x, const Vector &y) const;
virtual T Dot(const VectorMP<T> &x, const VectorMP<T> &y) const;
public:
/// Create an operator of size 1 (scalar).
InnerProductOperator() : Operator(1)
InnerProductOperatorMP() : OperatorMP<T>(1)
{
#ifdef MFEM_USE_MPI
dot_prod_type = 0;
@@ -1171,7 +1217,7 @@ public:
/// Operator application - not always needed/used but added
/// to satisfy the abstract base class interface.
virtual void Mult(const Vector &x, Vector &y) const override
virtual void Mult(const VectorMP<T> &x, VectorMP<T> &y) const override
{
MFEM_ABORT("Mult is not implemented.");
}
@@ -1179,9 +1225,11 @@ public:
/** @brief Compute the inner product (x,y) of vectors x and y.
This is an abstract method that must be
implemented in derived classes. */
virtual real_t Eval(const Vector &x, const Vector &y) = 0;
virtual real_t Eval(const VectorMP<T> &x, const VectorMP<T> &y) = 0;
};
using InnerProductOperator = InnerProductOperatorMP<real_t>;
/** @brief PowerMethod helper class to estimate the largest eigenvalue of an
operator using the iterative power method. */
class PowerMethod
+75
View File
@@ -0,0 +1,75 @@
#include "ordering.hpp"
namespace mfem
{
template <>
void Ordering::DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = 1; vd < vdim; vd++)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byNODES>(ndofs, vdim, dofs[i], vd);
}
}
}
template <>
void Ordering::DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = vdim-1; vd >= 0; vd--)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byVDIM>(ndofs, vdim, dofs[i], vd);
}
}
}
void Ordering::Reorder(Vector &v, int vdim, Ordering::Type in_ord,
Ordering::Type out_ord)
{
if (in_ord == out_ord)
{
return;
}
int nvdofs = v.Size();
int nldofs = nvdofs/vdim;
if (out_ord == Ordering::byNODES) // byVDIM -> byNODES
{
Vector temp = v;
for (int d = 0; d < vdim; d++)
{
int off = d * nldofs;
for (int i = 0; i < nldofs; i++)
{
v[i + off] = temp[Map<byVDIM>(nldofs,vdim,i,d)];
}
}
}
else // byNODES -> byVDIM
{
Vector temp = v;
for (int i = 0; i < nldofs; i++)
{
int off = i*vdim;
for (int d = 0; d < vdim; d++)
{
v[d + off] = temp[Map<byNODES>(nldofs,vdim,i,d)];
}
}
}
}
} // namespace mfem
+53
View File
@@ -0,0 +1,53 @@
#ifndef MFEM_ORDERING
#define MFEM_ORDERING
#include "../config/config.hpp"
#include "vector.hpp"
namespace mfem
{
/** @brief The ordering method used when the number of unknowns per mesh node
(vector dimension) is bigger than 1. */
class Ordering
{
public:
/// %Ordering methods:
enum Type
{
byNODES, /**< loop first over the nodes (inner loop) then over the vector
dimension (outer loop); symbolically it can be represented
as: XXX...,YYY...,ZZZ... */
byVDIM /**< loop first over the vector dimension (inner loop) then over
the nodes (outer loop); symbolically it can be represented
as: XYZ,XYZ,XYZ,... */
};
template <Type Ord>
static inline int Map(int ndofs, int vdim, int dof, int vd);
template <Type Ord>
static void DofsToVDofs(int ndofs, int vdim, Array<int> &dofs);
/// Reorder Vector \p v from its current ordering \p in_ord to \p out_ord
static void Reorder(Vector &v, int vdim, Type in_ord, Type out_ord);
};
template <> inline int
Ordering::Map<Ordering::byNODES>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? dof+ndofs*vd : dof-ndofs*vd;
}
template <> inline int
Ordering::Map<Ordering::byVDIM>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? vd+vdim*dof : -1-(vd+vdim*(-1-dof));
}
} // namespace mfem
#endif // MFEM_ORDERING
+313
View File
@@ -0,0 +1,313 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "particlevector.hpp"
namespace mfem
{
void ParticleVector::GrowSize(int min_num_vectors, bool keep_data)
{
const int nsize = std::max(min_num_vectors*vdim, 2 * data.Capacity());
Memory<real_t> p(nsize, data.GetMemoryType());
if (keep_data) { p.CopyFrom(data, size); }
p.UseDevice(data.UseDevice());
data.Delete();
data = p;
}
ParticleVector::ParticleVector(int vdim_, Ordering::Type ordering_)
: ParticleVector(vdim_, ordering_, 0) { }
ParticleVector::ParticleVector(int vdim_, Ordering::Type ordering_,
int num_nodes)
: Vector(num_nodes*vdim_), vdim(vdim_), ordering(ordering_)
{
Vector::operator=(0.0);
}
ParticleVector::ParticleVector(int vdim_, Ordering::Type ordering_,
const Vector &vec)
: Vector(vec), vdim(vdim_), ordering(ordering_)
{
MFEM_ASSERT(vec.Size() % vdim == 0,
"Incompatible Vector size of " << vec.Size() << " given vdim " << vdim);
}
void ParticleVector::GetValues(int i, Vector &nvals) const
{
nvals.SetSize(vdim);
if (ordering == Ordering::byNODES)
{
int nv = GetNumParticles();
for (int c = 0; c < vdim; c++)
{
nvals[c] = Vector::operator[](i+nv*c);
}
}
else
{
for (int c = 0; c < vdim; c++)
{
nvals[c] = Vector::operator[](c+vdim*i);
}
}
}
void ParticleVector::GetValuesRef(int i, Vector &nref)
{
MFEM_ASSERT(ordering == Ordering::byVDIM,
"GetValuesRef only valid when ordering byVDIM.");
nref.MakeRef(*this, i*vdim, vdim);
}
void ParticleVector::GetComponents(int vd, Vector &comp)
{
int vdim_temp = vdim;
// For byNODES: Treat each component as a vector temporarily
// For byVDIM: Treat each vector as a component temporarily
vdim = GetNumParticles();
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
GetValues(vd, comp);
// Reset ordering back to original
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
vdim = vdim_temp;
}
void ParticleVector::GetComponentsRef(int vd, Vector &nref)
{
MFEM_ASSERT(ordering == Ordering::byNODES,
"GetComponentsRef only valid when ordering byNODES.");
nref.MakeRef(*this, vd*GetNumParticles(), GetNumParticles());
}
void ParticleVector::SetValues(int i, const Vector &nvals)
{
if (ordering == Ordering::byNODES)
{
int nv = GetNumParticles();
for (int c = 0; c < vdim; c++)
{
Vector::operator[](i + c*nv) = nvals[c];
}
}
else
{
for (int c = 0; c < vdim; c++)
{
Vector::operator[](c + i*vdim) = nvals[c];
}
}
}
void ParticleVector::SetComponents(int vd, const Vector &comp)
{
int vdim_temp = vdim;
// For byNODES: Treat each component as a vector temporarily
// For byVDIM: Treat each vector as a component temporarily
vdim = GetNumParticles();
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
SetValues(vd, comp);
// Reset ordering back to original
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
vdim = vdim_temp;
}
real_t& ParticleVector::operator()(int i, int comp)
{
MFEM_ASSERT(i < GetNumParticles(),
"Particle index " << i <<
" is invalid for number of particles " << GetNumParticles());
MFEM_ASSERT(comp < vdim,
"Component index " << comp <<
" is invalid for vector dimension " << vdim);
if (ordering == Ordering::byNODES)
{
return Vector::operator[](i + comp*GetNumParticles());
}
else
{
return Vector::operator[](comp + i*vdim);
}
}
const real_t& ParticleVector::operator()(int i, int comp) const
{
MFEM_ASSERT(i < GetNumParticles(),
"Particle index " << i <<
" is invalid for number of particles " << GetNumParticles());
MFEM_ASSERT(comp < vdim,
"Component index " << comp <<
" is invalid for vector dimension " << vdim);
if (ordering == Ordering::byNODES)
{
return Vector::operator[](i + comp*GetNumParticles());
}
else
{
return Vector::operator[](comp + i*vdim);
}
}
void ParticleVector::DeleteParticles(const Array<int> &indices)
{
if (indices.Size() == 0) { return; }
// Convert list index array of "ldofs" to "vdofs"
Array<int> v_list;
v_list.Reserve(indices.Size()*vdim);
MFEM_VERIFY(indices.Max() < GetNumParticles(),
"Particle index " << indices.Max() <<
" is out-of-range for number of particles " <<
GetNumParticles());
if (ordering == Ordering::byNODES)
{
for (int l = 0; l < indices.Size(); l++)
{
for (int vd = 0; vd < vdim; vd++)
{
v_list.Append(Ordering::Map<Ordering::byNODES>(GetNumParticles(),
vdim,
indices[l], vd));
}
}
}
else
{
for (int l = 0; l < indices.Size(); l++)
{
for (int vd = 0; vd < vdim; vd++)
{
v_list.Append(Ordering::Map<Ordering::byVDIM>(GetNumParticles(),
vdim,
indices[l],
vd));
}
}
}
Vector::DeleteAt(v_list);
}
void ParticleVector::SetVDim(int vdim_, bool keep_data)
{
if (!keep_data)
{
int num_particles = GetNumParticles();
vdim = vdim_;
Vector::SetSize(num_particles*vdim_);
return;
}
// Reorder/shift existing entries
// For byNODES: Treat each component as a vector temporarily
// For byVDIM: Treat each vector as a component temporarily
vdim = GetNumParticles();
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
SetNumParticles(vdim_, keep_data);
// Reset ordering back to original
ordering = ordering == Ordering::byNODES ? Ordering::byVDIM :
Ordering::byNODES;
vdim = vdim_;
}
void ParticleVector::SetOrdering(Ordering::Type ordering_, bool keep_data)
{
if (keep_data)
{
Ordering::Reorder(*this, vdim, ordering, ordering_);
}
ordering = ordering_;
}
void ParticleVector::SetNumParticles(int num_vectors, bool keep_data)
{
int old_nv = GetNumParticles();
if (num_vectors == old_nv)
{
return;
}
// If resizing larger...
if (num_vectors > old_nv)
{
// Increase capacity if needed
if (num_vectors*vdim > Vector::Capacity())
{
GrowSize(num_vectors, keep_data);
}
// Set larger new size
Vector::SetSize(num_vectors*vdim);
if (!keep_data) { return; }
if (ordering == Ordering::byNODES)
{
// Shift entries for byNODES
for (int c = vdim-1; c > 0; c--)
{
for (int i = old_nv-1; i >= 0; i--)
{
Vector::operator[](i+c*num_vectors) = Vector::operator[](i+c*old_nv);
}
}
// Zero-out data now associated with new Vectors
for (int c = 0; c < vdim; c++)
{
for (int i = old_nv; i < num_vectors; i++)
{
Vector::operator[](i+c*num_vectors) = 0.0;
}
}
}
else // byVDIM
{
for (int i = old_nv*vdim; i < num_vectors*vdim; i++)
{
data[i] = 0.0;
}
}
}
else // Else just remove the trailing vector data
{
if (!keep_data) { Vector::SetSize(num_vectors*vdim); return; }
Array<int> rm_indices(old_nv-num_vectors);
for (int i = 0; i < rm_indices.Size(); i++)
{
rm_indices[i] = old_nv - rm_indices.Size() + i;
}
DeleteParticles(rm_indices);
}
}
} // namespace mfem

Some files were not shown because too many files have changed in this diff Show More