Compare commits

...
Author SHA1 Message Date
Victor DeCaria d31dfa15fa fix missing std::array and std::ignore 2024-11-12 06:36:03 -07:00
Julian Andrej e94b8b6c89 cleanup 2024-11-11 09:53:37 -08:00
Julian Andrej 3f8348d04b update 2024-11-08 10:03:21 -08:00
Julian Andrej e30f7e5aa9 the big refactor 2024-11-07 14:31:30 -08:00
Julian Andrej 047943cfda missing file 2024-10-31 14:50:32 -07:00
Julian Andrej 4e6c38cf6d add qfunction_dual 2024-10-31 14:47:45 -07:00
Julian Andrej c981633846 move integration rule to element operator 2024-10-29 09:34:01 -07:00
Julian Andrej ac2a5107da update nonlinear test 2024-10-28 15:00:30 -07:00
Julian Andrej e6d733aa95 native ad 2024-10-28 14:53:26 -07:00
Julian Andrej c6ee709eef directly pass through quadrature point data 2024-10-28 09:57:49 -07:00
Julian Andrej 0274b67ff6 demo updates 2024-10-28 08:30:45 -07:00
Julian Andrej 38da503958 refactors 2024-10-18 15:14:46 -07:00
Julian Andrej 6c4f9c69a9 refactor 2024-10-18 11:07:07 -07:00
Julian Andrej 0702eb2fb2 refactor 2024-10-18 10:01:41 -07:00
Julian Andrej 7a62a7fd4a add more tests 2024-10-15 16:13:04 -07:00
Julian Andrej 0f94d484a4 more updates 2024-10-15 12:44:49 -07:00
Julian Andrej 4b5798f905 updateees 2024-10-15 12:44:20 -07:00
Julian Andrej 87240f5619 reorder loops 2024-10-11 11:16:50 -07:00
Julian Andrej 3ccaa48cd4 fixes 2024-10-10 11:09:04 -07:00
Julian Andrej a08928dd97 benchmark 2024-10-09 13:11:03 -07:00
Julian Andrej a0694d5825 tweaks 2024-10-09 12:39:57 -07:00
Julian Andrej a3ecbef0ec partial assembly test for 3d diffusion 2024-10-09 09:16:05 -07:00
Julian Andrej db5bc1725e three dee 2024-10-08 07:19:57 -07:00
Julian Andrej 6d50ebc9c3 derpderp 2024-09-18 15:53:22 -07:00
Julian Andrej e713913177 derp 2024-09-18 15:41:36 -07:00
Julian Andrej dd9898af3b SYNC ALL THE SYNCS 2024-09-18 13:10:44 -07:00
Julian Andrej 0435ef5dac laghos progress 2024-09-17 15:02:56 -07:00
Julian Andrej 79e5020a18 device 2024-09-13 20:54:37 -07:00
Julian Andrej adc81c7f5d hd annot 2024-09-13 20:42:20 -07:00
Julian Andrej 18676c61b7 HD annotation 2024-09-13 20:39:29 -07:00
Julian Andrej 8f8deab121 add missing examples 2024-09-13 20:33:45 -07:00
Julian Andrej 8b0c779320 get laghos example to work 2024-09-13 19:32:18 -07:00
Julian Andrej a012769434 reintroduce derivatives 2024-09-11 16:29:34 -07:00
Julian Andrej ce5517b9af remove old dfem header 2024-09-11 16:25:43 -07:00
Julian Andrej 70ae37d5f0 sync 2024-08-28 16:23:59 -07:00
Julian Andrej feac718e95 Merge branch 'master' into dfem-coefficient
# Conflicts:
#	CMakeLists.txt
2024-08-26 15:19:10 -07:00
Tzanio Kolev ac2a21516c Merge pull request #4447 from mfem/det-d1d-q1d-fix
Fix switched D1D and Q1D in determinant kernels
2024-08-25 15:22:05 -07:00
Tzanio Kolev 8ba104788f Merge pull request #4444 from mfem/small_doc_update
update the documentation of two methods in fespace
2024-08-25 15:21:49 -07:00
Tzanio Kolev a454a5407c Merge pull request #4464 from mfem/najlkin/fix-point-attr
Fixed Point default attribute.
2024-08-25 15:21:23 -07:00
Jan Nikl 689b46e3d6 Removed the workaround for 1D in the dof-to-arrays test. 2024-08-23 10:14:03 -07:00
Jan Nikl 85ccdf210a Fixed Point default attribute. 2024-08-22 12:53:33 -07:00
Julian Andrej 9be8c15cf8 performance updates 2024-08-22 07:43:21 -07:00
Julian Andrej 08ba45fca3 more shmemenigans 2024-08-19 13:15:31 -07:00
Julian Andrej 3aacfbfab0 threaded loops 2024-08-19 11:13:38 -07:00
Julian Andrej 2a60b998c7 more shmem shenan 2024-08-19 10:59:56 -07:00
Julian Andrej 91a168929f maybe 2024-08-16 08:12:48 -07:00
Julian Andrej 1369f5e189 still bugs 2024-08-16 07:33:29 -07:00
Julian Andrej 993e4fbbe1 buuugs 2024-08-15 10:58:37 -07:00
Julian Andrej 07d8a17abe simplification 2024-08-15 09:38:23 -07:00
Julian Andrej 5531b82dbc buugs 2024-08-15 07:46:21 -07:00
Julian Andrej d9f60f401b shmem info doc 2024-08-15 07:41:58 -07:00
Julian Andrej fb876ba3a1 shared memory bug 2024-08-15 07:41:37 -07:00
Julian Andrej 32a94b438f more shared memory 2024-08-15 07:37:02 -07:00
Julian Andrej 917978d310 refactor for tensor product elements 2024-08-14 13:10:35 -07:00
Tzanio Kolev 5f373e8c2a Merge pull request #4389 from mfem/batched-linalg
Batched linear algebra with GPU backends
2024-08-14 11:11:43 -07:00
Tzanio Kolev 0fdb8a2709 Merge pull request #4443 from mfem/fix-cuda-warnings
Fix some nvcc warnings
2024-08-13 19:06:31 -07:00
Will Pazner 0303669e9a Fix switched D1D and Q1D in determinant kernels 2024-08-13 16:03:36 -07:00
Will Pazner 37d19f99de Merge pull request #4421 from mfem/windows-getaddrinfo-fix
Fix `getaddrinfo` on Windows with MS VC++
2024-08-13 16:02:51 -07:00
bslazarov fa10d89676 update the documentation of two methods in fesapce 2024-08-12 13:20:23 -07:00
Will Pazner 7534d86172 Fix some nvcc warnings 2024-08-12 13:13:31 -07:00
Will Pazner dc13eac9e6 Remove include of cublas_v2.h from cpp file 2024-08-12 10:32:49 -07:00
Will Pazner 993f478c54 Use cublas_v2.h header instead of cublas.h 2024-08-12 09:52:03 -07:00
Tzanio Kolev cdf9cfe1b6 Merge pull request #4440 from mfem/tmop-single-fix
Fix warning with TMOP in single precision
2024-08-10 12:23:15 -07:00
Tzanio Kolev 4e30030ab7 Merge pull request #4405 from mfem/paraview-component-name
Add component labels to ParaView VTU output
2024-08-10 12:22:44 -07:00
Will Pazner 22878ee681 Merge remote-tracking branch 'origin/master' into batched-linalg
# Conflicts:
#	CHANGELOG
2024-08-09 13:12:02 -07:00
Will Pazner aba85bf679 Update CHANGELOG
Add new entry for batched linear algebra, and categorize existing v4.7.1 entries
2024-08-09 13:10:26 -07:00
Will Pazner 3a59281601 Change batched linear algebra backend terminology
The "preferred backend" is now called the "active backend"
2024-08-09 12:24:39 -07:00
Will Pazner 2360809938 Factor multiplication out of inner loop in kernels::AddMult 2024-08-09 11:36:28 -07:00
Will Pazner 69b26cb722 Replace nullptr_t with std::nullptr_t
Also make sure to include the <cstddef> header.
2024-08-09 11:06:12 -07:00
Will Pazner f734b1bd92 Include gpu_blas.hpp from linalg.hpp
Make GPUBlas class work when compiling without CUDA or HIP; in this case, it has
no effect (and the handle is just nullptr).
2024-08-09 11:03:33 -07:00
Will Pazner 14c31edb4b Fix device backend check when enabling GPU BLAS 2024-08-09 10:54:10 -07:00
Will Pazner bbc9af2619 Use real_t instead of double in TMOP_Metric_000 2024-08-09 10:51:08 -07:00
Will Pazner 2a1cce4663 Include 'batched/solver.hpp' in 'linalg.hpp' 2024-08-08 14:28:27 -07:00
Will Pazner 80399de548 Doxygen comment for BatchedDirectSolver::SetOperator 2024-08-08 14:07:46 -07:00
Will Pazner 473bd4177e const correctness in BatchedDirectSolver::BatchedDirectSolver 2024-08-08 14:07:14 -07:00
Will Pazner 184c3cbbb8 Merge pull request #4426 from mfem/move-nodes-update-dev
`MoveNodes` calls `NodesUpdated`
2024-08-06 11:48:39 -07:00
Veselin Dobrev 29730813cb Merge pull request #4342 from mfem/amd-use-hypre-spmv
Better HYPRE SpMV
2024-08-06 11:47:37 -07:00
Will Pazner cbbfaede8c More adjustments for MAGMA with CMake build system 2024-07-30 14:04:17 -07:00
Will Pazner bb8a5b88da Adjustments for MAGMA with CMake build system 2024-07-30 10:18:03 -07:00
Tzanio Kolev 057a5a43b0 Merge pull request #4082 from mfem/najlkin/mixed-face-forms
Boundary/face integration in (Mixed)BilinearForm and Hybridization
2024-07-30 08:38:13 -07:00
Victor DeCaria 867a26ae4b MoveNodes calls NodesUpdated, and therefore updates nodes_sequence 2024-07-29 14:14:51 -06:00
Julian Andrej a62302b4cb make input qp memory thread safe 2024-07-29 12:28:00 -07:00
Tzanio Kolev acc8ba9df3 Merge pull request #4088 from mfem/najlkin/integral-els
Support for integral finite elements at multiple places
2024-07-28 15:38:54 -07:00
Tzanio Kolev 73ee69da91 Merge pull request #4414 from mfem/najlkin/fix-mesh-refine
[BUG] Fixed initialization of embedding geometry
2024-07-28 15:38:21 -07:00
Tzanio Kolev bfb3eb786b Merge pull request #4178 from mfem/cubit-pyramid-wedge-support-dev
Extend ReadCubit to Add Support For First/Second Order Pyramid and Wedge Elements and Mixed Meshes
2024-07-28 15:37:12 -07:00
Julian Andrej f5d2b82839 add device config to tests 2024-07-26 15:13:23 -07:00
Julian Andrej 1382f8aa1f more gpu compat 2024-07-26 14:58:08 -07:00
Julian Andrej f76a884d15 more device sanitizing 2024-07-26 14:00:52 -07:00
Julian Andrej 5c56659e46 add tuple impl 2024-07-26 12:59:09 -07:00
Julian Andrej 4bcd4586ba add serac::tuple 2024-07-26 12:51:00 -07:00
Julian Andrej 083d42c6ce try other initializer 2024-07-26 11:37:33 -07:00
Julian Andrej dbc3458db0 MFEM_HOST_DEVICE 2024-07-26 11:33:43 -07:00
Julian Andrej 4f694287ae host device annotations 2024-07-26 11:32:33 -07:00
Julian Andrej 5d8fbfee93 clean up use of Vector for device prep 2024-07-26 11:21:57 -07:00
Will Pazner bc729465d6 Build system fixes for MAGMA 2024-07-25 20:22:42 -07:00
Will Pazner c690c25058 Move #include inside #ifdef 2024-07-25 18:50:45 -07:00
Will Pazner 4f629c150d Revert changes to tests/unit/makefile 2024-07-25 18:49:41 -07:00
Will Pazner 7b35f3cfde Add MAGMA instructions to INSTALL 2024-07-25 18:49:27 -07:00
Will Pazner 9d78e8cc23 Revert some changes to makefile 2024-07-25 18:49:27 -07:00
Veselin Dobrev 1c252d79d8 Merge branch 'master' into batched-linalg 2024-07-25 17:03:09 -07:00
Julian Andrej 7799753053 forall capture 2024-07-24 12:48:51 -07:00
Veselin Dobrev fe8562e6e3 In general/socketstream.[ch]pp, send error messages to mfem::err
instead of mfem::out.
2024-07-23 18:01:51 -07:00
Veselin Dobrev ce5b362077 Merge pull request #4377 from mfem/EdwardPalmer99/add-missing-header-to-exodus-writer-fix
Fixes compilation issue when compiling for GPU -- missing header
2024-07-23 16:37:54 -07:00
Veselin Dobrev fd6e0e1659 Merge pull request #4363 from mfem/bugfix/chapman39/slepc-makefile-ordering
Fix SLEPc linking errors missing PETSc symbols when using `make`
2024-07-23 16:36:13 -07:00
Veselin Dobrev 73f84b48f8 On Windows, initialize additional fields in the 'addrinfo' struct
before calling 'getaddrinfo' -- without this the call fails.
2024-07-22 14:22:40 -07:00
Julian Andrej cb6db58ad3 simplify conversion 2024-07-22 10:12:34 -07:00
Julian Andrej 5da2bfc23d cruft 2024-07-22 10:12:16 -07:00
Julian Andrej 466a771ab0 bugfix 2024-07-22 10:11:44 -07:00
Tzanio Kolev 0952809e5e Merge pull request #4252 from farscape-project/makefile
Improve support for out-of-tree installations
2024-07-19 14:12:27 -07:00
Tzanio Kolev b9a1f8689e Merge pull request #4362 from mfem/tmop-fitting-energyfix
Fix energy calculation for surface fitting
2024-07-19 12:35:34 -07:00
Tzanio Kolev b479f97b54 Merge pull request #4407 from mfem/tmop-zero-metric
Zero metric in TMOP
2024-07-19 12:35:04 -07:00
Jan Nikl 5202ab09c5 Initialized embedding geometry. 2024-07-19 08:01:24 -07:00
Tzanio Kolev 5fc313d2f2 Merge pull request #4401 from adam-sim-dev/solvers_verify
Use MFEM_VERIFY instead of MFEM_ASSERT inside the solvers
2024-07-18 16:05:25 +01:00
Tzanio Kolev 299a8d6c14 Merge pull request #4395 from adam-sim-dev/SLISolver_typo
It should be the logical || operator
2024-07-18 16:04:11 +01:00
Tzanio Kolev 5683932a61 Merge pull request #4409 from mfem/gitlab-ci-cleanup-fix
Fix a small bug in Gitlab CI when cleaning up
2024-07-18 02:07:43 +01:00
Veselin Dobrev 2432eefc23 In the Gitlab CI, remove the '--exclusive' flag recently added to
'srun' on Ruby -- it seems to cause slowdown for some unknown reason.

Also, in the 'baseline' gitlab script, account for the case when the
'salloc' command returns an error -- in the current version some
failures were reported as success.
2024-07-17 16:19:58 -07:00
Veselin Dobrev 8030334a97 In the 'ruby-baseline' Gitlab pipeline, do not try to cleanup the
directory ${CI_PROJECT_DIR} -- we don't clone the repo in this step,
so the directory may be empty and that will generate a bogus error.
2024-07-16 18:35:52 -07:00
Tzanio Kolev e13989a293 Merge pull request #4402 from mfem/gitlab-ci-quartz-to-ruby
Switch Gitlab CI on Quartz to Ruby
2024-07-17 01:17:42 +01:00
Tzanio Kolev 7f47e0b7ac Merge branch 'master' into najlkin/mixed-face-forms 2024-07-17 00:44:08 +01:00
Jan Nikl e3ec06fbb8 Update CHANGELOG 2024-07-16 16:42:34 -07:00
Jan Nikl 2dd9c9ca65 Changed int -> bool for the flag for extern bdr constraint integrators. 2024-07-16 15:14:18 -07:00
Tzanio Kolev ab503a3fd0 Update CHANGELOG 2024-07-16 14:57:36 -07:00
Tzanio Kolev 46fdcd696f Update test_exodus_reader.cpp 2024-07-16 14:56:10 -07:00
Tzanio Kolev 9f42d495ca Update CHANGELOG 2024-07-16 14:55:34 -07:00
Jan Nikl 6be7547229 Removed virtual qualifier from VectorFEBoundaryFluxIntegrator. 2024-07-16 14:27:19 -07:00
Mittal, Ketan 63da26c00f double -> real_t 2024-07-16 11:42:07 -07:00
Joseph Signorelli 713f86c134 Fix minor typos for qfunctions labeling 2024-07-16 13:22:41 -05:00
Mittal, Ketan 708c477714 add zero metric 2024-07-16 11:12:35 -07:00
Will Pazner 96382fe2a6 Add component labels to ParaView VTU output
This fixes an inconsistency between the interpretation of components for 3x3
symmetric matrices.

Resolves #4398.
2024-07-15 13:42:50 -07:00
Julian Andrej a317e1a17d more options 2024-07-15 09:01:47 -07:00
Veselin Dobrev 5a38a2e712 In Gitlab CI on Ruby, increase the number of jobs for building and
add --exclusive to allocations.
2024-07-14 21:06:45 -07:00
Veselin Dobrev 5af524009e Switch Gitlab CI on Quartz to Ruby 2024-07-14 20:32:54 -07:00
Nuno Nobre 2137ce1f4f Document new install permission options 2024-07-15 00:10:23 +01:00
Nuno Nobre c32d62f83c Try to be a bit cleverer when looking for config.mk 2024-07-15 00:10:18 +01:00
Nuno NobreandVeselin Dobrev 8aa00234d8 Tweak permission settings for installed files/dirs
Co-authored-by: Veselin Dobrev <dobrev@llnl.gov>
2024-07-15 00:09:10 +01:00
adam-sim-dev fbc4083001 Use MFEM_VERIFY 2024-07-13 08:33:28 +08:00
Julian Andrej fcbde98cb6 working laghos example 2024-07-11 08:04:14 -07:00
adam-sim-dev ffa4f84108 It should be the logical || operator 2024-07-11 16:20:36 +08:00
Tzanio Kolev 988439f60a Merge pull request #4360 from mfem/print-mathematica-dev
Adding SparseMatrix::PrintMathematica member function [print-mathematica-dev]
2024-07-09 22:32:46 +01:00
Will Pazner 09d03b715e Support single precision with batched BLAS 2024-07-08 13:17:08 -07:00
Will Pazner 5398491cd4 Improve Doxygen for BatchedLinAlg 2024-07-08 13:17:08 -07:00
Will Pazner 0ff7174de2 Add BatchedLinAlg::AddMult and related functionality 2024-07-08 13:17:08 -07:00
Will Pazner 0108a83a43 Add hipBLAS to make and cmake builds 2024-07-08 13:17:08 -07:00
Will Pazner 151828f435 Add MFEM_USE_MAGMA to MFEM build configuration 2024-07-08 13:17:08 -07:00
Will Pazner 4d1564ff64 MAGMA implementation for batched linear algebra 2024-07-08 13:17:08 -07:00
Will Pazner ac906c8827 Doxygen for batched direct solver 2024-07-08 13:17:08 -07:00
Will Pazner 1710d10dbe Add batched linear algebra files to CMakeLists.txt 2024-07-08 13:17:08 -07:00
Will Pazner 5292971b7e Add new class BatchedDirectSolver 2024-07-08 13:17:08 -07:00
Will Pazner 9997d1b718 Add BatchedLinAlg::GetPreferredBackend 2024-07-08 13:17:08 -07:00
Will Pazner 967e0f5bac Add (comment-out) alternative implementation of NativeBatchedLinAlg::Mult 2024-07-08 13:17:08 -07:00
Will Pazner 194debfcc0 Test batched linear algebra with multiple right-hand sides 2024-07-08 13:17:08 -07:00
Will Pazner eae459cb15 Improve DenseTensor LinearSolve methods unit tests 2024-07-08 13:17:08 -07:00
Will Pazner 580578aa04 Add BatchedLinAlg::IsAvailable 2024-07-08 13:17:08 -07:00
Will Pazner 82ad111db5 Implement GPUBlasBatchedLinAlg::Mult 2024-07-08 13:17:08 -07:00
Will Pazner f575e45593 Doxygen comments for BatchedLinAlg 2024-07-08 13:17:08 -07:00
Will Pazner e5c31240d0 Support hipBLAS for batched linear algebra 2024-07-08 13:17:08 -07:00
Will Pazner 70f2f6677f Add NativeBatchedLinAlg::Mult 2024-07-08 13:17:08 -07:00
Will Pazner 31202eb904 Initial GPU BLAS implementation 2024-07-08 13:17:08 -07:00
Will Pazner bf79ef7f90 Initial framework for batched linear aglebra 2024-07-08 13:17:08 -07:00
Edward Palmer d81bab442d Merge branch 'master' into EdwardPalmer99/add-missing-header-to-exodus-writer-fix 2024-07-05 10:22:09 +01:00
Will Pazner 9343ffad7b Merge pull request #4378 from mfem/mish2/clarify_facestriction_limitations
clarify FaceRestriction limitations
2024-07-02 12:39:19 -07:00
Veselin Dobrev e2a8206d35 Merge branch 'master' into mish2/clarify_facestriction_limitations 2024-07-02 11:14:34 -07:00
Edward Palmer 62049a7990 Merge branch 'master' into EdwardPalmer99/add-missing-header-to-exodus-writer-fix 2024-07-02 15:42:22 +01:00
Edward Palmer d44ff2d39e Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-07-02 15:41:00 +01:00
Tzanio Kolev f5219a2484 Merge pull request #4035 from mfem/silence-duplicate-library-warnings
Silence duplicate libraries linker warnings on Mac
2024-07-01 05:49:18 -07:00
Edward Palmer 6d2dec6361 Checks fespace order in WriteElementBlockParameters. 2024-07-01 08:32:38 +00:00
Edward Palmer ce18700b07 Modifies switch statement cases in WriteElementBlockParameters. 2024-07-01 08:19:10 +00:00
samuelpmishLLNLandWill Pazner 36d28c6e3e Update fem/restriction.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2024-06-28 12:03:56 -07:00
samuelpmishLLNLandWill Pazner 098be49296 Update fem/fespace.hpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2024-06-28 12:03:48 -07:00
Sam Mish b5b4b5da7d improve error message when encountering unsupported element types, and update doxygen entry 2024-06-28 11:29:03 -07:00
Edward Palmer 24e2d0f959 Adds missing header. 2024-06-28 13:49:24 +00:00
Edward Palmer 3d80323862 Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-06-28 08:40:35 +01:00
Tzanio Kolev 7f1eb85c45 Merge pull request #4339 from mfem/hughcars/transfermap_surface_bugfix
Fix TransferMap between surface and volume submesh
2024-06-25 20:14:00 +01:00
Tzanio Kolev 878f94f998 Merge pull request #4190 from mfem/sjg/mesh-part-const
Const correctness for mesh partition
2024-06-25 20:13:39 +01:00
Alex Tyler Chapman 0b75377438 Merge branch 'master' into bugfix/chapman39/slepc-makefile-ordering 2024-06-24 16:36:32 -07:00
Will PaznerandVeselin Dobrev 74e538404e Silence duplicate library warnings in CMake build
Co-authored-by: Veselin Dobrev <dobrev@llnl.gov>
2024-06-24 14:45:34 -07:00
Tzanio Kolev a042d2db0d Merge branch 'master' into silence-duplicate-library-warnings 2024-06-24 18:53:45 +01:00
Tzanio Kolev 01819638f7 Merge pull request #4233 from mfem/mod-ex23
Small correction to example 23 [mod-ex23]
2024-06-24 18:11:39 +01:00
Tzanio Kolev acb5f5acf7 Merge branch 'master' into mod-ex23 2024-06-24 15:56:56 +01:00
Alex Tyler Chapman 5425dbeb5b Switch slepc ordering in dependency list 2024-06-20 11:20:27 -07:00
Mittal, Ketan fe918fcdd4 fix missing division in energy calculation by surf_fit_dof_count 2024-06-19 17:24:47 -07:00
Tzanio Kolev c240df5fbe Merge pull request #4049 from mfem/3942-add-test-for-sundials-usemfemmasslinearsolver
Refactored `ARKStepSolver` to use `ExplicitMult` when either `UseMFEMMassLinearSolver` or `UseSundialsMassLinearSolver` are called.
2024-06-19 00:43:11 +01:00
Tzanio Kolev 950198a3f2 Merge pull request #4354 from mfem/small-bugfixes-2024-06-12
Minor bugfixes
2024-06-19 00:42:34 +01:00
Tzanio Kolev d808463114 Updated CHANGELOG 2024-06-15 14:39:59 -07:00
Tzanio Kolev 5447bcf8a9 Merge branch 'master' into 3942-add-test-for-sundials-usemfemmasslinearsolver 2024-06-15 22:37:29 +01:00
Stowell, Mark L ef6d80189f Adding SparseMatrix::PrintMathematica member function 2024-06-14 11:28:35 -07:00
Edward PalmerEdward PalmerTzanio KolevStowell, Mark L <stowell1@llnl.gov>
7ace2dedf1 Exodus II Writer (#4208)
* Added WriteExodusII method to the Mesh; added exodus_writer cpp file; updated cmakelists.

* Added test_exodus_writer file for Exodus II writer unit tests.

* Setting title, num_dim, num_elem.

* Added function to generate Exodus II element blocks from MFEM mesh.

* Added a function to generate sideset information from an MFEM mesh.

* Added function to get num_nodes for an MFEM mesh.

* Writing coordinates to file.

* Rewritten GenerateExodusIIElementBlocksFromMesh to make use of element attributes.

* Now defining some element block parameters.

* Added WriteNodeConnectivityForBlock; fixed naming of one of the variables.

* Fixed naming for number of nodes per element variable.

* Added function to write sideset boundary IDs to file.

* Added function to write block IDs.

* Added incomplete functiono "GenerateExodusIISidesetsFromMesh" which generates key information about each boundary which can then be written to the file.

* Now also writing the number of elements for each sideset.

* Updated Exodus II writer to write boundary element IDs and side IDs to file.

* Corrected the side_ids_for_boundary_id mapping.

* Rewritten function to generate Exodus II boundary info.

* Fixed incorrect dimensions passed to nc_def_var.

* Added line length and version number info.

* Added header information.

* Removed NETCDF_4 flag (not supported by some programs). Manually setting nc_enddef and nc_redef.

* Added info for timesteps, updated file size info, added info for block element types.

* Added dummy variable to get-around bug in libMesh which prevents the x-coordinate from being read.

* Added ExodusII writer Hex8 test case.

* Updated exodus_writer to handle Tet4.

* Added Tet4 test case and a comparison test function.

* Fixed incorrect variable name.

* Added MFEM to ExodusII side map for Hex8.

* Added Tet4 test ExodusII file.

* Added Wedge6 support to ExodusII writer.

(cherry picked from commit 9412dac5c4dde84732d2f8e82787eac3f9e48906)

* Added ExodusII Wedge6 test case.

(cherry picked from commit 0e69f28a5f357fc44d22f52b1cd5f7a63e492b2b)

* Added Pyramid5 support.

(cherry picked from commit 679b1e4c3f9298513627d27ea130b5e87b80350c)

* Added Pyramid5 test case.

(cherry picked from commit 2f90b786a0228892f76f926af5e0fe04ab741977)

* Commented-out Wedge6 and Pyramid5 tests since the files cannot be read until ReadCubit is updated in a separate PR.

* Removed unused dimension definition; Added support for writing mixed first-order meshes.

(cherry picked from commit 4c111d6f9198f418d32df98fd5e711455f73170a)

* Commented-out test cases that cannot be run with existing ReadCubit ExodusII reader.

* STarted writing a class to encapsulate writing.

* Converted functions to methods in class.

* Removed mesh argument from methods.

* Added CreateEmptyFile, WriteTitle and WriteNumOFElements methods.

* Added database/api versions, floating point word size, max line/name lengths.

* writing element block parameters now handled in class method.

* Sideset information now stored inside class.

* writing nodal variables is now done in a method.

* Add functionality now added to class.

* Added a DefineDimension wrapper around nc_def_dim.

* Added DefineVar wrapper method.

* Added a static method for writing to a file.

* Reordered methods.

* Updated documentation.

* Added safety check to ensure mesh is first-order.

* Added PutVar wrapper method.

* Added PutAtt method.

* Replaced nc_put_att_text.

* Moved nc_redef and nc_enddef into methods.

* WriteNodalCoordinates is now a single method.

* Added DefineAndPutVar method to simplify code.

* Added a macro to check NetCDF status.

* Added a GenerateLabel method.

* Added global named C string labels.

* Updated documentation; merged methods.

* Merged methods for writing boundary info.

* WriteElementBlocks now contains all methods related to this.

* Moved ExodusII file information writer methods into a new method.

* Moved all mesh writer methods into new method.

* Added safety check method.

* Reordered globals; updated documentation; switched set to unordered_set.

* Added test case for Tet10; added additional dofs checks.

* Added handling of second-order Tet (Tet10) elements to exodus writer.

* Updated the "elem_type" names.

* Added Hex27 support to writer.

* Added Hex27 test.

* Added support for Wedge18.

* Added mapping for Pyramid14 (cannot test until reader is able to handle higher-order pyramids).

* Added test files; added additional unit tests.

* Added test comments.

* Commented-out mixed second-order writer test since current reader cannot handle multiple element types.

* Addresses compiler warnings.

* Updated documentation.

* Minor changes to increase readability.

* Updated changelog.

* Address build issue.

* Address compiler warning for unused function used in the unit tests.

* Moves "WriteExodusII" further down to live with the Print methods.

* Renamed "WriteExodusII" to "PrintExodusII" to be consistent.

* Moves ExodusII labels into their own namespace to avoid polluting mfem namespace.

* Temporary mesh output files are now placed in current directory.

* Removes temporary output meshes to avoid false positives.

* Adding Exodus II output option to mesh-explorer

* make style

* Moves ExodusII test meshes into mfem/data directory.

* Fixes minor typo for GenerateExodusIIElementBlocks  documentation.

* Adds a check to confirm that the nodes correspond to a 2nd order H1 space.

* Moves CheckNodalFESpaceIsSecondOrderH1 implementation to bottom.

* Applies style.

* Removes unneeded semi-colon from ExodusIILabels namespace.

* Moves side map arrays into ExodusIISideMaps labels.

Avoids polluting mfem namespace.

* Moves node ordering maps into ExodusIINodeOrderings namespace.

Ensures that mfem namespace is not polluted.

* Removes documentation from #define to fix failing check.

* Revert "Removes documentation from #define to fix failing check."

This reverts commit 39bc2ce836.

* Removes Doxygen documentation from #define to hopefully fix failing test.

* Fixes an issue where the writer failed on interior boundaries.

This initial fix skips internal boundaries. These are not added and a warning is printed indicating which interior boundaries have been skipped.

* Removes mesh test files.

* ExodusII write tests now use mfem/data repository.

* Adds ExodusII test tag.

* Removes varaible underscore prefixes.

* Applies style.

* Uses Generate macro to avoid test duplication.

* Adds link to libMesh issue.

---------

Co-authored-by: Edward Palmer <edward.palmer@ukaea.uk>
Co-authored-by: Tzanio Kolev <tzanio@llnl.gov>
Co-authored-by: Stowell, Mark L <stowell1@llnl.gov>
2024-06-14 10:26:55 -07:00
Will Pazner 36dea0cf38 Merge pull request #4348 from mfem/facerestriction-native-err
Actually throw an error when using native ordering in FaceRestriction
2024-06-13 13:25:25 -07:00
Will Pazner 94ca7d26e8 Check for A.Empty() instead of A == NULL in SparseMatrix 2024-06-13 08:53:05 -07:00
Tzanio Kolev 46e35d0bce Merge pull request #4270 from mfem/findpts-custom-interpolation
Support for custom interpolation procedure using FindPointsGSLIB.
2024-06-13 15:07:02 +01:00
Tzanio Kolev d582c31370 Merge branch 'master' into findpts-custom-interpolation 2024-06-13 15:06:45 +01:00
Veselin Dobrev 8876a84dd4 A set of fixes for small bugs uncovered during the more extensive
testing of https://github.com/spack/spack/pull/44010.
2024-06-12 12:06:56 -07:00
Tzanio Kolev 8153d11274 Merge pull request #3999 from mfem/hdiv-nurbs
Adding H(div) and H(curl) conforming NURBS FiniteElements
2024-06-11 22:37:08 +01:00
Tzanio Kolev 827ed64113 Merge branch 'master' into hdiv-nurbs 2024-06-11 22:36:58 +01:00
Tzanio Kolev ef9f02ba53 Merge pull request #2854 from mfem/var-order-href-op
Variable order space: h-(de)refinement transfer operator
2024-06-11 20:25:12 +01:00
Tzanio Kolev e9a0b0620a Update tests/unit/fem/test_hp_transfer.cpp 2024-06-11 12:24:39 -07:00
Tzanio Kolev 4d9d444248 Merge pull request #4343 from mfem/sample-runs-update
Sample runs update
2024-06-11 20:23:41 +01:00
Tzanio Kolev 6e2badecca Merge pull request #4337 from mfem/lapack-cleanup
Deduplicate single/double precision LAPACK
2024-06-11 20:22:01 +01:00
Tzanio Kolev 99b45fcb02 Merge pull request #4327 from mfem/sjg/mesh-vertex-bdr-table
Add `Mesh::GetVertexToBdrElementTable`
2024-06-11 20:20:12 +01:00
Mittal, Ketan 003dc46a84 Merge branch 'master' of https://github.com/mfem/mfem into findpts-custom-interpolation 2024-06-10 20:33:36 -07:00
Mittal, Ketan 7fac0fbd07 add gslib unit test file to CMakeLists.txt 2024-06-10 20:33:25 -07:00
Jan Nikl 9bae3b25ab Changed the allocation of empty matrices in MixedBilinearForm::ComputeBdrTraceFaceMatrix(). 2024-06-10 15:51:46 -07:00
Jan Nikl c28fd71214 Changed the allocation of empty matrices in (Mixed)BilinearForm::Compute*Matrix(). 2024-06-10 15:46:34 -07:00
Will Pazner 9a327eeca6 Merge pull request #4332 from mfem/sjg/cuda-atomicadd-fix
Fix compiler error for `real_t` and old CUDA architectures < 60
2024-06-10 17:07:12 -04:00
Julian Andrej 086f6c9847 assert -> verify 2024-06-10 10:30:58 -07:00
Julian Andrej 80cb02328c add normal test 2024-06-10 10:27:39 -07:00
Julian Andrej 480e90b41b actually throw an error when using native ordering in FaceRestriction 2024-06-10 08:11:07 -07:00
Julian Andrej 1847e460cf working boundary operators 2024-06-10 08:07:37 -07:00
Mittal, Ketan bc6ba0252a rename method 2024-06-06 12:39:24 -07:00
Will Pazner 262fa6173d Remove unneeded '#ifdef MFEM_USE_SINGLE' 2024-06-05 21:20:59 -07:00
Will Pazner e4b8584a16 Formatting 2024-06-05 21:20:59 -07:00
Socratis Petrides 7912d6915d silly mistake 2024-06-05 18:16:45 -07:00
Socratis Petrides 293d9009ae replace abort with skip so that make test passes with lapack 2024-06-05 17:25:00 -07:00
Socratis Petrides 51bde67bb1 fixing default tol in eltrans 2024-06-05 17:24:24 -07:00
Mittal, Ketan 9caa48d5c8 clean up 2024-06-05 17:15:51 -07:00
Mittal, Ketan f4d286b4b7 hide some arrays not needed by user 2024-06-05 16:12:54 -07:00
Sebastian Grimberg 45e2636921 Address PR feedback: Add const 2024-06-05 14:19:49 -07:00
Mittal, Ketan 57876fbfb0 minor 2024-06-05 09:27:12 -07:00
Julian Andrej 1cd46aa768 add qoi derivatives, dual type option and qoi derivative assembly 2024-06-05 08:31:01 -07:00
Julian Andrej 11edee7aca remove custom enzyme cmake module 2024-06-05 08:29:28 -07:00
Julian Andrej 28115e5de2 temporarily add cmake targets 2024-06-05 08:29:09 -07:00
Julian Andrej 20954328c3 add enzyme to cmake 2024-06-05 08:28:58 -07:00
Tzanio Kolev c3eb769a2a Merge pull request #4330 from mfem/CurlDim-bugfix
GridFunction::CurlDim() nullptr fix
2024-06-05 11:49:58 +01:00
Tzanio Kolev 9286d89b0e Merge branch 'master' into CurlDim-bugfix 2024-06-05 11:49:08 +01:00
Edward Palmer 2b9f428909 Removes test files. 2024-06-05 09:33:09 +00:00
Edward Palmer b230e5f594 Adds [MFEMData] tags. 2024-06-05 09:32:52 +00:00
Edward Palmer 3e8d7f21a2 Adds ExodusII tags. 2024-06-05 09:26:32 +00:00
Mittal, Ketan 47b519047a reviewer comments 2024-06-04 23:17:51 -07:00
Veselin Dobrev 35225e045e Updated the script 'config/sample-runs.sh' to
* run examples 40-99
* run autodiff, dpg, hdiv-linear-solver, and moonolith miniapps
* run additional meshing miniapps
* add 'todo' comments for other missing miniapps

Updated the formatting of sample runs in miniapps/dpg.
2024-06-04 14:17:13 -07:00
Socratis Petrides 22c1087503 VariableOrderRefinementMatrix_main -> VariableOrderRefinementMatrix 2024-06-04 11:32:24 -07:00
Tom Stitt a74663d634 Use HYPRE's SpMV since it is currently faster than rocSPARSE's 2024-06-04 10:51:23 -07:00
Hugh Carson 469096892f Address bug where transfer map failed for transfering between surface and a volume root mesh 2024-06-04 10:24:29 -04:00
Socratis Petrides cd5d2f7c04 minor tweaks 2024-06-03 22:21:20 -07:00
Ketan MittalandVladimir Tomov 7fc2ce350d Update fem/gslib.hpp based on reviewer's suggested change
Co-authored-by: Vladimir Tomov <tomov2@llnl.gov>
2024-06-03 15:18:05 -07:00
Edward Palmer 4314dc64db Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-06-03 14:33:00 +00:00
Will Pazner cbae29ad06 Suppress Doxygen warnings in LAPACK header
Otherwise Doxygen complains:

warning: documented symbol 'void mfem::MFEM_LAPACK_PREFIX' was not declared or defined.
2024-06-02 13:21:37 -07:00
Will Pazner 4c1d842c72 Adjust tolerances depending on single or double precision
ex38 was not converging in single precision with previous tolerances
2024-06-02 13:21:37 -07:00
Will Pazner 4aa0ed52af Don't duplicate single and double precision LAPACK code
Introduce MFEM_LAPACK_PREFIX and MFEM_LAPACK_COMPLEX to prepend 's', 'd', 'c',
or 'z' to the BLAS/LAPACK function name according to the precision.

Add new header lapack.hpp with declarations of the LAPACK functions.
2024-06-02 13:06:51 -07:00
Veselin Dobrev fca4c314d4 Remove the old field 'GridFunction::fec' since the deprecation
attribute does not work as expected with GCC.

Add a CHANGELOG entry about the renaming 'fec' -> 'fec_owned' in
class GridFunction.

Fixed two GCC warnings that do not show up in CI.
2024-05-31 17:53:18 -07:00
Socratis Petrides 8712d02570 remove leftover RT space from enum 2024-05-31 11:17:00 -07:00
Socratis Petrides e5fec6279b minor fix 2024-05-31 11:11:24 -07:00
Socratis Petrides e90e96f9a5 add unit L2 test 2024-05-31 11:04:51 -07:00
Veselin Dobrev 459def6d79 Fix the build with PUMI support enabled.
For backward compatibility, define `GridFunction::fec` as a deprecated
reference to `GridFunction::fec_owned`.

Fix a few warnings in the PUMI examples.
2024-05-31 10:15:56 -07:00
Sebastian Grimberg cc00ef7d90 Fix compiler error for real_t and old CUDA architectures pre-6.0 2024-05-30 13:52:27 -07:00
Christopher vogl 2c0346bc36 second pass to CHANGELOG addition to more specifically describe new functionality of ARKStepSolver 2024-05-29 17:33:46 -07:00
Christopher vogl 56186d8770 fixed typos in previous commit 2024-05-29 17:21:18 -07:00
Christopher vogl 1645b854a4 updated CHANGELOG to note refactoring in ARKStepSolver 2024-05-29 17:20:18 -07:00
Christopher vogl 69fd2f9051 Merge remote-tracking branch 'origin/master' into 3942-add-test-for-sundials-usemfemmasslinearsolver
- needed to pull in changes to CHANGELOG before adding to it
2024-05-29 17:00:44 -07:00
Socratis Petrides a5d230f199 minor 2024-05-29 12:11:32 -07:00
Christopher vogl ba4b627e68 updated comments in SUNDIALS examples to reflect target name differences between GNU make and CMake 2024-05-29 11:53:01 -07:00
IdoAkkerman 62c535d0ee Merge branch 'hdiv-nurbs' of github.com:mfem/mfem into hdiv-nurbs 2024-05-29 17:56:33 +02:00
IdoAkkerman 17829d1c38 Seperate NURBS examples in dox - small typo 2024-05-29 17:56:15 +02:00
IdoAkkerman dd198ce3f9 Seperate NURBS examples in dox 2024-05-29 17:54:43 +02:00
IdoAkkerman b9f36468ba Move changes to v 4.7.1 2024-05-29 17:45:30 +02:00
Tzanio Kolev 76bcd044d0 Merge branch 'master' into hdiv-nurbs 2024-05-29 07:49:33 -07:00
Socratis Petrides 84ce403ffb fix doxygen 2024-05-28 17:42:24 -07:00
Socratis Petrides df09aea4da rename GridFunction member variable fec 2024-05-28 17:31:15 -07:00
Socratis Petrides d2840464ba null fec pointer fix 2024-05-28 14:32:11 -07:00
Christopher vogl 0dff351b2e added new example 16 tests to GNU build system. 2024-05-28 12:05:03 -07:00
Christopher vogl 55a914321d refactored sample runs to exclude sundials_ prefix 2024-05-28 11:56:44 -07:00
Christopher vogl 1410aef639 updated examples/sundials/CMakeLists.txt so the executables are named the same as with GNU build system 2024-05-28 11:56:24 -07:00
Chris VoglandVeselin Dobrev e1ac8ca08c Adding precision to Save call in SUNDIALS ex16
Co-authored-by: Veselin Dobrev <v-dobrev@users.noreply.github.com>
2024-05-28 11:00:45 -07:00
IdoAkkerman 28916b23a4 Merge branch 'master' into hdiv-nurbs 2024-05-27 10:24:58 +02:00
IdoAkkerman 7476c00f2b Solenoidal convergence check SINGLE/DOUBLE 2024-05-27 10:23:03 +02:00
Socratis Petrides 7a0137c496 generalize unit test 2024-05-24 12:49:18 -07:00
Socratis Petrides 4267b2af05 fix for using Rhp in the deref case and addressing reviewer comments 2024-05-24 12:48:51 -07:00
Socratis Petrides e37daad5eb fix from Dylan 2024-05-24 12:48:10 -07:00
Jan Nikl a461f25b4a Revert "Fixed short circruiting."
This reverts commit 20b52574c6.
2024-05-24 09:10:57 -07:00
Socratis Petrides 20b4b72071 Merge branch 'master' into var-order-href-op 2024-05-23 19:25:48 -07:00
Sebastian Grimberg ddf80492c5 Add Mesh::GetVertexToBdrElementTable 2024-05-23 11:13:16 -07:00
Jan Nikl db8b4c9f20 Added a note about ignored markers. 2024-05-23 10:39:56 -07:00
Jan Nikl d4137f9c7a Renamed Compute*FaceElementMatrix() to be more consistent. 2024-05-23 10:37:31 -07:00
Jan Nikl 20b52574c6 Fixed short circruiting. 2024-05-23 10:12:51 -07:00
Jan Nikl e33344f539 Added real_t based matrix tolerance in Hybridization::ConstructC(). 2024-05-23 09:19:36 -07:00
Christopher vogl 712ae82026 updated documentation of ExplicitMult to mention ARKStep 2024-05-22 15:25:42 -07:00
Jan Nikl 606439cfb8 Minor comment styling in Hybridization. 2024-05-22 15:24:04 -07:00
Jan Nikl 147cbc014a Renamed boundary constraint integrators. 2024-05-22 15:10:37 -07:00
Jan Nikl aa07a1b175 Added override to VectorFEBoundaryNormalLFIntegrator. 2024-05-22 14:57:01 -07:00
Jan Nikl 53171de727 Added override to VectorFEBoundaryFluxIntegrator. 2024-05-22 14:54:53 -07:00
Christopher vogl 53dd97e0d8 updated checks to be 'not EXPLICIT' in ARKStepSolver, adding one to UseSundialsMassLinearSolver as well 2024-05-22 14:08:31 -07:00
Christopher vogl 0c413570c4 braces added to meet style requirements 2024-05-22 13:43:24 -07:00
Jan Nikl d6bfc6370e Added access to the boundary constraint integrators. 2024-05-22 11:50:18 -07:00
IdoAkkerman f8d18cd4be Merge branch 'master' into hdiv-nurbs 2024-05-22 17:45:58 +02:00
IdoAkkerman 994d83dd80 Modify tolerance for single precision - more relaxed 2024-05-22 17:44:17 +02:00
IdoAkkerman f218efae09 Modify tolerance for single precision 2024-05-22 16:06:11 +02:00
Christopher vogl 43bb865c26 updated SUNDIALS ex16 and ARKStepSolver to use TDO::Type 2024-05-21 19:51:22 -07:00
Christopher vogl dd9b723cfd reverted addition of SetImplicit in lieu of calling code using constructor... corrected typo 2024-05-21 19:39:59 -07:00
Christopher vogl 87362ca1ca added TimeDepedentOperator::SetImplicit function 2024-05-21 16:47:07 -07:00
Christopher vogl a6afefc6a5 updated documentation of SUNDIALS functions in TimeDependentOperator 2024-05-21 16:46:11 -07:00
Christopher vogl e1dc4680d3 using real_t in SUNDIALS ex16, ex16p, and ARKStepSolver 2024-05-21 15:01:11 -07:00
Christopher vogl 44985dacc0 Merge branch 'master' into 3942-add-test-for-sundials-usemfemmasslinearsolver 2024-05-21 14:21:32 -07:00
Mittal, Ketan 9c77f6b407 minor 2024-05-21 11:54:14 -07:00
Mittal, Ketan 8df0341e11 Merge branch 'master' of https://github.com/mfem/mfem into findpts-custom-interpolation 2024-05-21 11:51:37 -07:00
Julian Andrej 41e92219ee reorganize files and navier stokes example 2024-05-21 09:01:10 -07:00
Ido Akkerman af24eaea27 Merge branch 'master' into hdiv-nurbs 2024-05-21 10:57:13 +02:00
Sebastian Grimberg f8c6512cf9 Merge branch 'master' into sjg/mesh-part-const 2024-05-20 11:16:28 -07:00
Jan Nikl 4385e6d568 Merge branch 'master' into najlkin/integral-els 2024-05-20 10:01:20 -07:00
Jan Nikl 171346b20f Changed double to real_t in VectorFEBoundaryNormalLFIntegrator and VectorFEBoundaryFluxIntegrator. 2024-05-20 09:44:37 -07:00
Jan Nikl 6f7320c240 Merge branch 'master' into najlkin/mixed-face-forms 2024-05-20 09:41:11 -07:00
Sebastian Grimberg e6ce6e7532 Merge branch 'master' into sjg/mesh-part-const 2024-05-17 12:32:10 -07:00
Mittal, Ketan 7338e797bb merge with master and resolve conflict 2024-05-17 09:49:11 -07:00
Mittal, Ketan 31d931a99c Update changelog 2024-05-16 11:01:06 -07:00
Ketan Mittal 3ae930c93b Merge branch 'master' into findpts-custom-interpolation 2024-05-16 10:52:44 -07:00
Mittal, Ketan 36f882257e minor doc update 2024-05-16 10:34:49 -07:00
Jan Nikl 1508ae0886 Added a switch for external boundary constraint integrators in Hybridization. 2024-05-16 10:29:01 -07:00
IdoAkkerman 8adb7461b0 Comment on array of NURBSexts 2024-05-15 10:38:14 +02:00
IdoAkkerman 716e370d35 Fix merge 2024-05-14 14:21:58 +02:00
IdoAkkerman 37c0768fe3 Merge branch 'master' into hdiv-nurbs 2024-05-14 14:12:44 +02:00
Sebastian Grimberg d395caad9b Fix some missed corrections for parallel NURBS meshes 2024-05-13 11:04:29 -07:00
Sebastian Grimberg 3e5e18797c Address PR feedback: Rename variable and enforce 80 character width 2024-05-13 10:20:59 -07:00
Sebastian Grimberg aab273b303 Address PR feedback and borrow partitioning improvement from #2669 2024-05-13 10:20:37 -07:00
Edward Palmer e4d1a861c9 Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-05-13 13:02:49 +00:00
Mittal, Ketan 217b77d5f0 minor 2024-05-09 11:33:21 -07:00
Mittal, Ketan bb67d6cb98 add unit test 2024-05-09 11:30:27 -07:00
Jan Nikl cb6b94d9d9 Merge branch 'master' into najlkin/mixed-face-forms 2024-05-09 09:44:15 -07:00
Jan Nikl 7f118b0793 Fixed MixedBilinearForm::ComputeBdrTraceFaceElementMatrix(). 2024-05-09 09:41:54 -07:00
Jan Nikl a6f0a23998 Fixed a typo in Hybridization. 2024-05-09 09:34:38 -07:00
Jan Nikl 9eb4e5f947 Fixed boundary constraint integration. 2024-05-09 09:34:23 -07:00
Sebastian Grimberg 6a48f8e165 Const correctness for mesh partition 2024-05-06 12:21:37 -07:00
Julian Andrej a0c3620618 relocate restrictions to individual operators 2024-05-06 08:19:05 -07:00
Julian Andrej 9c791bed5a starting boundary and L2 2024-05-02 15:39:17 -07:00
Mittal, Ketan 696cbd05e8 improved documentation 2024-05-01 14:18:30 -07:00
Chris Vogl 6c8a4188a1 Merge branch 'master' into 3942-add-test-for-sundials-usemfemmasslinearsolver 2024-04-30 11:45:49 -07:00
Julian Andrej 42d0fc17a1 updates 2024-04-29 08:53:33 -07:00
Julian Andrej 2d2da417bb bugfixes 2024-04-25 15:05:21 -07:00
Mittal, Ketan 829b123641 minor fix for when there are no points received on a rank 2024-04-25 12:22:41 -07:00
Mittal, Ketan bd52201add minor 2024-04-25 12:07:33 -07:00
Mittal, Ketan 812ecce84a add doxygen comments 2024-04-25 12:06:00 -07:00
Mittal, Ketan 172c38b675 initial commit 2024-04-25 11:17:02 -07:00
Julian Andrej 2e86ccb948 working assembly 2024-04-22 08:36:39 -07:00
Ido Akkerman 875b5f3f52 Update nurbs_naca_cmesh.cpp 2024-04-05 22:34:26 +02:00
IdoAkkerman be1f36a523 Merge branch 'master' into hdiv-nurbs 2024-04-05 21:54:43 +02:00
IdoAkkerman d4c7dd3490 Double -> Real 2024-04-05 21:53:48 +02:00
IdoAkkerman e55fb21538 Merge branch 'master' into mod-ex23 2024-04-05 14:36:47 +02:00
IdoAkkerman a966b0502f Rewrite BC enforcement 2024-04-05 12:59:41 +02:00
IdoAkkerman 1bac61ad1c Typos 2024-04-05 09:04:06 +02:00
IdoAkkerman c7451115d8 Typos 2024-04-05 09:00:12 +02:00
IdoAkkerman 72ae003a00 double -> real_t 2024-04-05 08:51:50 +02:00
IdoAkkerman d67098b8f8 remove unused variable 2024-04-04 22:06:01 +02:00
IdoAkkerman 0e6dbaf050 remove 999 statement 2024-04-04 21:52:34 +02:00
IdoAkkerman 32afc8565c remove 9999 statement 2024-04-04 21:48:24 +02:00
IdoAkkerman 48ace60875 order[2] big fix + small cosmetic changes 2024-04-04 21:46:38 +02:00
Edward Palmer 9e75f9e19d Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-04-04 18:22:28 +01:00
IdoAkkerman 9b6ee6fcad Add papers -- comment flipped with previous commit 2024-04-04 14:44:54 +02:00
IdoAkkerman 1a1639b87e Add neumann example 2024-04-04 14:44:14 +02:00
IdoAkkerman 3735aa504b Comments dec 2023 2024-04-04 12:50:03 +02:00
Ido Akkerman 367dda6794 Merge branch 'master' into hdiv-nurbs 2024-04-04 11:59:43 +02:00
IdoAkkerman e4a85f79cd Remove multipatch examples in ex24 2024-04-03 22:30:00 +02:00
Will Pazner 22ec7e7ada Use LDFLAGS_INTERNAL for 'no_warn_duplicate_libraries' on Mac 2024-04-03 09:53:43 -07:00
IdoAkkerman 678101938b Merge branch 'master' into hdiv-nurbs 2024-04-03 13:37:00 +02:00
IdoAkkerman 9a3aa18c62 Fix 3D Hcurl 2024-04-03 13:24:20 +02:00
IdoAkkerman 5a3ba1424a Update comments 2024-04-02 13:29:48 +02:00
IdoAkkerman 3bb8419a96 Add miniapp source to doc 2024-04-02 13:07:45 +02:00
IdoAkkerman ce29282f63 Add Hdiv and Hcurl NURBS to changelog 2024-04-02 12:51:37 +02:00
IdoAkkerman 9575299ae3 Merge branch 'master' into hdiv-nurbs 2024-04-02 09:04:07 +02:00
IdoAkkerman 995ceca6c2 Cosmetic fix of 2D Hcurl dof count 2024-04-02 09:03:04 +02:00
Christopher vogl 012aa50cd3 accounted for residual differences in SUNImplicitSolve is mass linear solve is used 2024-03-28 23:03:25 -07:00
Christopher vogl d0193919c4 updated parallel version of ex16 2024-03-28 21:27:47 -07:00
Christopher vogl 422ca290b5 added more explanation for SUNImplicitSetup 2024-03-28 20:16:01 -07:00
Christopher vogl 59e1d7bf27 updated comments to reflect the linearization assumption used throughout 2024-03-28 20:09:49 -07:00
Christopher vogl 09dd9656c8 updated ex16 for SUNDIALS to make of ExplicitMult to unify TDO implementations 2024-03-28 19:50:04 -07:00
Christopher vogl e9afca2cd6 updated RHS1 and RHS2 in ARKStepSolver to use ExplicitMult for mass form ODEs 2024-03-28 19:49:14 -07:00
IdoAkkerman 6abd0e6002 Fix 2d Curl 2024-03-28 13:29:15 +01:00
IdoAkkerman 656e3062b4 Fix mesh name in example 2024-03-28 13:14:40 +01:00
IdoAkkerman 9c9c519175 Add exso.mesh to gitignore -- prevent regression test error 2024-03-27 14:23:14 +01:00
IdoAkkerman 8300809562 Fix double real_t conversion 2024-03-27 13:57:51 +01:00
IdoAkkerman cce7296ffe Tweak example runs in miniapps; add comments; remove comments 2024-03-27 13:39:50 +01:00
IdoAkkerman a445ad00da Update clean statement in makefile 2024-03-27 13:38:26 +01:00
IdoAkkerman 4db7e1a107 Add comments to NURB extension mode 2024-03-27 13:37:36 +01:00
IdoAkkerman c51a1c4aa9 Remove comment in nurbs, tweak error statement 2024-03-27 13:37:09 +01:00
IdoAkkerman d489908e50 Merge branch 'master' into hdiv-nurbs 2024-03-26 16:55:17 +01:00
IdoAkkerman c4eda188d5 Tweak nurbs ex24 2024-03-26 16:31:27 +01:00
IdoAkkerman 677eb4c876 Change solver params for nurbs ex5 2024-03-26 16:30:57 +01:00
IdoAkkerman 69a4aa70b9 Fix mapping in NURBS Hdiv 3D 2024-03-26 15:37:49 +01:00
IdoAkkerman f07c2f460d Update copyright statement 2024-03-26 10:29:49 +01:00
Julian Andrej c7fe1ff1f4 working most recent interface iteration 2024-03-25 09:38:56 -07:00
Socratis Petrides 09c557bdd7 Merge branch 'master' into var-order-href-op 2024-03-21 19:45:57 -07:00
Socratis Petrides e487da01c5 adding unit tests 2024-03-21 19:44:43 -07:00
IdoAkkerman b6b6843ad2 Merge branch 'master' into hdiv-nurbs 2024-03-21 15:46:53 +01:00
Edward Palmer 011f7b0350 Fixed compiler warning. 2024-03-18 12:21:16 +00:00
Edward Palmer 41eb57cee3 Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-03-18 12:03:55 +00:00
Edward Palmer c4b389a4c6 Merge branch 'cubit-mixed-mesh-support-dev' into cubit-pyramid-wedge-support-dev 2024-03-18 12:00:02 +00:00
Edward Palmer 6e6eeccf61 Updated changelog. 2024-03-15 18:20:59 +00:00
Edward Palmer a82ec2a298 Fixed incorrect cubit side maps for Hex8 and Pyramid5. 2024-03-15 18:00:49 +00:00
Edward Palmer ac956e53da Fixed typo. 2024-03-15 16:14:02 +00:00
Edward Palmer 7332f65373 Fixed GetFaceType (incorrect Wedge, Pyramid faces for side ids). 2024-03-15 16:06:10 +00:00
Edward Palmer 5229753c6f Boundary side ids now 1-indexed to be consistent with Exodus. 2024-03-15 15:41:12 +00:00
Edward Palmer 5cefe337dd Now correctly using boundary ID (1-index) to extract sideset information. 2024-03-15 14:52:28 +00:00
Edward Palmer 292051c8e3 Element IDs now numbered from 1 internally to be consistent with Exodus II. 2024-03-15 14:42:14 +00:00
Edward Palmer 68ccd510c2 Added mixed first/second-order Exodus unit tests. 2024-03-15 13:44:24 +00:00
Edward Palmer 3204614d51 Added back support for higher-order element types. 2024-03-15 13:37:33 +00:00
Edward Palmer bd95160e67 Modified ReadCubit functions to take-in a Cubit block class instance allowing for multiple elements; temporarily removed support for higher-order elements. 2024-03-15 13:34:12 +00:00
Edward Palmer bbf6f013af Added reverse mapping going from element ID to the block ID. 2024-03-15 11:52:42 +00:00
Edward Palmer a30390c306 Updated ReadCubitElementBlocks to use CubitBlock class; fixed potential issue where we assumed that blocks were numbered contiguously from 1 (not necessarily the case). 2024-03-15 11:36:11 +00:00
Edward Palmer 8e5de72407 Renamed ReadCubitNumNodesPerElement to ReadCubitBlocks; currently still limited to single element type 2024-03-15 11:30:24 +00:00
Edward Palmer 5691f60988 Add GetNumNodes method to CubitElement. 2024-03-15 11:20:37 +00:00
Edward Palmer 9f9ccdcc55 Added CubitBlock class which stores the element type for each block. 2024-03-14 18:22:59 +00:00
Edward Palmer 0b10bcbba5 Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-03-14 17:38:14 +00:00
Edward Palmer a4731e0031 Added new Exodus unit tests to changelog. 2024-03-14 17:36:25 +00:00
Edward Palmer e94d3b1628 Updated changelog. 2024-03-14 17:34:36 +00:00
Edward Palmer ecdf4717a9 Addressed compiler warnings. 2024-03-14 17:09:06 +00:00
Edward Palmer c181e720c9 Cleanup and documentation tweaks. 2024-03-14 16:56:06 +00:00
Edward Palmer fe8cac7082 Removed CubitElement constructor. 2024-03-14 16:30:58 +00:00
Edward Palmer b02f1e492f Renaming; minor changes. 2024-03-14 16:27:34 +00:00
Edward Palmer ad1bbd318a Moved additional methods into CubitElement class. 2024-03-14 16:13:08 +00:00
Edward Palmer 136eddd3ab Moved cubit enums out of CubitElement class and added static methods for determining element type base on number of nodes and dimension. 2024-03-14 15:46:31 +00:00
Edward Palmer e6725d8f5c Renamed CubitElementInfo to CubitElement. 2024-03-14 14:22:58 +00:00
Edward Palmer 484a27c13d Renamed _face_info to _face. 2024-03-14 13:52:27 +00:00
Edward Palmer 7b67a8bf15 Removed std use; Removed _num_faces member variable. 2024-03-14 13:52:02 +00:00
Edward Palmer 916c13a2ef Removed unnecessary methods. 2024-03-14 13:44:38 +00:00
Edward Palmer 5b4deb99b0 Removed CubitFaceInfo class; simplified CubitElementInfo class. 2024-03-14 13:41:41 +00:00
Edward Palmer 6e5a42c921 Fixed unused-variable and sign-comparison compiler-warnings. 2024-03-14 11:24:34 +00:00
Edward Palmer 8478f268ac Fixed bug in ReadCubitNodeCoordinates where coordy was unused. 2024-03-14 11:20:54 +00:00
Mittal, Ketan 66a1379947 Merge branch 'var-order-href-op' of https://github.com/mfem/mfem into var-order-href-op 2024-03-13 13:10:41 -07:00
Mittal, Ketan 546ba6c11e add support for h-derefinement 2024-03-13 13:10:24 -07:00
Mittal, Ketan 67ba63c2f4 Merge branch 'master' of https://github.com/mfem/mfem into var-order-href-op 2024-03-13 13:08:02 -07:00
Edward Palmer 01b25a54c9 Merge branch 'master' into cubit-pyramid-wedge-support-dev 2024-03-13 17:58:38 +00:00
Edward Palmer 51dd386f0e Added documentation to mesh header. 2024-03-13 16:51:44 +00:00
Edward Palmer f0d24c1ffd Cleaned-up ReadCubit documentation. 2024-03-13 16:46:30 +00:00
Edward Palmer 6174655771 Updated documentation and tidied-up NetCDFReader. 2024-03-13 16:35:39 +00:00
Edward Palmer d91656c768 Renamed CubitElementInfo methods to be consistent; updated documentation. 2024-03-13 16:29:45 +00:00
Edward Palmer 89d95adb09 Renamed CubitFaceInfo accessors to be consistent; added documentation. 2024-03-13 16:14:39 +00:00
Edward Palmer 639ae13d24 Added character buffer to NetCDFReader. 2024-03-13 15:53:18 +00:00
Edward Palmer 1df6c37ded Removed unused properties from CubitFaceInfo. 2024-03-13 15:45:31 +00:00
Edward Palmer 08406c8410 Removed dimension property of CubitElementInfo. 2024-03-13 15:42:36 +00:00
Edward Palmer 1be4bc267f Switched to C-style character array arguments for NetCDFReader. 2024-03-13 15:36:45 +00:00
Edward Palmer 4e61084655 Added HasVariable and HasDimension methods. 2024-03-13 15:32:53 +00:00
Edward Palmer 05e8ef16b8 Using NetCDFReader in ReadCubit methods to cleanup code. 2024-03-13 15:13:57 +00:00
Edward Palmer df59f3856a Added simple NetCDFReader class to wrap-around netcdf C-api. Cleans-up ReadCubit. 2024-03-13 11:53:14 +00:00
Edward Palmer 4d5a99f3de Added unit tests for Exodus reader. 2024-03-13 10:54:09 +00:00
Edward Palmer 5c6917697a Removed TODOs. 2024-03-12 16:40:15 +00:00
Edward Palmer 459d4a4940 Added support for Pyramid14 (although not currently handled by H1 FEC). 2024-03-12 16:35:43 +00:00
Edward Palmer 76a71f69bc Added support for Wedge18. 2024-03-12 16:25:43 +00:00
Edward Palmer 51279cb47f Removed unnecessary order 2 side-maps. 2024-03-12 15:50:03 +00:00
Edward Palmer f9845fe3fc Removed NumNodes. 2024-03-12 15:46:30 +00:00
Edward Palmer 802d249684 Removed unused NumFaceNodes. 2024-03-12 15:44:52 +00:00
Edward Palmer 6f86a4241b Revert "Cleaned-up BuildCubitBoundaries."
This reverts commit d37bbb7d93.
2024-03-12 15:28:31 +00:00
Edward Palmer 08ef67eafa Fixed incorrect array size for wedge6 mapping. 2024-03-12 15:17:19 +00:00
Edward Palmer 5103b31c6a Removed FACE_QUAD8. 2024-03-12 13:40:57 +00:00
Edward Palmer ce810be429 Made return-type void for BuildCubitBlockIDs. 2024-03-12 10:58:52 +00:00
Edward Palmer 287e4835f0 Renamed functions to be consistent. 2024-03-12 10:52:44 +00:00
Edward Palmer 261af4c74d Applied formatting. 2024-03-12 10:42:17 +00:00
Edward Palmer 3d3e0f991d Added BuildCubitToMFEMVertexMap function. 2024-03-12 10:40:12 +00:00
Edward Palmer d37bbb7d93 Cleaned-up BuildCubitBoundaries. 2024-03-11 17:58:14 +00:00
Edward Palmer 8f0b439eb8 Removed unused variable. 2024-03-11 17:50:52 +00:00
Edward Palmer 5f4fc3acb8 Applied style. 2024-03-11 17:02:13 +00:00
Edward Palmer 6b744436fb Renamed BuildMFEMBoundaryElements to BuildCubitBoundaries. 2024-03-11 17:01:53 +00:00
Edward Palmer bdfd326e8c Renamed BuildMFEMElements to BuildCubitElements. 2024-03-11 17:01:30 +00:00
Edward Palmer 7676d7fff6 Renamed BuildMFEMVertices to BuildCubitVertices. 2024-03-11 17:01:12 +00:00
Edward Palmer 1fdd927821 Renamed "corner_nodes" to "vertices". 2024-03-11 17:00:39 +00:00
Edward Palmer 2374ae0588 Commented-out FACE_QUAD8. 2024-03-11 16:58:44 +00:00
Edward Palmer deec36125c Consistent formatting for "IDs". 2024-03-11 16:56:34 +00:00
Edward Palmer abbf08a0f8 Added ReadCubitBoundaryIDs function. 2024-03-11 16:54:05 +00:00
Edward Palmer 2473af39c3 Added BuildMFEMBoundaryElements method. 2024-03-11 16:48:02 +00:00
Edward Palmer d3a2aa17b8 Added BuildMFEMElements method. 2024-03-11 16:37:30 +00:00
Edward Palmer 7b5f9a4157 Added back support for Hex27 and Tet10. 2024-03-11 15:53:53 +00:00
Edward Palmer 178ae5ea9d Added support for Pyramid 5 (now supporting Tet4, Hex8, Pyramid5, Wedge6). 2024-03-11 15:22:28 +00:00
Edward Palmer b9e4187a71 Added BuildMFEMVertices method. 2024-03-11 14:27:41 +00:00
Edward Palmer 053b114172 Fixed bug in GetElementIdsForBlockId. 2024-03-11 14:27:12 +00:00
Edward Palmer c8dcee1065 Extracted the creation of unique_vertex_ids into a function. 2024-03-11 12:38:10 +00:00
Socratis Petrides 99c2967920 Merge branch 'master' into var-order-href-op 2024-03-08 12:07:52 -08:00
Edward Palmer 3c561a774d Extracted code out of ReadCubit into BuildBOundaryNodeIds. 2024-03-08 20:06:27 +00:00
Edward Palmer 6679963094 Removed GetCubitBlockIndexForElement. 2024-03-08 19:45:16 +00:00
Edward Palmer eda46fef52 Tidied-up ReadCubitBoundaries. 2024-03-08 19:44:30 +00:00
Edward Palmer 0efb8c8c2d Added maps and simplified code to facilitate extending element type support; not currently working. 2024-03-08 19:24:43 +00:00
Edward Palmer ad90bb1acf Extracted block ids to function. 2024-03-08 16:13:57 +00:00
Edward Palmer 7e5fed72cf Comment-out all except wedge6. 2024-03-08 16:00:44 +00:00
= 6d81cb7748 Node orderings added between Genesis and MFEM for Pyramid14 and Wedge18. 2024-03-07 17:36:03 +00:00
= ca92a847f2 GetWedge6FaceInfo and similar methods now use MFEM face orderings. 2024-03-07 16:49:41 +00:00
= c8d4285c94 Switched to enum argument rather than integer. 2024-03-07 16:02:52 +00:00
= 323db614e1 Add FACE_QUAD8 to CreateCubitBoundaryElement. 2024-03-07 15:59:52 +00:00
= 9a2c0ce611 Extended CreateCubitElement to add support for Wedges and Pyramid elements. 2024-03-07 15:57:09 +00:00
= 2b024ff6b3 Removed existing cubit enums and existing functions (now using the CubitElementInfo and CubitFaceInfo classes). 2024-03-07 15:38:20 +00:00
= 38e37437bd Switched to Pascal case for methods to be consistent. 2024-03-07 15:14:19 +00:00
= 5e0c5469ec Added CubitElementInfo class to store information about an element type; not currently in use. 2024-03-07 15:07:17 +00:00
= a095387aa2 Added CubitFaceInfo class to store information about each face. Not currently used. 2024-03-07 14:59:41 +00:00
= c084361b5c Added pyramid and wedge element types to CubitElementType enum. 2024-03-07 14:54:02 +00:00
Socratis Petrides 2dc419f1ae Merge branch 'master' into var-order-href-op 2024-03-04 11:36:38 -08:00
Mittal, Ketan 450d6cea6d Merge branch 'master' of https://github.com/mfem/mfem into var-order-href-op 2024-03-04 09:51:05 -08:00
Socratis Petrides 75df4ad3e6 replace umfpack with PCG 2024-02-16 12:18:21 -08:00
Socratis Petrides 4048d46443 Merge branch 'master' into var-order-href-op 2024-02-16 12:02:58 -08:00
Jan Nikl 381cf25cbd Added const qualifiers to the Compute*ElementMatrix() methods of (Mixed)BilinearForm. 2024-02-08 15:04:56 -08:00
Jan Nikl 556b577818 Added BilinearForm::Compute(Bdr)FaceElementMatrix(). 2024-02-08 14:48:29 -08:00
Jan Nikl a5ca1b6a32 Added MixedBilinearForm::Compute(Bdr)TraceFaceElementMatrix(). 2024-02-08 14:47:56 -08:00
Jan Nikl 1b7f20af7f Added VectorFEBoundaryNormalLFIntegrator for integrating (f.n, v.n). 2024-02-08 14:31:47 -08:00
Jan Nikl a65f6064ba Added VectorFEBoundaryFluxIntegrator for integrating (Q u.n, v.n). 2024-02-08 14:31:47 -08:00
Jan Nikl 993d5cb831 Added boundary face constraint integrators to Hybridization. 2024-02-08 14:31:47 -08:00
Jan Nikl 458447caf0 Replaced the check by support of scalar integral FEs in VectorFECurlIntegrator. 2024-01-25 16:49:09 -08:00
Jan Nikl a6138169fa Removed the check from VectorFEDivergenceIntegrator as it already supports intefral fes. 2024-01-25 16:06:30 -08:00
Jan Nikl d65033932e Added support of the integral finite elements to the L2 GridTransfer. 2024-01-25 14:39:25 -08:00
Jan Nikl d49629b916 Added extrusion of L2 integral finite elements. 2024-01-25 14:13:12 -08:00
Jan Nikl a642f36524 Added support for integral scalar elements to BilinearFormIntegrators. 2024-01-25 12:46:41 -08:00
Christopher vogl bbd4edce83 style changes 2023-12-27 14:34:32 -08:00
Christopher vogl 1742616cac uncommited changes to add UseMFEMMassLinearSolver to CTest suite 2023-12-27 14:34:21 -08:00
Christopher vogl 2c64bbab79 refactored 16p with all changes made to 16 2023-12-27 12:36:16 -08:00
Christopher vogl 785fa7adc2 added some comments to clarify difference between MFEM and SUNDIALS solves 2023-12-27 12:33:46 -08:00
Christopher vogl 0248c58591 use newer GridFunction::Save 2023-12-27 12:11:52 -08:00
Christopher vogl 160e783638 whitespace cleanup 2023-12-27 12:11:35 -08:00
Christopher vogl ddd2500a9c removed deprecate SetParameters definition 2023-12-27 12:11:14 -08:00
Christopher vogl 2b5dee2b95 added more example to show speedup with mass form 2023-12-27 12:10:49 -08:00
Christopher vogl 26393f230f corrected typos and added runs to sample runs 2023-12-27 11:10:01 -08:00
Christopher vogl 8e9948d729 removed no longer necessary auxilliary variable 2023-12-22 17:27:11 -08:00
Christopher vogl 9bbbd8c324 refactor to eliminate the copy-paste in SetParameters 2023-12-22 17:23:40 -08:00
Christopher vogl a19e7cb38e some last touchups to ex16 2023-12-22 15:43:34 -08:00
Christopher vogl 3a2912bc0b factored ConductionOperator into separate classes 2023-12-22 14:37:48 -08:00
Christopher vogl 5cfd284cb8 implemented new mfem mass options 2023-12-22 13:45:14 -08:00
Christopher vogl eeae538115 updated comments 2023-12-21 22:45:36 -08:00
Christopher vogl 8a98c0332f fixed copy-paste bug in sundials: LSA should be LSM 2023-12-21 22:43:06 -08:00
Christopher vogl 84d44db3a7 Implemented new SUN routines & fixed tolerance bug 2023-12-21 22:22:57 -08:00
Christopher vogl c97af2f3dc Converted remaining raw pointers in SUNDIALS ex16
originally was going to keep the raw pointers in the ConductionOperator to
facilitate comparison with MFEM ex16, but now want to avoid incurring more
technial debt as additional TimeDependentOperator functions are implemented
2023-12-21 15:12:49 -08:00
Christopher vogl c4ca3bfc5f Cleanup of SUNDIALS ex16 ConductionOperator
added override keywords and removed unnecessary virtual specifications
2023-12-21 15:02:08 -08:00
IdoAkkerman b03cf507be Fix debug stuff 2023-12-21 17:04:02 +01:00
IdoAkkerman 066dc9b078 Us prev. unused variable 2023-12-21 15:16:29 +01:00
IdoAkkerman 70854254e7 Report Dofs in boundary for ex5 2023-12-21 14:45:42 +01:00
IdoAkkerman 646df28ac8 Add sign to Hdiv bdr dof indices 2023-12-21 14:45:09 +01:00
IdoAkkerman 73d4f987e4 Add direction to 2D Hdiv bdr indices 2023-12-21 14:14:39 +01:00
IdoAkkerman c59d519c89 Allow for negative dof indices 2023-12-21 14:12:21 +01:00
IdoAkkerman 1ec2cba9e8 Allow for negative dof indices in Table merge constructor 2023-12-21 14:11:11 +01:00
IdoAkkerman 26cc1f8387 Fix 2D curl 2023-12-21 09:40:38 +01:00
Christopher vogl 85fe35bec2 replaced c-style pointer main use in SUNDIALS ex16
-used std::unique_ptr and dynamic casting instead
-avoided changing ConductionOperator for comparison to MFEM ex16
2023-12-20 19:17:24 -08:00
IdoAkkerman b57fa2b127 Add neumann/periodic test case 2023-12-20 12:53:32 +01:00
IdoAkkerman 596909138a Cleaner bc selection/reporting in ex1 2023-12-20 12:53:07 +01:00
IdoAkkerman c11a76f2c1 Update make clean call 2023-12-19 16:52:34 +01:00
IdoAkkerman 51d32ad293 merge master - manual 2023-12-19 16:16:00 +01:00
IdoAkkerman a433e9e0b4 Fix small numbering issue 2023-12-19 14:33:07 +01:00
IdoAkkerman abac61f5b5 Add Hcurl boundary + allow GF read to set dim with seperate call 2023-12-19 14:32:26 +01:00
IdoAkkerman a121a9d186 Small order fix 2023-12-19 14:31:20 +01:00
IdoAkkerman 58ecbf6150 Use new ext and table modes 2023-12-19 14:30:23 +01:00
IdoAkkerman bad5ae41d1 Add bdr dof selection option 2023-12-19 14:29:35 +01:00
IdoAkkerman e773e07373 Make Table merge generic 2023-12-19 14:28:17 +01:00
IdoAkkerman c3ded3c003 Add dim for fe_coll and ess bc output 2023-12-19 14:27:47 +01:00
Will Pazner e706802ed5 Use new LDFLAGS only with new Mac linker 2023-12-15 12:12:32 -08:00
Will Pazner 9e97ad8bb3 Silence duplicate libraries linker warnings on Mac 2023-12-15 11:51:07 -08:00
IdoAkkerman 76e04c4606 Make style 2023-12-15 17:24:11 +01:00
IdoAkkerman 83ccf77d2f Fixing gitignore 2023-12-15 17:00:57 +01:00
IdoAkkerman a0d53975d8 Adding boundary elements to Hdiv collection pt2 2023-12-15 16:17:49 +01:00
IdoAkkerman 1252c0fbb9 Merge branch 'hdiv-nurbs' of /home/ido/Data/mfem/mfem into hdiv-nurbs 2023-12-15 16:14:34 +01:00
IdoAkkerman 04acf613ae Adding boundary elements to Hdiv collection 2023-12-15 16:14:20 +01:00
IdoAkkerman c8bddb8035 Merge branch 'master' into hdiv-nurbs 2023-12-15 14:38:11 +01:00
IdoAkkerman 467e83da31 Fixed nurbs miniapps 2023-12-15 14:36:31 +01:00
IdoAkkerman 8b29ef1335 style fix in fe_nurbs 2023-12-15 14:26:34 +01:00
IdoAkkerman 1076700714 Cmake fixes 2023-12-15 14:26:03 +01:00
IdoAkkerman 978f1155c5 Update Makefile 2023-12-15 14:25:34 +01:00
IdoAkkerman 3096d9d9cb Fixed some sloppiness 2023-12-15 13:51:27 +01:00
IdoAkkerman ffea75abb2 Fixed Table comments 2023-12-15 13:47:04 +01:00
IdoAkkerman a44a8b8789 Made booleans const 2023-12-15 13:43:09 +01:00
IdoAkkerman 05106096c3 Symmetrice delete call 2023-12-15 13:41:43 +01:00
IdoAkkerman ba7fd7a9a9 Remove superfluous hdiv check 2023-12-15 13:38:44 +01:00
IdoAkkerman 9b0e4e0085 Fix nurbs miniapp cmake 2023-12-15 13:38:25 +01:00
IdoAkkerman 3cb64f7f7e Merge branch 'hdiv-nurbs' of /home/ido/Data/mfem/mfem into hdiv-nurbs 2023-12-15 13:25:28 +01:00
IdoAkkerman 6642857437 Fix uninit error 2023-11-29 17:19:52 +01:00
IdoAkkerman 91f00d643a Fix shadow varaibel in fespace 2023-11-29 17:12:17 +01:00
IdoAkkerman ce8b62cfe7 Add override keyword for macosx 2 2023-11-29 16:51:03 +01:00
IdoAkkerman 4064bda60d Add override keyword for macosx 2023-11-29 16:50:15 +01:00
IdoAkkerman 4a21554986 Remove unused variables from table 2023-11-29 16:43:10 +01:00
IdoAkkerman 165968dc26 Make style 2023-11-29 16:40:48 +01:00
IdoAkkerman 9eb70f7be0 Add tests 2023-11-29 16:38:01 +01:00
IdoAkkerman 8797a9cb00 Small ex24 tweaks 2023-11-29 16:37:42 +01:00
IdoAkkerman 8366a5a6d6 Fix documentation error 2023-11-29 12:17:15 +01:00
IdoAkkerman 63a9d5749b Delete parallel miniapp 2023-11-29 12:16:58 +01:00
IdoAkkerman 0e6aa41245 Update nurbs miniapps 2023-11-29 12:11:37 +01:00
IdoAkkerman bbcb054814 example update 2023-11-29 12:08:38 +01:00
IdoAkkerman 2e5a86db7a Merge branch 'master' into hdiv-nurbs 2023-11-29 09:24:21 +01:00
IdoAkkerman 057b15cefb miniapps/nurbs/nurbs_ex24p.cpp 2023-11-28 16:52:13 +01:00
IdoAkkerman 7c06741f36 Remove debug print statement 2023-11-28 15:37:41 +01:00
IdoAkkerman 3a0c42aea5 Add L2 projection for NURBS 2023-11-28 15:37:16 +01:00
IdoAkkerman 75d5555a5f Component output not necessary anymore 2023-11-28 10:35:37 +01:00
IdoAkkerman 10c1ac9a66 Add nurbs hdiv hcurl examples/tests 2023-11-28 10:34:06 +01:00
IdoAkkerman 692e15c088 Make style 2023-11-28 09:57:11 +01:00
IdoAkkerman 6be9665bfb Add mappings to VShape Trans calls 2023-11-27 17:24:23 +01:00
IdoAkkerman 9be617d754 Add Hdiv and Hcurl NURBS to fe coll selection mechanism 2023-11-27 16:58:10 +01:00
IdoAkkerman ec4f37fe25 Make style 2023-11-27 16:57:28 +01:00
IdoAkkerman 6bb4ae9d50 Small correction in solenoidal test app 2023-11-27 16:57:00 +01:00
IdoAkkerman c7a3f188c0 Add output 2023-11-21 17:28:15 +01:00
IdoAkkerman 8e44509585 Make style 2023-11-21 17:27:26 +01:00
IdoAkkerman 787954715b Make style and small compile order fix nurbs fe 2023-11-21 17:26:30 +01:00
IdoAkkerman 9f03260dd2 Fix Hcurl boundary dof table 2 2023-11-21 17:24:23 +01:00
IdoAkkerman 5c3a3f7fdf Add Curl miniapp 2023-11-21 13:08:20 +01:00
IdoAkkerman 6693b22c83 Small compile fixes 2023-11-21 12:55:01 +01:00
IdoAkkerman 49c7f60a57 Add routines to make a Hcurl fespace 2023-11-21 12:49:37 +01:00
IdoAkkerman ace4608f10 Add H curl NURBS collection 2023-11-21 12:41:06 +01:00
IdoAkkerman 3899dfcc64 Remove interfaces -- will implement in follow-up PR 2023-11-21 12:40:39 +01:00
IdoAkkerman 2dc98e9153 Add H curl NURBS elements 2023-11-21 12:39:47 +01:00
IdoAkkerman 90f8a2409f Add divergence free test case 2023-11-21 11:59:37 +01:00
IdoAkkerman 92dc0db889 Fix patch check 2023-11-21 10:23:42 +01:00
IdoAkkerman d8fc48e608 Clean up of projection miniapp 2023-11-20 14:21:05 +01:00
IdoAkkerman 611802f990 Rename extension routine 2023-11-20 14:18:27 +01:00
IdoAkkerman 606df86b0a Component Extension generator 2023-11-20 14:08:27 +01:00
IdoAkkerman 2f7d38e6f6 Fix typo 2023-11-20 14:00:19 +01:00
IdoAkkerman b90d665ced Clean fe collection 2023-11-20 14:00:04 +01:00
IdoAkkerman 50dd77ffd4 Make style 2023-11-20 13:53:10 +01:00
IdoAkkerman 5e5b79783c Other comment -- make style 2023-11-20 13:51:37 +01:00
IdoAkkerman c815114661 Add explaination to new table constructors 2023-11-20 13:48:53 +01:00
IdoAkkerman 93f6a53201 FES cleaning, renaming and memleak fix 2023-11-20 13:35:55 +01:00
IdoAkkerman a3ee0cfe79 Add 3D gradient and hessian -- compile fixes 2023-11-20 12:06:55 +01:00
IdoAkkerman 95b0178514 Add 3D gradient and hessian 2023-11-20 11:40:34 +01:00
IdoAkkerman afacf3db45 Add 3D function 2023-11-17 17:58:20 +01:00
IdoAkkerman 4a80321420 Add 3D output 2023-11-17 17:52:35 +01:00
IdoAkkerman 9295c69249 Hdiv 3D fixes 2023-11-17 17:44:58 +01:00
IdoAkkerman 893f04967c add geom option to NURBS HDiv fecoll 2023-11-17 17:16:05 +01:00
IdoAkkerman fa410a6e02 Correct type in fespace 2023-11-17 17:15:19 +01:00
IdoAkkerman aed2687743 Add 3D HDiv elements 2023-11-17 17:03:08 +01:00
IdoAkkerman 978c0d10bc Add 3D to fespace 2023-11-17 16:41:43 +01:00
IdoAkkerman 25b540e804 Add 3D table merge 2023-11-17 16:41:06 +01:00
IdoAkkerman 9aa58cd5c2 Modify fe space to accomodate Hdiv NURBS 2023-11-17 15:38:39 +01:00
IdoAkkerman 691be01bcc Add constructors to table that merge existing tables 2023-11-17 15:38:10 +01:00
IdoAkkerman 402ed45ee4 Add Hdiv fe collection 2023-11-17 15:37:31 +01:00
IdoAkkerman 07dfcd83b9 Add 2D Hdiv NURBS basis 2023-11-17 15:37:10 +01:00
IdoAkkerman 9845dfda2c Add VectorBasis derivative interfaces 2023-11-17 15:36:42 +01:00
IdoAkkerman da5ee77e61 Add div free option 2023-11-17 14:07:06 +01:00
IdoAkkerman e1d2966e42 Add neumann bcs to nurbs_ex1 miniapp 2023-11-17 14:05:09 +01:00
IdoAkkerman 17428ce198 Add div-free option 2023-11-17 13:21:40 +01:00
IdoAkkerman e87e790215 add nurbs solenoidal miniapp for checking Hdif elemenet 2023-11-16 16:46:41 +01:00
Socratis Petrides a6d4e17911 Transfer Operator as a SparseMatrix 2022-02-25 13:50:07 -08:00
Socratis Petrides bde7846b5a style 2022-02-24 17:43:11 -08:00
Socratis Petrides 01283767a6 variable order href transfer for the 'ANY_TYPE' transfer operator 2022-02-24 17:42:41 -08:00
254 changed files with 27202 additions and 2955 deletions
+11
View File
@@ -272,16 +272,27 @@ miniapps/navier/*_output
miniapps/nurbs/nurbs_ex1
miniapps/nurbs/nurbs_ex1p
miniapps/nurbs/nurbs_ex3
miniapps/nurbs/nurbs_ex5
miniapps/nurbs/nurbs_ex11p
miniapps/nurbs/nurbs_ex24
miniapps/nurbs/nurbs_solenoidal
miniapps/nurbs/nurbs_printfunc
miniapps/nurbs/nurbs_patch_ex1
miniapps/nurbs/nurbs_curveint
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/sol_?.gf
miniapps/nurbs/sol.*
miniapps/nurbs/mode_*
miniapps/nurbs/Example1*
miniapps/nurbs/Example3*
miniapps/nurbs/Example5*
miniapps/nurbs/Solenoidal*
miniapps/nurbs/ParaView
miniapps/nurbs/sin-fit.mesh
miniapps/nurbs/ex5.mesh
miniapps/nurbs/exsol.mesh
miniapps/nurbs/CurveInt
miniapps/nurbs/nurbs_naca_cmesh
miniapps/nurbs/naca-cmesh.mesh
+5 -5
View File
@@ -22,7 +22,7 @@ include:
# the "needs" keyword and express the DAG of jobs for more efficiency.
# - We use setup and setup_baseline phases to download content outside of mfem
# directory.
# - Allocate/Release is where quartz resource are allocated/released once for all.
# - Allocate/Release is where ruby resource are allocated/released once for all.
# - Build and Test is where we build and MFEM for multiple toolchains.
# - Baseline_checks gathers baseline-type test suites execution
# - Baseline_publish, only available on master, allows to update baseline
@@ -53,7 +53,7 @@ variables:
AUTOTEST_COMMIT: "YES"
# Trigger subpipelines:
quartz-build-and-test:
ruby-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
@@ -61,10 +61,10 @@ quartz-build-and-test:
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/quartz-build-and-test.yml
include: .gitlab/ruby-build-and-test.yml
strategy: depend
quartz-baseline:
ruby-baseline:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
@@ -73,7 +73,7 @@ quartz-baseline:
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/quartz-baseline.yml
include: .gitlab/ruby-baseline.yml
strategy: depend
lassen-build-and-test:
+3 -3
View File
@@ -24,7 +24,7 @@ and `test type`.
Machines typically include:
* Quartz: Intel bi-socket x86
* Ruby: 2nd Gen Intel Xeon (Cascade Lake)
* Lassen: Power9 + Nvidia GPU
* Corona: AMD GPU
@@ -76,13 +76,13 @@ with a spack spec of MFEM, within the limits permitted by the MFEM spack
package.
In any build-and-test sub-pipeline a job basically consists in defining the
spack spec to use. Adding a job on quartz for example resumes to:
spack spec to use. Adding a job on ruby for example resumes to:
```yaml
<job_name>:
variables:
SPEC: "<spack_spec>"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
```
The remaining and non trivial work is to make sure this spec is working. To
+1 -1
View File
@@ -24,7 +24,7 @@ variables:
# TODO: add a clean-up mechanism
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${MACHINE_NAME}-pipeline-${CI_PIPELINE_ID}
# On LLNL's quartz, there is only one allocation shared among jobs in order to
# On LLNL's ruby, there is only one allocation shared among jobs in order to
# save time and resource. This allocation has to be uniquely named so that we
# are sure to retrieve it.
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
@@ -9,17 +9,17 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Quartz machine at LLNL
# GitLab pipelines configurations for the Ruby machine at LLNL
variables:
MACHINE_NAME: quartz
MACHINE_NAME: ruby
.on_quartz:
.on_ruby:
tags:
- shell
- quartz
- ruby
rules:
# Don't run quartz jobs if...
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_QUARTZ == "OFF"'
# Don't run ruby jobs if...
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_RUBY == "OFF"'
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
@@ -40,13 +40,13 @@ variables:
- when: on_success
# Spack helped builds
# Generic quartz build job, extending build script
.build_and_test_on_quartz:
extends: [.on_quartz]
# Generic ruby build job, extending build script
.build_and_test_on_ruby:
extends: [.on_ruby]
stage: build_and_test
script:
# THREADS is used by 'tests/gitlab/build_and_test', run below
- export THREADS=12
- export THREADS=16
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
+1 -1
View File
@@ -18,7 +18,7 @@
setup_baseline:
tags:
- shell
- quartz
- ruby
stage: setup
variables:
GIT_STRATEGY: none
+1 -1
View File
@@ -16,7 +16,7 @@
setup:
tags:
- shell
- quartz
- ruby
stage: setup
variables:
GIT_STRATEGY: none
@@ -19,8 +19,8 @@ stages:
- cleanup
- baseline_publish
baselinecheck_mfem_intel_quartz:
extends: [.on_quartz]
baselinecheck_mfem_intel_ruby:
extends: [.on_ruby]
stage: baseline_check
variables:
# TPLS_DIR is used in .gitlab/scripts/baseline to provide the tpls location
@@ -32,7 +32,7 @@ baselinecheck_mfem_intel_quartz:
- echo ${BUILD_ROOT}
- echo ${TPLS_DIR}
# Used by the tests in MFEM/tests:
- export MFEM_TEST_NP=32
- export MFEM_TEST_NP=48
# The next script uses the following environment variables:
# * BASELINE_TEST, SYS_TYPE, CI_PROJECT_DIR, ARTIFACTS_DIR,
# * BUILD_ROOT, TPLS_DIR, MACHINE_NAME
@@ -44,18 +44,16 @@ baselinecheck_mfem_intel_quartz:
allow_failure: true
cleanup:
extends: .on_quartz
extends: .on_ruby
stage: cleanup
variables:
GIT_STRATEGY: none
script:
- echo "BUILD_ROOT=${BUILD_ROOT}"
- rm -rf "${BUILD_ROOT}" || true
- echo "CI_PROJECT_DIR=${CI_PROJECT_DIR}"
- make -C "${CI_PROJECT_DIR}" distclean
report_baseline:
extends: [.on_quartz]
extends: [.on_ruby]
stage: baseline_report
script:
- echo ${MACHINE_NAME}
@@ -115,8 +113,8 @@ report_baseline:
exit $err
) 9> autotest.lock
baselinepublish_mfem_quartz:
extends: [.on_quartz]
baselinepublish_mfem_ruby:
extends: [.on_ruby]
stage: baseline_publish
rules:
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "YES"'
@@ -131,5 +129,5 @@ baselinepublish_mfem_quartz:
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/quartz-config.yml
- local: .gitlab/configs/ruby-config.yml
- local: .gitlab/configs/setup-baseline.yml
@@ -19,54 +19,54 @@ stages:
allocate_resource:
variables:
GIT_STRATEGY: none
extends: .on_quartz
extends: .on_ruby
stage: allocate_resource
script:
- echo ${ALLOC_NAME}
- salloc --exclusive --nodes=1 --reservation=ci --time=60 --no-shell --job-name=${ALLOC_NAME}
timeout: 6h
# GitLab jobs for the Quartz machine at LLNL
# GitLab jobs for the Ruby machine at LLNL
debug_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug~mpi"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
debug_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug+mpi"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
opt_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 ~mpi"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
opt_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
opt_par_gcc_10_sundials:
variables:
SPEC: "%gcc@10.3.1 +sundials"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
opt_par_gcc_10_petsc:
variables:
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
opt_par_gcc_10_pumi:
variables:
SPEC: "%gcc@10.3.1 +pumi"
extends: .build_and_test_on_quartz
extends: .build_and_test_on_ruby
# Release
release_resource:
variables:
GIT_STRATEGY: none
extends: .on_quartz
extends: .on_ruby
stage: release_resource_and_report
script:
- echo ${ALLOC_NAME}
@@ -78,17 +78,17 @@ release_resource:
report_job_success:
stage: release_resource_and_report
extends:
- .on_quartz
- .on_ruby
- .report_job_success
report_job_failure:
stage: release_resource_and_report
extends:
- .on_quartz
- .on_ruby
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/quartz-config.yml
- local: .gitlab/configs/ruby-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
+4 -4
View File
@@ -14,7 +14,7 @@
# locals
glob_err=${BASELINE_TEST}.err
base=${BASELINE_TEST}-${SYS_TYPE}
if [[ "${MACHINE_NAME}" == "quartz" ]]; then
if [[ "${MACHINE_NAME}" == "ruby" ]]; then
base="${BASELINE_TEST}-${MACHINE_NAME}"
fi
base_diff=${base}.diff
@@ -31,8 +31,8 @@ cd tests
mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
# run
if [[ "${MACHINE_NAME}" == "quartz" || "${MACHINE_NAME}" == "ruby" ]]; then
salloc --nodes=1 --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
if [[ "${MACHINE_NAME}" == "ruby" ]]; then
salloc --nodes=1 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "lassen" ]]; then
@@ -41,11 +41,11 @@ else
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
exit 1
fi
status="$?"
# post
mkdir ${artifacts_path}
status=0
if [[ -f ${BASELINE_TEST}.out ]]; then
cp ${BASELINE_TEST}.out ${artifacts_path}
fi
+2 -2
View File
@@ -11,7 +11,7 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# There will be collision between corona and quartz baselines.
# There will be collision between corona and ruby baselines.
# Once the corresponding files have been generated, we can switch to machine
# specific ref.
ARTIFACT_PATH=${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/baseline-${SYS_TYPE}
@@ -21,7 +21,7 @@ PATCH_FILE=${ARTIFACT_PATH}.patch
FULL_FILE=${ARTIFACT_PATH}.out
DIFF_FILE=${ARTIFACT_PATH}.diff
# There will be collision between corona and quartz baselines.
# There will be collision between corona and ruby baselines.
# Once the corresponding files have been generated, we can switch to machine
# specific ref.
SAVED_NAME=baseline-${SYS_TYPE}.saved
+42
View File
@@ -11,9 +11,48 @@
Version 4.7.1 (development)
===========================
Discretization improvements
---------------------------
- Added NURBS-based H(div) and H(curl) elements in 2D and 3D. Only on single
patch meshes. Only implemented for serial computations.
- Added support for boundary constraints to the hybridization class.
Meshing improvements
--------------------
- The ExodusII reader now handles pyramid and wedge element types. Mixed meshes
are also supported.
New and updated examples and miniapps
-------------------------------------
- Added miniapps to demonstrate the H(div) and H(curl) NURBS elements.
- Added an MFEM example for the eikonal equation. This new solver is based on
the proximal Galerkin method introduced by Keith and Surowiec.
GPU computing
-------------
- Added support for GPU-accelerated batched linear algebra (using cuBLAS,
hipBLAS, MAGMA, or native MFEM functionality) through the BatchedLinAlg class.
Miscellaneous
-------------
- Refactored the `ARKStepSolver` class (ARKODE interface) to use
`TimeDependentOperator::Mult` only when the associated ODE operator is
expressed in explicit form (i.e., `TimeDependentOperator::isExplicit()`),
otherwise `TimeDependentOperator::ExplicitMult` is used. A check has been
added to `ARKStepSolver` to verify that the associated ODE operator is not in
explicit form when a mass matrix solver is enabled via a call to either the
`UseMFEMMassLinearSolver` or `UseSundialsMassLinearSolver` methods. This is
because enabling a mass matrix solver assumes that F(u,k,t) = M k in the
associated ODE operator.
- Added support for custom interpolation procedure in FindPointsGSLIB.
API changes
-----------
- API change: in class GridFunction, 'fec' was renamed to 'fec_owned'.
Version 4.7, released on May 7, 2024
====================================
@@ -38,6 +77,9 @@ Meshing improvements
- Added support for internal boundary elements in nonconforming meshes.
- Added ExodusII output capability. The writer can handle first-order (Pyramid5,
Wedge6, Hex8, Tet4) and second-order FE types (Pyramid14, Wedge18, Hex27, Tet10).
- The ReadCubit Genesis mesh importer has been rewritten to improve readability.
Discretization improvements
+21 -5
View File
@@ -146,7 +146,9 @@ if (MFEM_USE_CUDA)
set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} ${CUDA_FLAGS}")
find_package(CUDAToolkit REQUIRED)
set(CUSPARSE_FOUND TRUE)
set(CUBLAS_FOUND TRUE)
get_target_property(CUSPARSE_LIBRARIES CUDA::cusparse LOCATION)
get_target_property(CUBLAS_LIBRARIES CUDA::cublas LOCATION)
endif()
if (XSDK_ENABLE_C)
@@ -231,6 +233,7 @@ if (MFEM_USE_HIP)
list(INSERT CMAKE_PREFIX_PATH 0 ${ROCM_PATH})
endif()
find_package(HIP REQUIRED)
find_package(HIPBLAS REQUIRED)
find_package(HIPSPARSE REQUIRED)
endif()
@@ -396,6 +399,10 @@ if (MFEM_USE_AMGX)
find_package(AMGX REQUIRED)
endif()
if (MFEM_USE_MAGMA)
find_package(MAGMA REQUIRED)
endif()
if (MFEM_USE_CONDUIT)
find_package(Conduit REQUIRED conduit relay blueprint)
endif()
@@ -515,7 +522,10 @@ endif()
# Enzyme
if (MFEM_USE_ENZYME)
find_package(ENZYME REQUIRED)
find_package(Enzyme REQUIRED HINTS ${ENZYME_DIR})
message(STATUS "Enzyme found in ${ENZYME_DIR}.")
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
set(ENZYME_FOUND 1)
endif()
# MFEM_TIMER_TYPE
@@ -557,8 +567,9 @@ find_package(Threads REQUIRED)
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
ADIOS2 CUSPARSE MKL_CPARDISO MKL_PARDISO AMGX CALIPER CODIPACK
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPSPARSE MOONOLITH BLITZ ALGOIM ENZYME)
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
ALGOIM ENZYME)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
@@ -621,6 +632,11 @@ set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX} CACHE PATH
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
if (MFEM_USE_ENZYME)
target_link_libraries(mfem PUBLIC ClangEnzymeFlags)
endif()
if (MINGW)
target_link_libraries(mfem PRIVATE ws2_32)
endif()
@@ -673,7 +689,7 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
#include \"${PROJECT_SOURCE_DIR}/${Header}\"
")
execute_process(COMMAND ${CMAKE_COMMAND} -E copy_if_different
execute_process(COMMAND ${CMAKE_COMMAND} -E copy_if_different
"${PROJECT_BINARY_DIR}/${Header}.tmp"
"${PROJECT_BINARY_DIR}/${Header}"
)
@@ -687,7 +703,7 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
#include \"mfem/${Header}\"
")
execute_process(COMMAND ${CMAKE_COMMAND} -E copy_if_different
execute_process(COMMAND ${CMAKE_COMMAND} -E copy_if_different
"${PROJECT_BINARY_DIR}/InstallHeaders/${Header}.tmp"
"${PROJECT_BINARY_DIR}/InstallHeaders/${Header}"
)
+17 -1
View File
@@ -273,7 +273,13 @@ Installation options:
PREFIX - Specify the installation directory. The library (libmfem.a) will be
installed in $(PREFIX)/lib, the headers in $(PREFIX)/include, and
the configuration makefile (config.mk) in $(PREFIX)/share/mfem.
INSTALL - Specify the install program, e.g /usr/bin/install
INSTALL - Specify the install program, default = /usr/bin/install
INSTALL_DEF_PERM - Specify the default install permissions. This affects
headers and configuration makefiles, default = 644
INSTALL_BIN_PERM - Specify the install permissions for binaries. This only
affects the shared version of the library, default = 755
INSTALL_DIR_PERM - Specify the install permissions for directories and,
on macOS/BSD, for symlinks as well, default = 755
MFEM library features/options (GNU make)
----------------------------------------
@@ -388,6 +394,11 @@ MFEM_USE_AMGX = YES/NO
Allows the user to use SparseMatrices and HypreParMatrices to solve linear
systems with the routines from the AmgX library.
MFEM_USE_MAGMA = YES/NO
Enable MFEM functionality based on the MAGMA high-performance linear algebra
library. The MAGMA library provides a BLAS/LAPACK interface, with
implementations that have been optimized for Nvidia and AMD GPUs.
MFEM_USE_GNUTLS = YES/NO
Enable secure socket support in class socketstream, using the auxiliary
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
@@ -699,6 +710,11 @@ The specific libraries and their options are:
Options: AMGX_OPT, AMGX_LIB.
Versions: AmgX >= 2.1, older versions may work too.
- MAGMA (optional), used with MFEM_USE_MAGMA = YES.
URL: https://icl.utk.edu/magma/
Options: MAGMA_OPT, MAGMA_LIB
Versions: MAGMA >= 2.8.0
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
one can get the library through the Homebrew package manager (http://brew.sh).
+1
View File
@@ -37,6 +37,7 @@ set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
set(MFEM_USE_HIOP @MFEM_USE_HIOP@)
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
set(MFEM_USE_GSLIB @MFEM_USE_GSLIB@)
+3
View File
@@ -114,6 +114,9 @@
// Enable MFEM functionality based on the AmgX library.
#cmakedefine MFEM_USE_AMGX
// Enable MFEM functionality based on the MAGMA library.
#cmakedefine MFEM_USE_MAGMA
// Enable secure socket streams based on the GNUTLS library.
#cmakedefine MFEM_USE_GNUTLS
-27
View File
@@ -1,27 +0,0 @@
# Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
message(STATUS "Looking for ENZYME ...")
message(STATUS " in ENZYME_DIR = ${ENZYME_DIR}")
# Make sure the directory and version combination works. Do nothing otherwise.
if(EXISTS "${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
message(STATUS "Found ENZYME: ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
# Set ENZYME_FOUND
set(ENZYME_FOUND TRUE CACHE BOOL "ENZYME was found." FORCE)
# Set CXX flags to accommodate the Enzyme Clang plugin
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Xclang -load -Xclang ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so -mllvm -enzyme-loose-types=1")
set(MFEM_USE_ENZYME YES)
else()
endif()
+37
View File
@@ -0,0 +1,37 @@
# Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables:
# - MAGMA_FOUND
# - MAGMA_LIBRARIES
# - MAGMA_INCLUDE_DIRS
include(MfemCmakeUtilities)
mfem_find_package(MAGMA MAGMA MAGMA_DIR "include" "magma.h" "lib" "magma"
"Paths to headers required by MAGMA." "Libraries required by MAGMA.")
if (MAGMA_FOUND AND MFEM_USE_CUDA)
get_target_property(CUSPARSE_LIBRARIES CUDA::cusparse LOCATION)
get_target_property(CUBLAS_LIBRARIES CUDA::cublas LOCATION)
list(APPEND MAGMA_LIBRARIES ${CUSPARSE_LIBRARIES} ${CUBLAS_LIBRARIES})
set(MAGMA_LIBRARIES ${MAGMA_LIBRARIES} CACHE STRING
"MAGMA libraries + dependencies." FORCE)
message(STATUS "Updated MAGMA_LIBRARIES: ${MAGMA_LIBRARIES}")
endif()
if (MAGMA_FOUND AND MFEM_USE_HIP)
find_package(HIPBLAS REQUIRED)
find_package(HIPSPARSE REQUIRED)
list(APPEND MAGMA_LIBRARIES ${HIPBLAS_LIBRARIES} ${HIPSPARSE_LIBRARIES})
set(MAGMA_LIBRARIES ${MAGMA_LIBRARIES} CACHE STRING
"MAGMA libraries + dependencies." FORCE)
message(STATUS "Updated MAGMA_LIBRARIES: ${MAGMA_LIBRARIES}")
endif()
@@ -846,14 +846,14 @@ function(mfem_export_mk_files)
MFEM_USE_ZLIB MFEM_USE_LIBUNWIND MFEM_USE_LAPACK MFEM_THREAD_SAFE
MFEM_USE_LEGACY_OPENMP MFEM_USE_OPENMP MFEM_USE_MEMALLOC MFEM_USE_SUNDIALS
MFEM_USE_SUITESPARSE MFEM_USE_SUPERLU MFEM_USE_SUPERLU5 MFEM_USE_MUMPS
MFEM_USE_STRUMPACK MFEM_USE_GINKGO MFEM_USE_AMGX MFEM_USE_GNUTLS
MFEM_USE_NETCDF MFEM_USE_PETSC MFEM_USE_SLEPC MFEM_USE_MPFR MFEM_USE_SIDRE
MFEM_USE_FMS MFEM_USE_CONDUIT MFEM_USE_PUMI MFEM_USE_HIOP MFEM_USE_GSLIB
MFEM_USE_CUDA MFEM_USE_HIP MFEM_USE_RAJA MFEM_USE_OCCA MFEM_USE_CEED
MFEM_USE_CALIPER MFEM_USE_UMPIRE MFEM_USE_SIMD MFEM_USE_ADIOS2
MFEM_USE_MKL_CPARDISO MFEM_USE_MKL_PARDISO MFEM_USE_ADFORWARD
MFEM_USE_CODIPACK MFEM_USE_BENCHMARK MFEM_USE_PARELAG MFEM_USE_TRIBOL
MFEM_USE_MOONOLITH MFEM_USE_ALGOIM MFEM_USE_ENZYME)
MFEM_USE_STRUMPACK MFEM_USE_GINKGO MFEM_USE_AMGX MFEM_USE_MAGMA
MFEM_USE_GNUTLS MFEM_USE_NETCDF MFEM_USE_PETSC MFEM_USE_SLEPC
MFEM_USE_MPFR MFEM_USE_SIDRE MFEM_USE_FMS MFEM_USE_CONDUIT MFEM_USE_PUMI
MFEM_USE_HIOP MFEM_USE_GSLIB MFEM_USE_CUDA MFEM_USE_HIP MFEM_USE_RAJA
MFEM_USE_OCCA MFEM_USE_CEED MFEM_USE_CALIPER MFEM_USE_UMPIRE MFEM_USE_SIMD
MFEM_USE_ADIOS2 MFEM_USE_MKL_CPARDISO MFEM_USE_MKL_PARDISO
MFEM_USE_ADFORWARD MFEM_USE_CODIPACK MFEM_USE_BENCHMARK MFEM_USE_PARELAG
MFEM_USE_TRIBOL MFEM_USE_MOONOLITH MFEM_USE_ALGOIM MFEM_USE_ENZYME)
foreach(var ${CONFIG_MK_BOOL_VARS})
if (${var})
set(${var} YES)
+3
View File
@@ -114,6 +114,9 @@
// Enable MFEM functionality based on the AmgX library.
// #define MFEM_USE_AMGX
// Enable MFEM functionality based on the MAGMA library.
// #define MFEM_USE_MAGMA
// Enable secure socket streams based on the GNUTLS library.
// #define MFEM_USE_GNUTLS
+1
View File
@@ -38,6 +38,7 @@ MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_AMGX = @MFEM_USE_AMGX@
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
MFEM_USE_PETSC = @MFEM_USE_PETSC@
+6 -1
View File
@@ -40,6 +40,7 @@ option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
option(MFEM_USE_GSLIB "Enable GSLIB usage" OFF)
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
@@ -183,6 +184,10 @@ set(Ginkgo_DIR "${MFEM_DIR}/../ginkgo" CACHE PATH "Path to the Ginkgo library.")
set(AMGX_DIR "${MFEM_DIR}/../amgx" CACHE PATH "Path to AmgX")
set(MAGMA_DIR "${MFEM_DIR}/../magma" CACHE PATH "Path to MAGMA")
set(MAGMA_REQUIRED_PACKAGES "BLAS" "LAPACK" CACHE STRING
"Additional packages required by MAGMA.")
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
set(GSLIB_DIR "" CACHE PATH "Path to the GSLIB library.")
@@ -259,7 +264,7 @@ set(PARELAG_LIBRARIES "${PARELAG_DIR}/build/src/libParELAG.a" CACHE STRING
"The ParELAG library.")
set(TRIBOL_DIR "${MFEM_DIR}/../tribol" CACHE PATH "Path to Tribol")
set(Tribol_REQUIRED_PACKAGES "Axom/core/mint/slam/slic" CACHE STRING
set(Tribol_REQUIRED_PACKAGES "Axom/core/mint/slam/slic" CACHE STRING
"Additional packages required by Tribol")
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
+12 -2
View File
@@ -95,6 +95,10 @@ else
# Silence unused command line argument warnings when generating dependencies
# with mpicxx and clang
DEP_FLAGS := -Wno-unused-command-line-argument $(DEP_FLAGS)
# Silence "ignoring duplicate libraries" warnings on new (Xcode 15) linker
ifneq (,$(findstring PROJECT:dyld,$(shell ld -v 2>&1)))
LDFLAGS_INTERNAL = -Xlinker -no_warn_duplicate_libraries
endif
endif
# Set CXXFLAGS to overwrite the default selection of DEBUG_FLAGS/OPTIM_FLAGS
@@ -139,6 +143,7 @@ MFEM_USE_MUMPS = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_GINKGO = NO
MFEM_USE_AMGX = NO
MFEM_USE_MAGMA = NO
MFEM_USE_GNUTLS = NO
MFEM_USE_NETCDF = NO
MFEM_USE_PETSC = NO
@@ -390,6 +395,11 @@ AMGX_DIR = @MFEM_DIR@/../amgx
AMGX_OPT = -I$(AMGX_DIR)/include
AMGX_LIB = -L$(AMGX_DIR)/lib -lamgx -lcusparse -lcusolver -lcublas -lnvToolsExt
# MAGMA library configuration
MAGMA_DIR = @MFEM_DIR@/../magma
MAGMA_OPT = -I$(MAGMA_DIR)/include
MAGMA_LIB = -L$(MAGMA_DIR)/lib -l:libmagma.a -lcublas -lcusparse $(LAPACK_LIB)
# GnuTLS library configuration
GNUTLS_OPT =
GNUTLS_LIB = -lgnutls
@@ -497,11 +507,11 @@ GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
# CUDA library configuration
CUDA_OPT =
CUDA_LIB = -lcusparse
CUDA_LIB = -lcusparse -lcublas
# HIP library configuration
HIP_OPT =
HIP_LIB = -L$(HIP_DIR)/lib $(XLINKER)-rpath,$(HIP_DIR)/lib -lhipsparse
HIP_LIB = -L$(HIP_DIR)/lib $(XLINKER)-rpath,$(HIP_DIR)/lib -lhipsparse -lhipblas
# OCCA library configuration
OCCA_DIR = @MFEM_DIR@/../occa
+83 -13
View File
@@ -32,7 +32,7 @@ groups_serial=(
'"examples"
"Examples:"
"examples"
"ex{,1,2,3}[0-9].cpp"'
"ex{,[1-9]}[0-9].cpp"'
# "ex1.cpp"'
'"sundials"
"SUNDIALS examples:"
@@ -58,6 +58,10 @@ groups_serial=(
"HiOp examples:"
"examples/hiop"
"ex9.cpp"'
'"moonolith"
"Moonolith examples:"
"examples/moonolith"
"ex1.cpp"'
'"pumi"
"PUMI examples:"
"examples/pumi"
@@ -66,25 +70,38 @@ groups_serial=(
'"meshing"
"Meshing miniapps:"
"miniapps/meshing"
"mobius-strip.cpp klein-bottle.cpp extruder.cpp toroid.cpp
"mobius-strip.cpp klein-bottle.cpp extruder.cpp toroid.cpp mesh-quality.cpp
polar-nc.cpp reflector.cpp shaper.cpp trimmer.cpp twist.cpp
mesh-optimizer.cpp minimal-surface.cpp"'
'"adjoint"
"Adjoint miniapps:"
"miniapps/adjoint"
"cvsRoberts_ASAi_dns.cpp"'
'"autodiff"
"Autodiff miniapps:"
"miniapps/autodiff"
"seq_example.cpp seq_test.cpp"' # 'seq_test.cpp' has no sample runs
'"dpg"
"DPG miniapps:"
"miniapps/dpg"
"{acoustics,convection-diffusion,diffusion,maxwell}.cpp"'
'"gslib"
"GSLIB miniapps:"
"miniapps/gslib"
"field-diff.cpp field-interp.cpp findpts.cpp schwarz_ex1.cpp "'
# todo: miniapps/mtop
'"nurbs"
"NURBS miniapps:"
"miniapps/nurbs"
"nurbs_ex1.cpp"'
# todo: add other nurbs miniapps
# todo: miniapps/solvers (serial)
'"tools"
"Tools miniapps:"
"miniapps/tools"
"convert-dc.cpp display-basis.cpp get-values.cpp load-dc.cpp
lor-transfer.cpp"'
# todo: add other tools miniapps
'"toys"
"Toys miniapps:"
"miniapps/toys"
@@ -100,7 +117,7 @@ groups_parallel=(
'"examples"
"Examples:"
"examples"
"ex{,1,2,3}[0-9]p.cpp"'
"ex{,[1-9]}[0-9]p.cpp"'
# "ex1p.cpp"'
'"sundials"
"SUNDIALS examples:"
@@ -126,6 +143,10 @@ groups_parallel=(
"HiOp examples:"
"examples/hiop"
"ex9p.cpp"'
'"moonolith"
"Moonolith examples:"
"examples/moonolith"
"ex{1,2}p.cpp"'
'"pumi"
"PUMI examples:"
"examples/pumi"
@@ -138,24 +159,41 @@ groups_parallel=(
'"meshing"
"Meshing miniapps:"
"miniapps/meshing"
"pmesh-optimizer.cpp pmesh-fitting.cpp pminimal-surface.cpp"'
"pmesh-optimizer.cpp pmesh-fitting.cpp pminimal-surface.cpp
fit-node-position.cpp"'
'"electromagnetics"
"Electromagnetics miniapps:"
"miniapps/electromagnetics"
"joule.cpp"'
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
# "{joule,maxwell,tesla,volta}.cpp"' # todo: multiline sample runs
'"adjoint"
"Adjoint miniapps:"
"miniapps/adjoint"
"adjoint_advection_diffusion.cpp"'
'"autodiff"
"Autodiff miniapps:"
"miniapps/autodiff"
"par_example.cpp"'
'"dpg"
"DPG miniapps:"
"miniapps/dpg"
"p{acoustics,convection-diffusion,diffusion,maxwell}.cpp"'
'"gslib"
"GSLIB miniapps:"
"miniapps/gslib"
"pfindpts.cpp schwarz_ex1p.cpp"'
'"hdiv-linear-solver"
"H(div) linear solver miniapps:"
"miniapps/hdiv-linear-solver"
"grad_div.cpp darcy.cpp"'
# 'miniapps/hooke/hooke.cpp' has no sample runs
# todo: miniapps/mtop
# todo: miniapps/multidomain
'"navier"
"Navier miniapps:"
"miniapps/navier"
"navier_cht.cpp"'
# todo: add other navier miniapps
'"nurbs"
"NURBS miniapps:"
"miniapps/nurbs"
@@ -164,14 +202,18 @@ groups_parallel=(
"Shifted miniapps:"
"miniapps/shifted"
"distance.cpp"'
# todo: add other shifted miniapps
'"solvers"
"Solvers miniapps:"
"miniapps/solvers"
"block-solvers.cpp"'
# todo: add other solvers miniapps
# todo: miniapps/spde
'"tools"
"Tools miniapps:"
"miniapps/tools"
"convert-cd.cpp get-values.cpp load-dc.cpp"'
"convert-dc.cpp get-values.cpp load-dc.cpp"'
# todo: add other tools miniapps
'"convergence"
"Convergence tests:"
"tests/convergence"
@@ -186,7 +228,7 @@ groups_all=(
'"examples"
"Examples:"
"examples"
"ex\"{,1,2,3}[0-9]\"{,p}.cpp"'
"ex\"{,[1-9]}[0-9]\"{,p}.cpp"'
'"sundials"
"SUNDIALS examples:"
"examples/sundials"
@@ -215,10 +257,14 @@ groups_all=(
"HiOp examples:"
"examples/hiop"
"ex9.cpp ex9p.cpp"'
'"moonolith"
"Moonolith examples:"
"examples/moonolith"
"ex1.cpp ex{1,2}p.cpp"'
'"pumi"
"PUMI examples:"
"examples/pumi"
"ex1.cpp ex1p.cpp ex2.cpp ex6p.cpp"'
"ex1.cpp ex2.cpp ex1p.cpp ex6p.cpp"'
'"superlu"
"Superlu examples:"
"examples/superlu"
@@ -226,43 +272,67 @@ groups_all=(
'"meshing"
"Meshing miniapps:"
"miniapps/meshing"
"mobius-strip.cpp klein-bottle.cpp extruder.cpp toroid.cpp
{,p}mesh-optimizer.cpp pmesh-fitting.cpp {,p}minimal-surface.cpp"'
"mobius-strip.cpp klein-bottle.cpp extruder.cpp toroid.cpp mesh-quality.cpp
polar-nc.cpp reflector.cpp shaper.cpp trimmer.cpp twist.cpp
{,p}mesh-optimizer.cpp pmesh-fitting.cpp {,p}minimal-surface.cpp
fit-node-position.cpp"'
'"electromagnetics"
"Electromagnetics miniapps:"
"miniapps/electromagnetics"
"joule.cpp"'
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
# "{joule,maxwell,tesla,volta}.cpp"' # todo: multiline sample runs
'"adjoint"
"Adjoint miniapps:"
"miniapps/adjoint"
"adjoint_advection_diffusion.cpp cvsRoberts_ASAi_dns.cpp"'
"cvsRoberts_ASAi_dns.cpp adjoint_advection_diffusion.cpp"'
'"autodiff"
"Autodiff miniapps:"
"miniapps/autodiff"
"seq_example.cpp seq_test.cpp par_example.cpp"'
# 'seq_test.cpp' has no sample runs
'"dpg"
"DPG miniapps:"
"miniapps/dpg"
"{,p}{acoustics,convection-diffusion,diffusion,maxwell}.cpp"'
'"gslib"
"GSLIB miniapps:"
"miniapps/gslib"
"field-diff.cpp field-interp.cpp findpts.cpp schwarz_ex1.cpp pfindpts.cpp
schwarz_ex1p.cpp"'
'"hdiv-linear-solver"
"H(div) linear solver miniapps:"
"miniapps/hdiv-linear-solver"
"grad_div.cpp darcy.cpp"'
# 'miniapps/hooke/hooke.cpp' has no sample runs
# todo: miniapps/mtop
# todo: miniapps/multidomain
'"navier"
"Navier miniapps:"
"miniapps/navier"
"navier_cht.cpp"'
# todo: add other navier miniapps
'"nurbs"
"NURBS miniapps:"
"miniapps/nurbs"
"nurbs_ex1.cpp nurbs_ex1p.cpp nurbs_ex11p.cpp"'
# todo: add other nurbs miniapps
'"shifted"
"Shifted miniapps:"
"miniapps/shifted"
"distance.cpp"'
# todo: add other shifted miniapps
'"solvers"
"Solvers miniapps:"
"miniapps/solvers"
"block-solvers.cpp"'
# todo: add other solvers miniapps
# todo: miniapps/spde
'"tools"
"Tools miniapps:"
"miniapps/tools"
"convert-dc.cpp display-basis.cpp get-values.cpp load-dc.cpp
lor-transfer.cpp"'
# todo: add other tools miniapps
'"toys"
"Toys miniapps:"
"miniapps/toys"
@@ -386,7 +456,7 @@ function help_message()
mfem_config [${mfem_config}]
Set MFEM configuration options
make [${make}], mpiexec [${mpiexec}], mpiexec_np [${mpiexec_np}]
Their values can also set using the respective uppercase environment
Their values can also be set using the respective uppercase environment
variable
mfem_build_dir [${mfem_build_dir}]
Same as '-d': set this variable to something different from <mfem_dir>
+3 -3
View File
@@ -18,9 +18,9 @@ elements
boundary
4
1 1 0 1
1 1 2 3
1 1 3 0
1 1 1 2
2 1 2 3
3 1 3 0
4 1 1 2
edges
4
+35
View File
@@ -0,0 +1,35 @@
MFEM mesh v1.0
#
# MFEM Geometry Types (see mesh/geom.hpp):
#
# POINT = 0
# SEGMENT = 1
# TRIANGLE = 2
# SQUARE = 3
# TETRAHEDRON = 4
# CUBE = 5
# PRISM = 6
#
dimension
2
elements
1
1 3 0 1 2 3
boundary
4
1 1 0 1
2 1 1 2
3 1 2 3
4 1 3 0
vertices
4
2
0 0
1 0.3
1.4 1.2
0.25 1.34
+3 -1
View File
@@ -938,6 +938,7 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/config \
@MFEM_SOURCE_DIR@/general \
@MFEM_SOURCE_DIR@/linalg \
@MFEM_SOURCE_DIR@/linalg/batched \
@MFEM_SOURCE_DIR@/linalg/simd \
@MFEM_SOURCE_DIR@/mesh \
@MFEM_SOURCE_DIR@/mesh/submesh \
@@ -1049,7 +1050,8 @@ RECURSIVE = NO
EXCLUDE = @MFEM_SOURCE_DIR@/config/_config.hpp \
@MFEM_SOURCE_DIR@/config/get_hypre_version.cpp \
@MFEM_SOURCE_DIR@/general/tinyxml2.h \
@MFEM_SOURCE_DIR@/general/tinyxml2.cpp
@MFEM_SOURCE_DIR@/general/tinyxml2.cpp \
@MFEM_SOURCE_DIR@/linalg/lapack.hpp
# The EXCLUDE_SYMLINKS tag can be used to select whether or not files or
# directories that are symbolic links (a Unix file system feature) are excluded
+15
View File
@@ -182,6 +182,21 @@ namespace mfem {
* <a class="el" href="examples_2superlu_2ex1p_8cpp_source.html">1p</a>,
* demonstrating the use of MFEM's \link superlu.hpp SuperLU integration\endlink.
*
* <H4>NURBS Examples</H4>
* - Variants of Examples
* <a class="el" href="nurbs__ex1_8cpp_source.html">1</a>,
* <a class="el" href="nurbs__ex1p_8cpp_source.html">1p</a>,
* <a class="el" href="nurbs__ex3_8cpp_source.html">3</a>,
* <a class="el" href="nurbs__ex5_8cpp_source.html">5</a>,
* <a class="el" href="nurbs__ex11p_8cpp_source.html">11p</a>, and
* <a class="el" href="nurbs__ex24_8cpp_source.html">24</a>,
* demonstrating howto perform NURBS-based Isogeometric Analysis.
* - Variant of Example <a class="el" href="nurbs__patch__ex1_8cpp_source.html">1</a>: demonstrates the use of patch integration
* - <a class="el" href="nurbs__solenoidal_8cpp_source.html">NURBS Divergence-free</a>: solve a solenoidal vector projection with NURBS-based H(div) elements
* - <a class="el" href="nurbs__curveint_8cpp_source.html">NURBS Interpolation</a>: NURBS interpolation of given geometry
* - <a class="el" href="nurbs__naca__cmesh_8cpp_source.html">NURBS NACA Mesher</a>: generate NURBS based mesh around a NACA foil
* - <a class="el" href="nurbs__printfunc_8cpp_source.html">NURBS Printer</a>: print the NURBS-basis
*
* <H3>Miniapps</H3>
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
+33
View File
@@ -50,6 +50,28 @@ list(APPEND ALL_EXE_SRCS
if (MFEM_USE_MPI)
list(APPEND ALL_EXE_SRCS
dfem_poisson.cpp
dfem_stokes.cpp
enzyme_interface_smoketest.cpp
test_dfem_dual.cpp
test_dfem.cpp
dfem_laghos.cpp
dfem_minimal_example.cpp
dfem_test_diffusion_2d.cpp
dfem_test_diffusion_3d.cpp
dfem_test_diffusion_3d_refactor.cpp
dfem_test_ordering.cpp
dfem_test_vector_diffusion.cpp
dfem_test_elasticity.cpp
dfem_test_nonlinear_elasticity_3d.cpp
dfem_test_nonlinear_diffusion_3d.cpp
dfem_test_interpolate_linear_scalar.cpp
dfem_test_interpolate_linear_scalar_3d.cpp
dfem_test_interpolate_gradient_linear_scalar_3d.cpp
dfem_test_mass_scalar_3d.cpp
dfem_test_mass_scalar_2d.cpp
dfem_test_interpolate_linear_vector.cpp
dfem_test_interpolate_linear_vector_3d.cpp
ex0p.cpp
ex1p.cpp
ex2p.cpp
@@ -110,6 +132,17 @@ include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Add one executable per cpp file
add_mfem_examples(ALL_EXE_SRCS)
target_link_libraries(dfem_poisson ClangEnzymeFlags)
target_link_libraries(dfem_stokes ClangEnzymeFlags)
target_link_libraries(enzyme_interface_smoketest ClangEnzymeFlags)
target_link_libraries(test_dfem ClangEnzymeFlags)
target_link_libraries(dfem_laghos ClangEnzymeFlags)
target_link_libraries(dfem_minimal_example ClangEnzymeFlags)
target_link_libraries(dfem_test_diffusion_3d ClangEnzymeFlags)
target_link_libraries(dfem_test_diffusion_3d_refactor ClangEnzymeFlags)
target_link_libraries(dfem_test_nonlinear_diffusion_3d ClangEnzymeFlags)
target_link_libraries(dfem_test_nonlinear_elasticity_3d ClangEnzymeFlags)
# Add a test for each example
if (MFEM_ENABLE_TESTING)
foreach(SRC_FILE ${ALL_EXE_SRCS})
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/amgx/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/caliper,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+184
View File
@@ -0,0 +1,184 @@
/*
MIT License
Copyright (c) 2017 André L. Maravilha
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
*/
#ifndef CXX_TIMER_HPP
#define CXX_TIMER_HPP
#include <chrono>
namespace cxxtimer {
/**
* This class works as a stopwatch.
*/
class Timer {
public:
/**
* Constructor.
*
* @param start
* If true, the timer is started just after construction.
* Otherwise, it will not be automatically started.
*/
Timer(bool start = false);
/**
* Copy constructor.
*
* @param other
* The object to be copied.
*/
Timer(const Timer& other) = default;
/**
* Transfer constructor.
*
* @param other
* The object to be transferred.
*/
Timer(Timer&& other) = default;
/**
* Destructor.
*/
virtual ~Timer() = default;
/**
* Assignment operator by copy.
*
* @param other
* The object to be copied.
*
* @return A reference to this object.
*/
Timer& operator=(const Timer& other) = default;
/**
* Assignment operator by transfer.
*
* @param other
* The object to be transferred.
*
* @return A reference to this object.
*/
Timer& operator=(Timer&& other) = default;
/**
* Start/resume the timer.
*/
void start();
/**
* Stop/pause the timer.
*/
void stop();
/**
* Reset the timer.
*/
void reset();
/**
* Return the elapsed time.
*
* @param duration_t
* The duration type used to return the time elapsed. If not
* specified, it returns the time as represented by
* std::chrono::milliseconds.
*
* @return The elapsed time.
*/
template <class duration_t = std::chrono::milliseconds>
typename duration_t::rep count() const;
private:
bool started_;
bool paused_;
std::chrono::steady_clock::time_point reference_;
std::chrono::duration<long double> accumulated_;
};
}
inline cxxtimer::Timer::Timer(bool start) :
started_(false), paused_(false),
reference_(std::chrono::steady_clock::now()),
accumulated_(std::chrono::duration<long double>(0)) {
if (start) {
this->start();
}
}
inline void cxxtimer::Timer::start() {
if (!started_) {
started_ = true;
paused_ = false;
accumulated_ = std::chrono::duration<long double>(0);
reference_ = std::chrono::steady_clock::now();
} else if (paused_) {
reference_ = std::chrono::steady_clock::now();
paused_ = false;
}
}
inline void cxxtimer::Timer::stop() {
if (started_ && !paused_) {
std::chrono::steady_clock::time_point now = std::chrono::steady_clock::now();
accumulated_ = accumulated_ + std::chrono::duration_cast< std::chrono::duration<long double> >(now - reference_);
paused_ = true;
}
}
inline void cxxtimer::Timer::reset() {
if (started_) {
started_ = false;
paused_ = false;
reference_ = std::chrono::steady_clock::now();
accumulated_ = std::chrono::duration<long double>(0);
}
}
template <class duration_t>
typename duration_t::rep cxxtimer::Timer::count() const {
if (started_) {
if (paused_) {
return std::chrono::duration_cast<duration_t>(accumulated_).count();
} else {
return std::chrono::duration_cast<duration_t>(
accumulated_ + (std::chrono::steady_clock::now() - reference_)).count();
}
} else {
return duration_t(0).count();
}
}
#endif
+4
View File
@@ -0,0 +1,4 @@
#pragma once
#include "dfem_differentiable_operator.hpp"
#include "dfem_element_operator.hpp"
+232
View File
@@ -0,0 +1,232 @@
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels>::Action::create_action_callback(
kernel_t kernel,
mult_func_t &func)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
const int num_entities = GetNumEntities<entity_t>(op.mesh);
const int num_qp = op.integration_rule.GetNPoints();
// All solutions T-vector sizes make up the width of the operator, since
// they are explicitly provided in Mult() for example.
op.width = GetTrueVSize(op.fields[test_space_field_idx]);
op.residual_lsize = GetVSize(op.fields[test_space_field_idx]);
if constexpr (std::is_same_v<decltype(output_fop), One>)
{
op.height = 1;
}
else
{
op.height = op.residual_lsize;
}
residual_l.SetSize(op.residual_lsize);
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : op.fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
doftoquad_mode));
}
const int q1d = (int)floor(pow(num_qp, 1.0/op.mesh.Dimension()) + 0.5);
residual_e.SetSize(R->Height());
const int residual_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(kernel.outputs),
op.fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
koutput_to_field);
auto input_fops = create_bare_fops(kernel.inputs);
auto output_fops = create_bare_fops(kernel.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(output_fops).vdim /
num_entities;
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
num_qp);
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
output_dtq_maps,
op.fields,
num_entities,
kernel.inputs,
num_qp,
input_size_on_qp,
residual_size_on_qp);
Vector shmem_cache(shmem_info.total_size);
print_shared_memory_info(shmem_info);
func = [=](Vector &ye_mem) mutable
{
restriction<entity_t>(op.solutions, solutions_l, this->fields_e,
op.element_dof_ordering);
restriction<entity_t>(op.parameters, parameters_l, this->fields_e,
op.element_dof_ordering,
op.solutions.size());
auto ye = Reshape(ye_mem.ReadWrite(), test_vdim, num_test_dof, num_entities);
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
// printf("\ne: %d\n", e);
// tic();
auto input_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
shmem_info.input_dtq_sizes,
input_dtq_maps);
auto output_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
shmem_info.output_dtq_sizes,
output_dtq_maps);
auto fields_shmem = load_field_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::FIELD],
shmem_info.field_sizes,
kinput_to_field,
wrapped_fields_e,
e);
// These methods don't copy, they simply create a `DeviceTensor` object
// that points to correct chunks of the shared memory pool.
auto input_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT],
shmem_info.input_sizes,
num_qp);
auto residual_shmem = load_residual_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT],
shmem_info.residual_size,
num_qp);
auto scratch_mem = load_scratch_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::TEMP],
shmem_info.temp_sizes);
MFEM_SYNC_THREAD;
// printf("shmem load elapsed: %.1fus\n", toc() * 1e6);
// tic();
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
std::make_index_sequence<kernel.num_kinputs> {});
// printf("interpolate elapsed: %.1fus\n", toc() * 1e6);
// tic();
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), residual_size_on_qp);
apply_kernel(r, kernel.func, kernel_args, input_shmem, q);
}
}
}
MFEM_SYNC_THREAD;
// printf("qf elapsed: %.1fus\n", toc() * 1e6);
// tic();
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(output_fops),
output_dtq_shmem[hardcoded_output_idx],
scratch_mem);
// printf("integrate elapsed: %.1fus\n", toc() * 1e6);
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
if constexpr (std::is_same_v<decltype(output_fop), None>)
{
residual_l = ye_mem;
}
else
{
R->MultTranspose(ye_mem, residual_l);
}
};
if constexpr (std::is_same_v<decltype(output_fop), None>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
y = r_local;
};
}
else if constexpr (std::is_same_v<decltype(output_fop), One>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
double local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
op.mesh.GetComm());
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
};
}
else
{
auto P = get_prolongation(op.fields[test_space_field_idx]);
prolongation_transpose = [P](const Vector &r_local, Vector &y)
{
P->MultTranspose(r_local, y);
};
}
}
@@ -0,0 +1,308 @@
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels
>
template <
size_t derivative_idx
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels>::Derivative<derivative_idx>::assemble_hypreparmatrix_impl(
kernel_t kernel, HypreParMatrix &A)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs,
std::make_index_sequence<kernel.num_koutputs> {});
auto output_fop = std::get<0>(kernel.outputs);
constexpr int hardcoded_output_idx = 0;
int num_qp = op.integration_rule.GetNPoints();;
int num_el = 0;
int dimension = 0;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
num_el = op.mesh.GetNE();
dimension = op.dim;
}
else if (std::is_same_v<entity_t, Entity::Face>)
{
num_el = op.mesh.GetNumFacesWithGhost();
dimension = op.dim - 1;
}
else
{
static_assert(always_false<entity_t>, "not implemented");
}
std::vector<const DofToQuad*> dtqmaps;
for (const auto &field : op.fields)
{
dtqmaps.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
doftoquad_mode));
}
// Allocate memory for fields on quadrature points
auto input_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto directions_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
for (auto &d_qp_mem : directions_qp_mem)
{
d_qp_mem = 0.0;
}
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
bool no_kinput_is_dependent = true;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_to_field[i] == derivative_idx)
{
no_kinput_is_dependent = false;
kinput_is_dependent[i] = true;
// out << "function input " << i << " is dependent on "
// << op.fields[kinput_to_field[i]].field_label << "\n";
}
else
{
kinput_is_dependent[i] = false;
}
}
if (no_kinput_is_dependent)
{
return;
}
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
DeviceTensor<1, const double> integration_weights(
this->op.integration_rule.GetWeights().Read(), num_qp);
Vector zero;
GeometricFactorMaps geometric_factors
{
DeviceTensor<3, const double>(zero.Read(), 0, 0, 0)
};
// fields interpolated to the quadrature points in the order of
// kernel function arguments
auto input_qp = map_inputs_to_memory(input_qp_mem, num_qp,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto directions_qp = map_inputs_to_memory(directions_qp_mem, num_qp,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto input_dtq_ops = create_dtq_operators<entity_t>(kernel.inputs, dtqmaps,
kinput_to_field);
auto dependent_input_dtq_ops = create_dtq_operators_conditional<entity_t>(
kernel.inputs,
dtqmaps,
kinput_to_field,
kinput_is_dependent, std::make_index_sequence<kernel.num_kinputs> {});
auto output_dtq_ops = create_dtq_operators<entity_t>(kernel.outputs, dtqmaps,
koutput_to_field);
constexpr int fixed_output_idx = 0;
auto Bv = output_dtq_ops[fixed_output_idx];
auto [num_test_qp, test_op_dim, num_test_dof] = Bv.GetShape();
const int test_vdim = std::get<0>(kernel.outputs).vdim;
const int num_trial_dof = dependent_input_dtq_ops[0].GetShape()[2];
int trial_vdim = 0;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_is_dependent[i])
{
trial_vdim = GetVDim(op.fields[kinput_to_field[i]]);
break;
}
}
// All trial operators dimensions accumulated
int total_trial_op_dim = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
total_trial_op_dim += dependent_input_dtq_ops[s].GetShape()[1];
}
Vector a_qp_mem(test_vdim * test_op_dim * trial_vdim * total_trial_op_dim *
num_qp *
num_el);
const auto a_qp = Reshape(a_qp_mem.ReadWrite(), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp,
num_el);
Vector Ae_mem(num_test_dof * test_vdim * num_trial_dof * trial_vdim * num_el);
Ae_mem = 0.0;
auto A_e = Reshape(Ae_mem.ReadWrite(), num_test_dof, test_vdim, num_trial_dof,
trial_vdim, num_el);
for (int e = 0; e < num_el; e++)
{
map_fields_to_quadrature_data(
input_qp, e, this->fields_e,
kinput_to_field, input_dtq_ops,
integration_weights, geometric_factors, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
for (int q = 0; q < num_qp; q++)
{
for (int j = 0; j < trial_vdim; j++)
{
size_t m_offset = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
auto Bu = dependent_input_dtq_ops[s];
auto [unused1, trial_op_dim, unused2] = Bu.GetShape();
auto d_qp = Reshape(&(directions_qp[Bu.which_input])[0], trial_vdim,
trial_op_dim, num_qp);
for (int m = 0; m < trial_op_dim; m++)
{
d_qp(j, m, q) = 1.0;
Vector f_qp = apply_kernel_fwddiff_enzyme(
kernel.func,
kernel_args,
input_qp,
kernel_shadow_args,
directions_qp,
q);
// Vector f_qp = apply_kernel_fwddiff_dual(
// kernel.func,
// kernel_args,
// input_qp,
// directions_qp,
// q);
d_qp(j, m, q) = 0.0;
auto f = Reshape(f_qp.Read(), test_vdim, test_op_dim);
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
a_qp(i, k, j, m + m_offset, q, e) = f(i, k);
}
}
}
m_offset += trial_op_dim;
}
}
}
Vector fhat_mem(test_op_dim * num_qp * dimension);
auto fhat = Reshape(fhat_mem.ReadWrite(), test_vdim, test_op_dim, num_qp);
for (int J = 0; J < num_trial_dof; J++)
{
for (int j = 0; j < trial_vdim; j++)
{
fhat_mem = 0.0;
size_t m_offset = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
auto Bu = dependent_input_dtq_ops[s];
int trial_op_dim = dependent_input_dtq_ops[s].GetShape()[1];
for (int q = 0; q < num_qp; q++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
for (int m = 0; m < trial_op_dim; m++)
{
fhat(i, k, q) += a_qp(i, k, j, m + m_offset, q, e) * Bu(q, m, J);
}
}
}
}
m_offset += trial_op_dim;
}
auto bvtfhat = Reshape(&A_e(0, 0, J, j, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields(bvtfhat, fhat, output_fop,
output_dtq_ops[hardcoded_output_idx]);
}
}
}
bool same_test_and_trial = false;
if (koutput_to_field[0] ==
kinput_to_field[dependent_input_dtq_ops[0].which_input])
{
same_test_and_trial = true;
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&op.fields[kinput_to_field[dependent_input_dtq_ops[0].which_input]].data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&op.fields[koutput_to_field[0]].data);
SparseMatrix mat(test_fes->GlobalVSize(), trial_fes->GlobalVSize());
if (test_fes == nullptr)
{
MFEM_ABORT("error");
}
for (int e = 0; e < num_el; e++)
{
auto tmp = Reshape(Ae_mem.ReadWrite(), num_test_dof * test_vdim,
num_trial_dof * trial_vdim,
num_el);
DenseMatrix A_e(&tmp(0, 0, e), num_test_dof * test_vdim,
num_trial_dof * trial_vdim);
Array<int> test_vdofs, trial_vdofs;
test_fes->GetElementVDofs(e, test_vdofs);
GetElementVDofs(
op.fields[kinput_to_field[dependent_input_dtq_ops[0].which_input]], e,
trial_vdofs);
mat.AddSubMatrix(test_vdofs, trial_vdofs, A_e, 1);
}
mat.Finalize();
if (same_test_and_trial)
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
&mat);
A = *RAP(&tmp, test_fes->Dof_TrueDof_Matrix());
A.EliminateBC(op.ess_tdof_list, DiagonalPolicy::DIAG_ONE);
}
else
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
trial_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
trial_fes->GetDofOffsets(),
&mat);
A = *RAP(test_fes->Dof_TrueDof_Matrix(), &tmp, trial_fes->Dof_TrueDof_Matrix());
// A.EliminateBC(op.ess_tdof_list, DiagonalPolicy::DIAG_ONE);
}
}
+233
View File
@@ -0,0 +1,233 @@
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels
>
template <
size_t derivative_idx
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels>::Derivative<derivative_idx>::assemble_vector_impl(
kernel_t kernel, Vector &v)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs,
std::make_index_sequence<kernel.num_koutputs> {});
auto output_fop = std::get<0>(kernel.outputs);
constexpr int hardcoded_output_idx = 0;
int num_qp = op.integration_rule.GetNPoints();;
int num_el = 0;
int dimension = 0;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
num_el = op.mesh.GetNE();
dimension = op.dim;
}
else if (std::is_same_v<entity_t, Entity::Face>)
{
num_el = op.mesh.GetNumFacesWithGhost();
dimension = op.dim - 1;
}
else
{
static_assert(always_false<entity_t>, "not implemented");
}
std::vector<const DofToQuad*> dtqmaps;
for (const auto &field : op.fields)
{
dtqmaps.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
doftoquad_mode));
}
// Allocate memory for fields on quadrature points
auto input_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto directions_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
for (auto &d_qp_mem : directions_qp_mem)
{
d_qp_mem = 0.0;
}
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
bool no_kinput_is_dependent = true;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_to_field[i] == derivative_idx)
{
no_kinput_is_dependent = false;
kinput_is_dependent[i] = true;
// out << "function input " << i << " is dependent on "
// << op.fields[kinput_to_field[i]].field_label << "\n";
}
else
{
kinput_is_dependent[i] = false;
}
}
if (no_kinput_is_dependent)
{
return;
}
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
DeviceTensor<1, const double> integration_weights(
this->op.integration_rule.GetWeights().Read(), num_qp);
Vector zero;
GeometricFactorMaps geometric_factors
{
DeviceTensor<3, const double>(zero.Read(), 0, 0, 0)
};
// fields interpolated to the quadrature points in the order of
// kernel function arguments
auto input_qp = map_inputs_to_memory(input_qp_mem, num_qp,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto directions_qp = map_inputs_to_memory(directions_qp_mem, num_qp,
kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto input_dtq_ops = create_dtq_operators<entity_t>(kernel.inputs, dtqmaps,
kinput_to_field);
auto dependent_input_dtq_ops = create_dtq_operators_conditional<entity_t>(
kernel.inputs,
dtqmaps,
kinput_to_field,
kinput_is_dependent, std::make_index_sequence<kernel.num_kinputs> {});
auto output_dtq_ops = create_dtq_operators<entity_t>(kernel.outputs, dtqmaps,
koutput_to_field);
constexpr int fixed_output_idx = 0;
auto Bv = output_dtq_ops[fixed_output_idx];
auto [num_test_qp, test_op_dim, num_test_dof] = Bv.GetShape();
const int test_vdim = std::get<0>(kernel.outputs).vdim;
const int num_trial_dof = dependent_input_dtq_ops[0].GetShape()[2];
int trial_vdim = 0;
int dependent_field_idx = -1;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_is_dependent[i])
{
dependent_field_idx = kinput_to_field[i];
break;
}
}
trial_vdim = GetVDim(op.fields[dependent_field_idx]);
// All trial operators dimensions accumulated
int total_trial_op_dim = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
total_trial_op_dim += dependent_input_dtq_ops[s].GetShape()[1];
}
Vector a_qp_mem(trial_vdim * total_trial_op_dim * num_qp * num_el);
const auto a_qp = Reshape(a_qp_mem.ReadWrite(), trial_vdim,
total_trial_op_dim, num_qp, num_el);
Vector ve_mem(num_trial_dof * trial_vdim * num_el);
ve_mem = 0.0;
for (int e = 0; e < num_el; e++)
{
map_fields_to_quadrature_data(
input_qp, e, this->fields_e,
kinput_to_field, input_dtq_ops,
integration_weights, geometric_factors, kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
for (int q = 0; q < num_qp; q++)
{
for (int j = 0; j < trial_vdim; j++)
{
size_t m_offset = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
auto Bu = dependent_input_dtq_ops[s];
auto [unused1, trial_op_dim, unused2] = Bu.GetShape();
auto d_qp = Reshape(&(directions_qp[Bu.which_input])[0], trial_vdim,
trial_op_dim, num_qp);
for (int m = 0; m < trial_op_dim; m++)
{
d_qp(j, m, q) = 1.0;
// Vector f_qp = apply_kernel_fwddiff_dual(
// kernel.func,
// kernel_args,
// input_qp,
// directions_qp,
// q);
Vector f_qp = apply_kernel_fwddiff_enzyme(
kernel.func,
kernel_args,
input_qp,
kernel_shadow_args,
directions_qp,
q);
d_qp(j, m, q) = 0.0;
auto f = Reshape(f_qp.Read(), test_vdim);
a_qp(j, m + m_offset, q, e) = f(0);
}
m_offset += trial_op_dim;
}
}
}
auto shat = Reshape(ve_mem.ReadWrite(), num_trial_dof, trial_vdim, num_el);
for (int J = 0; J < num_trial_dof; J++)
{
for (int j = 0; j < trial_vdim; j++)
{
size_t m_offset = 0;
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
{
auto Bu = dependent_input_dtq_ops[s];
int trial_op_dim = dependent_input_dtq_ops[s].GetShape()[1];
for (int q = 0; q < num_qp; q++)
{
for (int m = 0; m < trial_op_dim; m++)
{
shat(J, j, e) += a_qp(j, m + m_offset, q, e) * Bu(q, m, J);
}
}
m_offset += trial_op_dim;
}
}
}
}
auto R = get_element_restriction(op.fields[dependent_field_idx],
element_dof_ordering);
Vector ve(R->Width());
R->MultTranspose(ve_mem, ve);
get_prolongation(op.fields[dependent_field_idx])->MultTranspose(ve, v);
}
+244
View File
@@ -0,0 +1,244 @@
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels
>
template <
size_t derivative_idx
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels>::Derivative<derivative_idx>::create_callback(kernel_t kernel,
mult_func_t &func)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
const int num_entities = GetNumEntities<entity_t>(op.mesh);
const int num_qp = op.integration_rule.GetNPoints();
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : op.fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
doftoquad_mode));
}
const int q1d = dtq[0]->nqpt;
derivative_action_e.SetSize(R->Height());
const int da_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(kernel.outputs),
op.fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
koutput_to_field);
auto input_fops = create_bare_fops(kernel.inputs);
auto output_fops = create_bare_fops(kernel.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(output_fops).vdim /
num_entities;
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
num_qp);
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
// Check which qf inputs are dependent on the dependent variable
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
bool no_kinput_is_dependent = true;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_to_field[i] == derivative_idx)
{
no_kinput_is_dependent = false;
kinput_is_dependent[i] = true;
// out << "function input " << i << " is dependent on "
// << op.fields[kinput_to_field[i]].field_label << "\n";
}
else
{
kinput_is_dependent[i] = false;
}
}
bool with_derivatives = true;
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
output_dtq_maps,
op.fields,
num_entities,
kernel.inputs,
num_qp,
input_size_on_qp,
da_size_on_qp,
derivative_idx);
Vector shmem_cache(shmem_info.total_size);
print_shared_memory_info(shmem_info);
func = [=](Vector &ye_mem) mutable
{
if (no_kinput_is_dependent)
{
return;
}
restriction<entity_t>(direction, direction_l, direction_e,
op.element_dof_ordering);
auto ye = Reshape(ye_mem.ReadWrite(), num_test_dof, test_vdim, num_entities);
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
auto wrapped_direction_e = Reshape(direction_e.Read(), shmem_info.direction_size, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
{
auto input_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
shmem_info.input_dtq_sizes,
input_dtq_maps);
auto output_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
shmem_info.output_dtq_sizes,
output_dtq_maps);
auto fields_shmem = load_field_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::FIELD],
shmem_info.field_sizes,
kinput_to_field,
wrapped_fields_e,
e);
auto direction_shmem = load_direction_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::DIRECTION],
shmem_info.direction_size,
wrapped_direction_e,
e);
// These methods don't copy, they simply create a `DeviceTensor` object
// that points to correct chunks of the shared memory pool.
auto input_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT],
shmem_info.input_sizes,
num_qp);
auto shadow_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::SHADOW],
shmem_info.input_sizes,
num_qp);
auto residual_shmem = load_residual_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT],
shmem_info.residual_size,
num_qp);
auto scratch_mem = load_scratch_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::TEMP],
shmem_info.temp_sizes);
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
std::make_index_sequence<kernel.num_kinputs> {});
zero_all(shadow_shmem);
map_direction_to_quadrature_data_conditional<TensorProduct>(
shadow_shmem, direction_shmem, input_dtq_shmem, input_fops, ir_weights,
scratch_mem, kinput_is_dependent,
std::make_index_sequence<kernel.num_kinputs> {});
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), da_size_on_qp);
apply_kernel_fwddiff_enzyme(
r,
kernel.func,
kernel_args,
input_shmem,
kernel_shadow_args,
shadow_shmem,
q);
// printf(">>>>> WARNING: AD DISABLED\n");
}
}
}
MFEM_SYNC_THREAD;
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(output_fops),
output_dtq_shmem[hardcoded_output_idx],
scratch_mem);
}, num_entities, q1d, q1d, 1, shmem_info.total_size, shmem_cache.ReadWrite());
R->MultTranspose(ye_mem, derivative_action_l);
};
if constexpr (std::is_same_v<decltype(output_fop), One>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
double local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
op.mesh.GetComm());
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
};
}
else
{
auto P = get_prolongation(op.fields[test_space_field_idx]);
prolongation_transpose = [P](const Vector &r_local, Vector &y)
{
P->MultTranspose(r_local, y);
};
}
}
@@ -0,0 +1,820 @@
#pragma once
#include <algorithm>
#include <cstdlib>
#include <functional>
#include <iostream>
#include <utility>
#include <variant>
#include <vector>
#include <type_traits>
#include <mfem.hpp>
#include <type_traits>
#include "dfem_fieldoperator.hpp"
#include "dfem_parametricspace.hpp"
#include "general/tic_toc.hpp"
#include "tuple.hpp"
#include <linalg/tensor.hpp>
#include <enzyme/utils>
#include <enzyme/enzyme>
#include "dfem_util.hpp"
#include "dfem_interpolate.hpp"
#include "dfem_qfunction.hpp"
#include "dfem_qfunction_dual.hpp"
#include "dfem_integrate.hpp"
namespace mfem
{
using mult_func_t = std::function<void(Vector &)>;
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields = num_solutions + num_parameters,
size_t num_kernels = mfem::tuple_size<kernels_tuple>::value,
typename autodiff_t = AutoDiff::NativeDualNumber
>
class DifferentiableOperator : public Operator
{
public:
DifferentiableOperator(DifferentiableOperator&) = delete;
DifferentiableOperator(DifferentiableOperator&&) = delete;
class Action : public Operator
{
public:
template <typename kernel_t>
void create_action_callback(kernel_t kernel, mult_func_t &func);
template<std::size_t... idx>
void materialize_callbacks(kernels_tuple &ks,
std::array<mult_func_t, num_kernels>,
std::index_sequence<idx...> const&)
{
(create_action_callback(mfem::get<idx>(ks), funcs[idx]), ...);
}
Action(DifferentiableOperator &op, kernels_tuple &ks) : op(op)
{
materialize_callbacks(ks, funcs,
std::make_index_sequence<mfem::tuple_size<kernels_tuple>::value>());
}
void Mult(const Vector &x, Vector &y) const
{
prolongation(op.solutions, x, solutions_l);
residual_e = 0.0;
for (const auto &f : funcs)
{
f(residual_e);
}
prolongation_transpose(residual_l, y);
y.SetSubVector(op.ess_tdof_list, 0.0);
}
void SetParameters(std::vector<Vector *> p) const
{
MFEM_ASSERT(num_parameters == p.size(),
"number of parameters doesn't match descriptors");
for (int i = 0; i < num_parameters; i++)
{
p[i]->Read();
parameters_l[i] = *p[i];
// parameters_l[i].MakeRef(p[i], 0, p[i]->Size());
}
}
protected:
DifferentiableOperator &op;
std::array<mult_func_t, num_kernels> funcs;
std::function<void(Vector &, Vector &)> prolongation_transpose;
mutable std::array<Vector, num_solutions> solutions_l;
mutable std::array<Vector, num_parameters> parameters_l;
mutable Vector residual_l;
mutable std::array<Vector, num_fields> fields_e;
mutable Vector residual_e;
};
template <size_t derivative_idx>
class Derivative : public Operator
{
public:
template <typename kernel_t>
void create_callback(kernel_t kernel, mult_func_t &func);
template<std::size_t... idx>
void materialize_callbacks(kernels_tuple &ks,
std::array<mult_func_t, num_kernels>,
std::index_sequence<idx...> const&)
{
(create_callback(mfem::get<idx>(ks), funcs[idx]), ...);
}
Derivative(
DifferentiableOperator &op,
std::array<Vector *, num_solutions> &solutions,
std::array<Vector *, num_parameters> &parameters,
kernels_tuple &ks) : op(op), ks(ks)
{
for (int i = 0; i < num_solutions; i++)
{
solutions_l[i] = *solutions[i];
}
for (int i = 0; i < num_parameters; i++)
{
parameters_l[i] = *parameters[i];
}
// G
// if constexpr (std::is_same_v<OperatesOn, OperatesOnElement>)
// {
element_restriction(op.solutions, solutions_l, fields_e,
op.element_dof_ordering);
element_restriction(op.parameters, parameters_l, fields_e,
op.element_dof_ordering,
op.solutions.size());
// }
// else
// {
// MFEM_ABORT("restriction not implemented for OperatesOn");
// }
direction = op.fields[derivative_idx];
size_t derivative_action_l_size = 0;
for (auto &s : op.solutions)
{
derivative_action_l_size += GetVSize(s);
this->width += GetTrueVSize(s);
}
this->height = derivative_action_l_size;
derivative_action_l.SetSize(derivative_action_l_size);
materialize_callbacks(ks, funcs,
std::make_index_sequence<num_kernels>());
}
void Mult(const Vector &x, Vector &y) const override
{
current_direction_t = x;
current_direction_t.SetSubVector(op.ess_tdof_list, 0.0);
prolongation(direction, current_direction_t, direction_l);
derivative_action_e = 0.0;
for (const auto &f : funcs)
{
f(derivative_action_e);
}
prolongation_transpose(derivative_action_l, y);
y.SetSubVector(op.ess_tdof_list, 0.0);
}
template <typename kernel_t>
void assemble_vector_impl(kernel_t kernel, Vector &v);
template<std::size_t... idx>
void assemble_vector(
kernels_tuple &ks,
Vector &v,
std::index_sequence<idx...> const&)
{
(assemble_vector_impl(mfem::get<idx>(ks), v), ...);
}
void Assemble(Vector &v)
{
assemble_vector(ks, v, std::make_index_sequence<num_kernels>());
}
template <typename kernel_t>
void assemble_hypreparmatrix_impl(kernel_t kernel, HypreParMatrix &A);
template<std::size_t... idx>
void assemble_hypreparmatrix(
kernels_tuple &ks,
HypreParMatrix &A,
std::index_sequence<idx...> const&)
{
(assemble_hypreparmatrix_impl(mfem::get<idx>(ks), A), ...);
}
void Assemble(HypreParMatrix &A)
{
assemble_hypreparmatrix(ks, A, std::make_index_sequence<num_kernels>());
}
void AssembleDiagonal(Vector &d) const override {}
protected:
DifferentiableOperator &op;
kernels_tuple &ks;
std::array<mult_func_t, num_kernels> funcs;
std::function<void(Vector &, Vector &)> prolongation_transpose;
FieldDescriptor direction;
std::array<Vector, num_solutions> solutions_l;
std::array<Vector, num_parameters> parameters_l;
mutable Vector direction_l;
mutable Vector derivative_action_l;
mutable std::array<Vector, num_fields> fields_e;
mutable Vector direction_e;
mutable Vector derivative_action_e;
mutable Vector current_direction_t;
};
DifferentiableOperator(std::array<FieldDescriptor, num_solutions> s,
std::array<FieldDescriptor, num_parameters> p,
kernels_tuple ks,
ParMesh &m,
autodiff_t ad = AutoDiff::NativeDualNumber{}) :
kernels(ks),
mesh(m),
dim(mesh.Dimension()),
solutions(s),
parameters(p)
{
for (int i = 0; i < num_solutions; i++)
{
fields[i] = solutions[i];
}
for (int i = 0; i < num_parameters; i++)
{
fields[i + num_solutions] = parameters[i];
}
residual.reset(new Action(*this, kernels));
}
void SetParameters(std::vector<Vector *> p) const
{
residual->SetParameters(p);
}
void Mult(const Vector &x, Vector &y) const override
{
residual->Mult(x, y);
}
template <int derivative_idx>
std::shared_ptr<Derivative<derivative_idx>>
GetDerivativeWrt(std::array<Vector *, num_solutions> solutions,
std::array<Vector *, num_parameters> parameters)
{
return std::shared_ptr<Derivative<derivative_idx>>(
new Derivative<derivative_idx>(*this, solutions, parameters, kernels));
}
void SetEssentialTrueDofs(const Array<int> &l)
{
l.Copy(ess_tdof_list);
}
kernels_tuple kernels;
ParMesh &mesh;
const int dim;
std::array<FieldDescriptor, num_solutions> solutions;
std::array<FieldDescriptor, num_parameters> parameters;
// solutions and parameters
std::array<FieldDescriptor, num_fields> fields;
int residual_lsize = 0;
mutable std::array<Vector, num_solutions> current_state_l;
mutable Vector direction_l;
mutable Vector current_direction_t;
Array<int> ess_tdof_list;
static constexpr ElementDofOrdering element_dof_ordering =
ElementDofOrdering::LEXICOGRAPHIC;
static constexpr DofToQuad::Mode doftoquad_mode =
DofToQuad::Mode::TENSOR;
// static constexpr ElementDofOrdering element_dof_ordering =
// ElementDofOrdering::NATIVE;
// static constexpr DofToQuad::Mode doftoquad_mode =
// DofToQuad::Mode::FULL;
std::shared_ptr<Action> residual;
};
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels,
typename autodiff_t
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels,
autodiff_t>::Action::create_action_callback(
kernel_t kernel,
mult_func_t &func)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
const int num_entities = GetNumEntities<entity_t>(op.mesh);
const int num_qp = kernel.integration_rule.GetNPoints();
// All solutions T-vector sizes make up the width of the operator, since
// they are explicitly provided in Mult() for example.
op.width = GetTrueVSize(op.fields[test_space_field_idx]);
op.residual_lsize = GetVSize(op.fields[test_space_field_idx]);
if constexpr (std::is_same_v<decltype(output_fop), One>)
{
op.height = 1;
}
else
{
op.height = op.residual_lsize;
}
residual_l.SetSize(op.residual_lsize);
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : op.fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(field, kernel.integration_rule,
doftoquad_mode));
}
const int q1d = (int)floor(pow(num_qp, 1.0/op.mesh.Dimension()) + 0.5);
residual_e.SetSize(R->Height());
const int residual_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(kernel.outputs),
op.fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
koutput_to_field);
auto input_fops = create_bare_fops(kernel.inputs);
auto output_fops = create_bare_fops(kernel.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(output_fops).vdim /
num_entities;
auto ir_weights = Reshape(kernel.integration_rule.GetWeights().Read(), num_qp);
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
output_dtq_maps,
op.fields,
num_entities,
kernel.inputs,
num_qp,
input_size_on_qp,
residual_size_on_qp);
Vector shmem_cache(shmem_info.total_size);
// print_shared_memory_info(shmem_info);
func = [=](Vector &ye_mem) mutable
{
restriction<entity_t>(op.solutions, solutions_l, this->fields_e,
op.element_dof_ordering);
restriction<entity_t>(op.parameters, parameters_l, this->fields_e,
op.element_dof_ordering,
op.solutions.size());
auto ye = Reshape(ye_mem.ReadWrite(), test_vdim, num_test_dof, num_entities);
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
// printf("\ne: %d\n", e);
// tic();
auto input_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
shmem_info.input_dtq_sizes,
input_dtq_maps);
auto output_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
shmem_info.output_dtq_sizes,
output_dtq_maps);
auto fields_shmem = load_field_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::FIELD],
shmem_info.field_sizes,
kinput_to_field,
input_fops,
wrapped_fields_e,
e,
std::make_index_sequence<kernel.num_kinputs> {});
// These functions don't copy, they simply create a `DeviceTensor` object
// that points to correct chunks of the shared memory pool.
auto input_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT],
shmem_info.input_sizes,
num_qp);
auto residual_shmem = load_residual_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT],
shmem_info.residual_size,
num_qp);
auto scratch_mem = load_scratch_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::TEMP],
shmem_info.temp_sizes);
MFEM_SYNC_THREAD;
// printf("shmem load elapsed: %.1fus\n", toc() * 1e6);
// tic();
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
std::make_index_sequence<kernel.num_kinputs> {});
// printf("interpolate elapsed: %.1fus\n", toc() * 1e6);
// tic();
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), residual_size_on_qp);
apply_kernel(r, kernel.func, kernel_args, input_shmem, q);
}
}
}
MFEM_SYNC_THREAD;
// printf("qf elapsed: %.1fus\n", toc() * 1e6);
// tic();
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(output_fops),
output_dtq_shmem[hardcoded_output_idx],
scratch_mem);
// printf("integrate elapsed: %.1fus\n", toc() * 1e6);
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
if constexpr (std::is_same_v<decltype(output_fop), None>)
{
residual_l = ye_mem;
}
else
{
R->MultTranspose(ye_mem, residual_l);
}
};
if constexpr (std::is_same_v<decltype(output_fop), None>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
y = r_local;
};
}
else if constexpr (std::is_same_v<decltype(output_fop), One>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
double local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
op.mesh.GetComm());
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
};
}
else
{
auto P = get_prolongation(op.fields[test_space_field_idx]);
prolongation_transpose = [P](const Vector &r_local, Vector &y)
{
P->MultTranspose(r_local, y);
};
}
}
template <
typename kernels_tuple,
size_t num_solutions,
size_t num_parameters,
size_t num_fields,
size_t num_kernels,
typename autodiff_t
>
template <
size_t derivative_idx
>
template <
typename kernel_t
>
void DifferentiableOperator<kernels_tuple,
num_solutions,
num_parameters,
num_fields,
num_kernels,
autodiff_t>::Derivative<derivative_idx>::create_callback(kernel_t kernel,
mult_func_t &func)
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
const int num_entities = GetNumEntities<entity_t>(op.mesh);
const int num_qp = kernel.integration_rule.GetNPoints();
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : op.fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(field, kernel.integration_rule,
doftoquad_mode));
}
const int q1d = dtq[0]->nqpt;
derivative_action_e.SetSize(R->Height());
const int da_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(kernel.outputs),
op.fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
koutput_to_field);
auto input_fops = create_bare_fops(kernel.inputs);
auto output_fops = create_bare_fops(kernel.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(output_fops).vdim /
num_entities;
auto ir_weights = Reshape(kernel.integration_rule.GetWeights().Read(), num_qp);
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
// Check which qf inputs are dependent on the dependent variable
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
bool no_kinput_is_dependent = true;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_to_field[i] == derivative_idx)
{
no_kinput_is_dependent = false;
kinput_is_dependent[i] = true;
// out << "function input " << i << " is dependent on "
// << op.fields[kinput_to_field[i]].field_label << "\n";
}
else
{
kinput_is_dependent[i] = false;
}
}
bool with_derivatives = true;
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
output_dtq_maps,
op.fields,
num_entities,
kernel.inputs,
num_qp,
input_size_on_qp,
da_size_on_qp,
derivative_idx);
Vector shmem_cache(shmem_info.total_size);
// print_shared_memory_info(shmem_info);
func = [=](Vector &ye_mem) mutable
{
if (no_kinput_is_dependent)
{
return;
}
restriction<entity_t>(direction, direction_l, direction_e,
op.element_dof_ordering);
auto ye = Reshape(ye_mem.ReadWrite(), num_test_dof, test_vdim, num_entities);
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(), shmem_info.direction_size, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
{
auto input_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
shmem_info.input_dtq_sizes,
input_dtq_maps);
auto output_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
shmem_info.output_dtq_sizes,
output_dtq_maps);
auto fields_shmem = load_field_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::FIELD],
shmem_info.field_sizes,
kinput_to_field,
input_fops,
wrapped_fields_e,
e,
std::make_index_sequence<kernel.num_kinputs> {});
auto direction_shmem = load_direction_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::DIRECTION],
shmem_info.direction_size,
wrapped_direction_e,
e);
// These methods don't copy, they simply create a `DeviceTensor` object
// that points to correct chunks of the shared memory pool.
auto input_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT],
shmem_info.input_sizes,
num_qp);
auto shadow_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::SHADOW],
shmem_info.input_sizes,
num_qp);
auto residual_shmem = load_residual_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT],
shmem_info.residual_size,
num_qp);
auto scratch_mem = load_scratch_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::TEMP],
shmem_info.temp_sizes);
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
std::make_index_sequence<kernel.num_kinputs> {});
zero_all(shadow_shmem);
map_direction_to_quadrature_data_conditional<TensorProduct>(
shadow_shmem, direction_shmem, input_dtq_shmem, input_fops, ir_weights,
scratch_mem, kinput_is_dependent,
std::make_index_sequence<kernel.num_kinputs> {});
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto r = Reshape(&residual_shmem(0, q), da_size_on_qp);
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
if constexpr (std::is_same_v<autodiff_t, AutoDiff::EnzymeForward>)
{
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
apply_kernel_fwddiff_enzyme(
r,
kernel.func,
kernel_args,
kernel_shadow_args,
input_shmem,
shadow_shmem,
q);
}
else if constexpr (std::is_same_v<autodiff_t, AutoDiff::NativeDualNumber>)
{
apply_kernel_native_dual(
r,
kernel.func,
kernel_args,
input_shmem,
shadow_shmem,
q);
}
else
{
static_assert(always_false<autodiff_t>, "unknown autodiff type");
}
}
}
}
MFEM_SYNC_THREAD;
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(output_fops),
output_dtq_shmem[hardcoded_output_idx],
scratch_mem);
}, num_entities, q1d, q1d, 1, shmem_info.total_size, shmem_cache.ReadWrite());
R->MultTranspose(ye_mem, derivative_action_l);
};
if constexpr (std::is_same_v<decltype(output_fop), One>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
double local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
op.mesh.GetComm());
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
};
}
else
{
auto P = get_prolongation(op.fields[test_space_field_idx]);
prolongation_transpose = [P](const Vector &r_local, Vector &y)
{
P->MultTranspose(r_local, y);
};
}
}
} // namespace mfem
+79
View File
@@ -0,0 +1,79 @@
#include "dfem_util.hpp"
namespace mfem
{
template <typename func_t, typename input_t, typename output_t, typename dependency_map_t>
struct ElementOperator;
template <typename func_t, typename... input_ts, typename... output_ts, typename dependency_map_t>
struct ElementOperator<func_t, mfem::tuple<input_ts...>, mfem::tuple<output_ts...>, dependency_map_t>
{
using entity_t = Entity::Element;
func_t qfunc;
mfem::tuple<input_ts...> inputs;
mfem::tuple<output_ts...> outputs;
dependency_map_t dependency_map;
using qf_param_ts = typename create_function_signature<
decltype(&func_t::operator())>::type::parameter_ts;
using qf_output_t = typename create_function_signature<
decltype(&func_t::operator())>::type::return_t;
static constexpr size_t num_inputs =
mfem::tuple_size<decltype(inputs)>::value;
static constexpr size_t num_outputs =
mfem::tuple_size<decltype(outputs)>::value;
ElementOperator(func_t qfunc,
mfem::tuple<input_ts...> inputs,
mfem::tuple<output_ts...> outputs)
: qfunc(qfunc), inputs(inputs), outputs(outputs),
dependency_map(make_dependency_map(inputs))
{
// Consistency checks
if constexpr (num_outputs > 1)
{
static_assert(always_false<func_t>,
"more than one output per kernel is not supported right now");
}
constexpr size_t num_qfinputs = mfem::tuple_size<qf_param_ts>::value;
static_assert(num_qfinputs == num_inputs,
"kernel function inputs and descriptor inputs have to match");
constexpr size_t num_qf_outputs = mfem::tuple_size<qf_output_t>::value;
static_assert(num_qf_outputs == num_qf_outputs,
"kernel function outputs and descriptor outputs have to match");
}
};
template <typename func_t, typename... input_ts, typename... output_ts>
ElementOperator(func_t, mfem::tuple<input_ts...>, mfem::tuple<output_ts...>)
-> ElementOperator<func_t, mfem::tuple<input_ts...>, mfem::tuple<output_ts...>,
decltype(make_dependency_map(std::declval<mfem::tuple<input_ts...>>()))>;
// template <typename func_t, typename input_t, typename output_t>
// struct BoundaryElementOperator : public
// ElementOperator<func_t, input_t, output_t>
// {
// public:
// using entity_t = Entity::BoundaryElement;
// BoundaryElementOperator(func_t func, input_t inputs, output_t outputs)
// : ElementOperator<func_t, input_t, output_t>(func, inputs, outputs) {}
// };
// template <typename func_t, typename input_t, typename output_t>
// struct FaceOperator : public
// ElementOperator<func_t, input_t, output_t>
// {
// public:
// using entity_t = Entity::Face;
// FaceOperator(func_t func, input_t inputs, output_t outputs)
// : ElementOperator<func_t, input_t, output_t>(func, inputs, outputs) {}
// };
} // namespace mfem
+227
View File
@@ -0,0 +1,227 @@
#pragma once
#include <string>
namespace mfem
{
template <int FIELD_ID = -1>
class FieldOperator
{
public:
constexpr FieldOperator(int size_on_qp = 0) :
size_on_qp(size_on_qp) {};
static constexpr int GetFieldId() { return FIELD_ID; }
int size_on_qp = -1;
int dim = -1;
int vdim = -1;
};
template <int FIELD_ID = -1>
class None : public FieldOperator<FIELD_ID>
{
public:
constexpr None() : FieldOperator<FIELD_ID>() {}
};
template< typename T >
struct is_none_fop
{
static const bool value = false;
};
template <int FIELD_ID>
struct is_none_fop<None<FIELD_ID>>
{
static const bool value = true;
};
template <typename T>
struct DisableAD
{
T& operator()() const { return fop; }
T fop;
};
class Weight : public FieldOperator<-1>
{
public:
constexpr Weight() : FieldOperator<-1>() {};
};
template< typename T >
struct is_weight_fop
{
static const bool value = false;
};
template <>
struct is_weight_fop<Weight>
{
static const bool value = true;
};
template <int FIELD_ID = -1>
class Value : public FieldOperator<FIELD_ID>
{
public:
constexpr Value() : FieldOperator<FIELD_ID>() {};
};
template< typename T >
struct is_value_fop
{
static const bool value = false;
};
template <int FIELD_ID>
struct is_value_fop<Value<FIELD_ID>>
{
static const bool value = true;
};
template <typename T>
struct is_value_fop<DisableAD<T>>
{
static const bool value = is_value_fop<T>::value;
};
template <int FIELD_ID = -1>
class Gradient : public FieldOperator<FIELD_ID>
{
public:
constexpr Gradient() : FieldOperator<FIELD_ID>() {};
};
template< typename T >
struct is_gradient_fop
{
static const bool value = false;
};
template <int FIELD_ID>
struct is_gradient_fop<Gradient<FIELD_ID>>
{
static const bool value = true;
};
// class FieldOperator
// {
// public:
// FieldOperator(std::string field_label = "", int size_on_qp = 0) :
// field_label(field_label),
// size_on_qp(size_on_qp) {};
// std::string field_label;
// int size_on_qp = -1;
// int dim = -1;
// int vdim = -1;
// };
// class None : public FieldOperator
// {
// public:
// None(std::string field_label) :
// FieldOperator(field_label) {}
// };
// class Weight : public FieldOperator
// {
// public:
// Weight() : FieldOperator("quadrature_weights") {};
// };
// class Value : public FieldOperator
// {
// public:
// Value(std::string field_label) : FieldOperator(field_label) {};
// };
// class Gradient : public FieldOperator
// {
// public:
// Gradient(std::string field_label) : FieldOperator(field_label) {};
// };
// class Curl : public FieldOperator
// {
// public:
// Curl(std::string field_label) : FieldOperator(field_label) {};
// };
// class Div : public FieldOperator
// {
// public:
// Div(std::string field_label) : FieldOperator(field_label) {};
// };
// class FaceValueLeft : public FieldOperator
// {
// public:
// FaceValueLeft(std::string field_label) : FieldOperator(field_label) {};
// };
// class FaceValueRight : public FieldOperator
// {
// public:
// FaceValueRight(std::string field_label) : FieldOperator(field_label) {};
// };
// class FaceNormal : public FieldOperator
// {
// public:
// FaceNormal(std::string field_label) : FieldOperator(field_label) {};
// };
// class One : public FieldOperator
// {
// public:
// One(std::string field_label) : FieldOperator(field_label) {};
// };
// namespace BareFieldOperator
// {
// struct Base
// {
// Base(FieldOperator &o)
// {
// size_on_qp = o.size_on_qp;
// dim = o.dim;
// vdim = o.vdim;
// };
// int size_on_qp = -1;
// int dim = -1;
// int vdim = -1;
// };
// struct None : Base
// {
// None(FieldOperator &o) : Base(o) {}
// };
// struct Weight : Base
// {
// Weight(FieldOperator &o) : Base(o) {}
// };
// struct Value : Base
// {
// Value(FieldOperator &o) : Base(o) {}
// };
// struct Gradient : Base
// {
// Gradient(FieldOperator &o) : Base(o) {}
// };
// }
} // namespace mfem
+290
View File
@@ -0,0 +1,290 @@
#pragma once
#include "dfem_util.hpp"
#include <type_traits>
namespace mfem
{
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_impl(DeviceTensor<2, double> &y,
const DeviceTensor<3, double> &f,
const output_t &output,
const DofToQuadMap &dtq)
{
auto B = dtq.B;
auto G = dtq.G;
// assuming the quadrature point residual has to "play nice with
// the test function"
if constexpr (std::is_same_v<std::decay_t<output_t>, Value<>>)
{
const auto [num_qp, cdim, num_dof] = B.GetShape();
const int vdim = output.vdim > 0 ? output.vdim : cdim ;
for (int dof = 0; dof < num_dof; dof++)
{
for (int vd = 0; vd < vdim; vd++)
{
double acc = 0.0;
for (int qp = 0; qp < num_qp; qp++)
{
acc += B(qp, 0, dof) * f(vd, 0, qp);
}
y(dof, vd) += acc;
}
}
}
else if constexpr (
std::is_same_v<std::decay_t<output_t>, Gradient<>>)
{
const auto [num_qp, dim, num_dof] = G.GetShape();
const int vdim = output.vdim;
for (int dof = 0; dof < num_dof; dof++)
{
for (int vd = 0; vd < vdim; vd++)
{
double acc = 0.0;
for (int d = 0; d < dim; d++)
{
for (int qp = 0; qp < num_qp; qp++)
{
acc += G(qp, d, dof) * f(vd, d, qp);
}
}
y(dof, vd) += acc;
}
}
}
// else if constexpr (std::is_same_v<std::decay_t<output_t>, One>)
// {
// // This is the "integral over all quadrature points type" applying
// // B = 1 s.t. B^T * C \in R^1.
// const auto [a, b, num_qp] = B.GetShape();
// auto cc = Reshape(&c(0, 0, 0), num_qp);
// for (int i = 0; i < num_qp; i++)
// {
// y(0, 0) += cc(i);
// }
// }
else if constexpr (
std::is_same_v<std::decay_t<output_t>, None<>>)
{
const auto [vdim, dim, num_qp] = G.GetShape();
auto cc = Reshape(&f(0, 0, 0), num_qp * vdim);
auto yy = Reshape(&y(0, 0), num_qp * vdim);
for (int i = 0; i < num_qp * vdim; i++)
{
yy(i) = cc(i);
}
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor");
}
}
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_tensor_impl(DeviceTensor<2, double> &y,
const DeviceTensor<3, double> &f,
const output_t &output,
const DofToQuadMap &dtq,
std::array<DeviceTensor<1>, 6> &scratch_mem)
{
auto B = dtq.B;
auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
auto s1 = Reshape(&scratch_mem[1](0), q1d, d1d, d1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
double acc = 0.0;
for (int qx = 0; qx < q1d; qx++)
{
acc += fqp(vd, 0, qx, qy, qz) * B(qx, 0, dx);
}
s0(qz, qy, dx) = acc;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
double acc = 0.0;
for (int qy = 0; qy < q1d; qy++)
{
acc += s0(qz, qy, dx) * B(qy, 0, dy);
}
s1(qz, dy, dx) = acc;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
MFEM_FOREACH_THREAD(dz, z, d1d)
{
double acc = 0.0;
for (int qz = 0; qz < q1d; qz++)
{
acc += s1(qz, dy, dx) * B(qz, 0, dz);
}
yd(dx, dy, dz, vd) += acc;
}
}
}
MFEM_SYNC_THREAD;
}
}
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = G.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
auto s1 = Reshape(&scratch_mem[1](0), q1d, q1d, d1d);
auto s2 = Reshape(&scratch_mem[2](0), q1d, q1d, d1d);
auto s3 = Reshape(&scratch_mem[3](0), q1d, d1d, d1d);
auto s4 = Reshape(&scratch_mem[4](0), q1d, d1d, d1d);
auto s5 = Reshape(&scratch_mem[5](0), q1d, d1d, d1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t uvw[3] = {0.0, 0.0, 0.0};
for (int qx = 0; qx < q1d; qx++)
{
uvw[0] += fqp(vd, 0, qx, qy, qz) * G(qx, 0, dx);
uvw[1] += fqp(vd, 1, qx, qy, qz) * B(qx, 0, dx);
uvw[2] += fqp(vd, 2, qx, qy, qz) * B(qx, 0, dx);
}
s0(qz, qy, dx) = uvw[0];
s1(qz, qy, dx) = uvw[1];
s2(qz, qy, dx) = uvw[2];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t uvw[3] = {0.0, 0.0, 0.0};
for (int qy = 0; qy < q1d; qy++)
{
uvw[0] += s0(qz, qy, dx) * B(qy, 0, dy);
uvw[1] += s1(qz, qy, dx) * G(qy, 0, dy);
uvw[2] += s2(qz, qy, dx) * B(qy, 0, dy);
}
s3(qz, dy, dx) = uvw[0];
s4(qz, dy, dx) = uvw[1];
s5(qz, dy, dx) = uvw[2];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, d1d)
{
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t uvw[3] = {0.0, 0.0, 0.0};
for (int qz = 0; qz < q1d; qz++)
{
uvw[0] += s3(qz, dy, dx) * B(qz, 0, dz);
uvw[1] += s4(qz, dy, dx) * B(qz, 0, dz);
uvw[2] += s5(qz, dy, dx) * G(qz, 0, dz);
}
yd(dx, dy, dz, vd) += uvw[0] + uvw[1] + uvw[2];
}
}
}
MFEM_SYNC_THREAD;
}
}
else if constexpr (is_none_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d, q1d, q1d);
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d, q1d, q1d);
for (int sq = 0; sq < output.size_on_qp; sq++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
yqp(sq, qx, qy, qz) = fqp(sq, qx, qy, qz);
}
}
}
MFEM_SYNC_THREAD;
}
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
}
}
template <typename T = NonTensorProduct, typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields(DeviceTensor<2, double> &y,
const DeviceTensor<3, double> &f,
const output_t &output,
const DofToQuadMap &dtq,
std::array<DeviceTensor<1>, 6> &scratch_mem)
{
if constexpr (std::is_same_v<T, NonTensorProduct>)
{
map_quadrature_data_to_fields_impl(y, f, output, dtq);
}
else if constexpr (std::is_same_v<T, TensorProduct>)
{
map_quadrature_data_to_fields_tensor_impl(y, f, output, dtq, scratch_mem);
}
}
}
+403
View File
@@ -0,0 +1,403 @@
#pragma once
#include "dfem_util.hpp"
namespace mfem
{
template <typename field_operator_t>
MFEM_HOST_DEVICE inline
void map_field_to_quadrature_data_tensor_product(
DeviceTensor<2> &field_qp,
const DofToQuadMap &dtq,
const DeviceTensor<1> &field_e,
const field_operator_t &input,
const DeviceTensor<1, const double> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem)
{
auto B = dtq.B;
auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
{
auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, q1d, q1d, q1d);
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
auto s1 = Reshape(&scratch_mem[1](0), d1d, q1d, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dz, z, d1d)
{
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
double acc = 0.0;
for (int dx = 0; dx < d1d; dx++)
{
acc += B(qx, 0, dx) * field(dx, dy, dz, vd);
}
s0(dz, dy, qx) = acc;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, d1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
double acc = 0.0;
for (int dy = 0; dy < d1d; dy++)
{
acc += s0(dz, dy, qx) * B(qy, 0, dy);
}
s1(dz, qy, qx) = acc;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
double acc = 0.0;
for (int dz = 0; dz < d1d; dz++)
{
acc += s1(dz, qy, qx) * B(qz, 0, dz);
}
fqp(vd, qx, qy, qz) = acc;
}
}
}
MFEM_SYNC_THREAD;
}
}
else if constexpr (
is_gradient_fop<std::decay_t<field_operator_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const int dim = input.dim;
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
auto s1 = Reshape(&scratch_mem[1](0), d1d, d1d, q1d);
auto s2 = Reshape(&scratch_mem[2](0), d1d, q1d, q1d);
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dz, z, d1d)
{
MFEM_FOREACH_THREAD(dy, y, d1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t uv[2] = {0.0, 0.0};
for (int dx = 0; dx < d1d; dx++)
{
const real_t f = field(dx, dy, dz, vd);
uv[0] += f * B(qx, 0, dx);
uv[1] += f * G(qx, 0, dx);
}
s0(dz, dy, qx) = uv[0];
s1(dz, dy, qx) = uv[1];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, d1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t uvw[3] = {0.0, 0.0, 0.0};
for (int dy = 0; dy < d1d; dy++)
{
const real_t s0i = s0(dz, dy, qx);
uvw[0] += s1(dz, dy, qx) * B(qy, 0, dy);
uvw[1] += s0i * G(qy, 0, dy);
uvw[2] += s0i * B(qy, 0, dy);
}
s2(dz, qy, qx) = uvw[0];
s3(dz, qy, qx) = uvw[1];
s4(dz, qy, qx) = uvw[2];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t uvw[3] = {0.0, 0.0, 0.0};
for (int dz = 0; dz < d1d; dz++)
{
uvw[0] += s2(dz, qy, qx) * B(qz, 0, dz);
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
}
fqp(vd, 0, qx, qy, qz) = uvw[0];
fqp(vd, 1, qx, qy, qz) = uvw[1];
fqp(vd, 2, qx, qy, qz) = uvw[2];
}
}
}
MFEM_SYNC_THREAD;
}
}
// TODO: Create separate function for clarity
else if constexpr (
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
{
const int num_qp = integration_weights.GetShape()[0];
// TODO: eeek
const int q1d = (int)floor(pow(num_qp, 1.0/input.dim) + 0.5);
auto w = Reshape(&integration_weights[0], q1d, q1d, q1d);
auto f = Reshape(&field_qp[0], q1d, q1d, q1d);
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
f(qx, qy, qz) = w(qx, qy, qz);
}
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_none_fop<std::decay_t<field_operator_t>>::value)
{
const int q1d = B.GetShape()[0];
auto field = Reshape(&field_e[0], input.size_on_qp, q1d * q1d * q1d);
field_qp = field;
}
else
{
static_assert(always_false<std::decay_t<field_operator_t>>,
"can't map field to quadrature data");
}
}
template <typename field_operator_t>
MFEM_HOST_DEVICE
void map_field_to_quadrature_data(
DeviceTensor<2> field_qp,
const DofToQuadMap &dtq,
const DeviceTensor<1, const double> &field_e,
field_operator_t &input,
DeviceTensor<1, const double> integration_weights)
{
auto B = dtq.B;
auto G = dtq.G;
if constexpr (is_value_fop<field_operator_t>::value)
{
auto [num_qp, dim, num_dof] = B.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e(0), num_dof, vdim);
for (int vd = 0; vd < vdim; vd++)
{
for (int qp = 0; qp < num_qp; qp++)
{
double acc = 0.0;
for (int dof = 0; dof < num_dof; dof++)
{
acc += B(qp, 0, dof) * field(dof, vd);
}
field_qp(vd, qp) = acc;
}
}
}
else if constexpr (is_gradient_fop<field_operator_t>::value)
{
const auto [num_qp, dim, num_dof] = G.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e(0), num_dof, vdim);
auto f = Reshape(&field_qp[0], vdim, dim, num_qp);
for (int qp = 0; qp < num_qp; qp++)
{
for (int vd = 0; vd < vdim; vd++)
{
for (int d = 0; d < dim; d++)
{
double acc = 0.0;
for (int dof = 0; dof < num_dof; dof++)
{
acc += G(qp, d, dof) * field(dof, vd);
}
f(vd, d, qp) = acc;
}
}
}
}
// else if constexpr (std::is_same_v<field_operator_t, FaceNormal>)
// {
// auto normal = geometric_factors.normal;
// auto [num_qp, dim, num_entities] = normal.GetShape();
// auto f = Reshape(&field_qp[0], dim, num_qp);
// for (int qp = 0; qp < num_qp; qp++)
// {
// for (int d = 0; d < dim; d++)
// {
// f(d, qp) = normal(qp, d, entity_idx);
// }
// }
// }
// TODO: Create separate function for clarity
else if constexpr (std::is_same_v<field_operator_t, Weight>)
{
const int num_qp = integration_weights.GetShape()[0];
auto f = Reshape(&field_qp[0], num_qp);
for (int qp = 0; qp < num_qp; qp++)
{
f(qp) = integration_weights(qp);
}
}
else if constexpr (is_none_fop<field_operator_t>::value)
{
auto [num_qp, unused, num_dof] = B.GetShape();
const int size_on_qp = input.size_on_qp;
const auto field = Reshape(&field_e[0], size_on_qp * num_qp);
auto f = Reshape(&field_qp[0], size_on_qp * num_qp);
for (int i = 0; i < size_on_qp * num_qp; i++)
{
f(i) = field(i);
}
}
else
{
static_assert(always_false<field_operator_t>,
"can't map field to quadrature data");
}
}
template <typename T = NonTensorProduct, typename field_operator_ts, size_t num_inputs, size_t num_fields>
MFEM_HOST_DEVICE inline
void map_fields_to_quadrature_data(
std::array<DeviceTensor<2>, num_inputs> &fields_qp,
const std::array<DeviceTensor<1>, num_fields> &fields_e,
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
const std::array<int, num_inputs> &input_to_field,
const field_operator_ts &fops,
const DeviceTensor<1, const double> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem)
{
for_constexpr<num_inputs>([&](auto i)
{
if constexpr (std::is_same_v<T, TensorProduct>)
{
map_field_to_quadrature_data_tensor_product(
fields_qp[i],
dtqmaps[i],
fields_e[input_to_field[i]],
mfem::get<i>(fops),
integration_weights,
scratch_mem);
}
else
{
map_field_to_quadrature_data(
fields_qp[i],
dtqmaps[i],
fields_e[i],
mfem::get<i>(fops),
integration_weights);
}
});
}
template <typename T, typename field_operator_t>
MFEM_HOST_DEVICE
void map_field_to_quadrature_data_conditional(
DeviceTensor<2> &field_qp,
const DeviceTensor<1> &field_e,
const DofToQuadMap &dtqmap,
field_operator_t &fop,
const DeviceTensor<1, const double> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem,
const bool &condition)
{
if (condition)
{
if constexpr (std::is_same_v<T, TensorProduct>)
{
map_field_to_quadrature_data_tensor_product(field_qp, dtqmap,
field_e, fop,
integration_weights,
scratch_mem);
}
else
{
map_field_to_quadrature_data(field_qp, dtqmap, field_e, fop,
integration_weights);
}
}
}
template <typename T = NonTensorProduct, size_t num_fields, size_t num_kinputs, typename field_operator_ts, std::size_t... i>
MFEM_HOST_DEVICE
void map_fields_to_quadrature_data_conditional(
std::array<DeviceTensor<2>, num_kinputs> &fields_qp,
const std::array<DeviceTensor<1, const double>, num_fields> &fields_e,
const std::array<DofToQuadMap, num_kinputs> &dtqmaps,
field_operator_ts fops,
const DeviceTensor<1, const double> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem,
const std::array<bool, num_kinputs> &conditions,
std::index_sequence<i...>)
{
(map_field_to_quadrature_data_conditional<T>(fields_qp[i],
fields_e[i],
dtqmaps[i],
mfem::get<i>(fops),
integration_weights,
scratch_mem,
conditions[i]),
...);
}
template <typename T = NonTensorProduct, size_t num_inputs, typename field_operator_ts>
MFEM_HOST_DEVICE
void map_direction_to_quadrature_data_conditional(
std::array<DeviceTensor<2>, num_inputs> &directions_qp,
const DeviceTensor<1> &direction_e,
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
field_operator_ts fops,
const DeviceTensor<1, const double> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem,
const std::array<bool, num_inputs> &conditions)
{
for_constexpr<num_inputs>([&](auto i)
{
map_field_to_quadrature_data_conditional<T>(directions_qp[i],
direction_e,
dtqmaps[i],
mfem::get<i>(fops),
integration_weights,
scratch_mem,
conditions[i]);
});
}
}
+99
View File
@@ -0,0 +1,99 @@
#pragma once
#include <mfem.hpp>
namespace mfem
{
class ParametricSpace
{
public:
ParametricSpace(int spatial_dim, int local_size, int element_size,
int total_size) :
spatial_dim(spatial_dim),
local_size(local_size),
element_size(element_size),
total_size(total_size),
identity(total_size)
{
dtq.ndof = (int)floor(pow(element_size, 1.0/spatial_dim) + 0.5);
dtq.nqpt = dtq.ndof;
}
ParametricSpace(int local_size) :
local_size(local_size),
element_size(local_size),
total_size(local_size),
identity(local_size)
{
dtq.ndof = (int)floor(pow(element_size, 1.0/spatial_dim) + 0.5);
dtq.nqpt = dtq.ndof;
}
int Dimension() const
{
return spatial_dim;
}
int GetLocalSize() const
{
return local_size;
}
int GetElementSize() const
{
return element_size;
}
int GetTotalSize() const
{
return total_size;
}
const DofToQuad &GetDofToQuad() const
{
return dtq;
}
const Operator *GetProlongation() const
{
return &identity;
}
const Operator *GetRestriction() const
{
return &identity;
}
private:
int spatial_dim;
// Hint for the local dimension. E.g. the size on the quadrature point or vdim.
int local_size;
// Size of the data on an element
int element_size;
int total_size;
IdentityOperator identity;
DofToQuad dtq;
};
class ParametricFunction : public Vector
{
public:
ParametricFunction(ParametricSpace &space) :
Vector(space.GetTotalSize()),
space(space)
{}
ParametricSpace &space;
using Vector::operator=;
};
}
+253
View File
@@ -0,0 +1,253 @@
#pragma once
#include "dfem_util.hpp"
#ifdef MFEM_USE_ENZYME
#include <enzyme/utils>
#include <enzyme/enzyme>
#endif
namespace mfem
{
template <typename T0, typename T1>
MFEM_HOST_DEVICE
void process_kf_arg(const T0 &, T1 &)
{
static_assert(always_false<T0, T1>,
"process_kf_arg not implemented for arg type");
}
template <typename T>
MFEM_HOST_DEVICE
void process_kf_arg(
const DeviceTensor<1, T> &u,
T &arg)
{
arg = u(0);
}
template <typename T>
MFEM_HOST_DEVICE
void process_kf_arg(
const DeviceTensor<1, T> &u,
internal::tensor<T> &arg)
{
arg(0) = u(0);
}
template <typename T, int n>
MFEM_HOST_DEVICE
void process_kf_arg(
const DeviceTensor<1> &u,
internal::tensor<T, n> &arg)
{
for (int i = 0; i < n; i++)
{
arg(i) = u(i);
}
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE
void process_kf_arg(
const DeviceTensor<1> &u,
internal::tensor<T, n, m> &arg)
{
for (int i = 0; i < m; i++)
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * m) + j);
}
}
}
template <typename arg_type>
MFEM_HOST_DEVICE
void process_kf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
{
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
process_kf_arg(u_qp, arg);
}
template <size_t num_fields, typename kf_args, std::size_t... i>
MFEM_HOST_DEVICE
void process_kf_args(
const std::array<DeviceTensor<2>, num_fields> &u,
kf_args &args,
const int &qp,
std::index_sequence<i...>)
{
(process_kf_arg(u[i], mfem::get<i>(args), qp), ...);
}
template <typename T0, typename T1> inline
Vector process_kf_result(T0, T1)
{
static_assert(always_false<T0, T1>,
"process_kf_result not implemented for result type");
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const double &x)
{
r(0) = x;
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const internal::tensor<T> &x)
{
r(0) = x(0);
}
template <typename T, int n>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const internal::tensor<T, n> &x)
{
for (size_t i = 0; i < n; i++)
{
r(i) = x(i);
}
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const internal::tensor<T, n, m> &x)
{
for (size_t i = 0; i < n; i++)
{
for (size_t j = 0; j < m; j++)
{
r(i + n * j) = x(i, j);
}
}
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
const DeviceTensor<1> &v,
double &arg)
{
arg = u(0);
}
template <int n, int m>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
const DeviceTensor<1> &v,
internal::tensor<double, n, m> &arg)
{
for (int i = 0; i < m; i++)
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * m) + j);
}
}
}
template <typename kernel_func_t, typename kernel_args_ts, size_t num_args>
MFEM_HOST_DEVICE inline
void apply_kernel(
DeviceTensor<1, double> &f_qp,
const kernel_func_t &kf,
kernel_args_ts &args,
const std::array<DeviceTensor<2>, num_args> &u,
int qp)
{
process_kf_args(u, args, qp,
std::make_index_sequence<mfem::tuple_size<kernel_args_ts>::value> {});
process_kf_result(f_qp, mfem::get<0>(mfem::apply(kf, args)));
}
#ifdef MFEM_USE_ENZYME
// Version for active function arguments only
//
// This is an Enzyme regression and can be removed in later versions.
template <typename kernel_t, typename arg_ts, std::size_t... Is,
typename inactive_arg_ts>
MFEM_HOST_DEVICE inline
auto fwddiff_apply_enzyme_indexed(kernel_t kernel, arg_ts &&args,
arg_ts &&shadow_args,
std::index_sequence<Is...>,
inactive_arg_ts &&inactive_args,
std::index_sequence<>)
{
using kf_return_t = typename create_function_signature<
decltype(&kernel_t::operator())>::type::return_t;
return __enzyme_fwddiff<kf_return_t>(
+kernel, enzyme_dup, &mfem::get<Is>(args)..., enzyme_interleave,
&mfem::get<Is>(shadow_args)...);
}
// Interleave function arguments for enzyme
template <typename kernel_t, typename arg_ts, std::size_t... Is,
typename inactive_arg_ts, std::size_t... Js>
MFEM_HOST_DEVICE inline
auto fwddiff_apply_enzyme_indexed(kernel_t kernel, arg_ts &&args,
arg_ts &&shadow_args,
std::index_sequence<Is...>,
inactive_arg_ts &&inactive_args,
std::index_sequence<Js...>)
{
using kf_return_t = typename create_function_signature<
decltype(&kernel_t::operator())>::type::return_t;
return __enzyme_fwddiff<kf_return_t>(
+kernel, enzyme_dup, &std::get<Is>(args)..., enzyme_const,
&mfem::get<Js>(inactive_args)..., enzyme_interleave,
&mfem::get<Is>(shadow_args)...);
}
template <typename kernel_t, typename arg_ts, typename inactive_arg_ts>
MFEM_HOST_DEVICE inline
auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args,
arg_ts &&shadow_args,
inactive_arg_ts &&inactive_args)
{
auto arg_indices = std::make_index_sequence<
mfem::tuple_size<std::remove_reference_t<arg_ts>>::value> {};
auto inactive_arg_indices = std::make_index_sequence<
mfem::tuple_size<std::remove_reference_t<inactive_arg_ts>>::value> {};
return fwddiff_apply_enzyme_indexed(kernel, args, shadow_args, arg_indices,
inactive_args, inactive_arg_indices);
}
template <typename kf_t, typename kernel_arg_ts, size_t num_args>
MFEM_HOST_DEVICE inline
void apply_kernel_fwddiff_enzyme(
DeviceTensor<1, double> &f_qp,
const kf_t &kf,
kernel_arg_ts &args,
kernel_arg_ts &shadow_args,
const std::array<DeviceTensor<2>, num_args> &u,
const std::array<DeviceTensor<2>, num_args> &v,
int qp_idx)
{
process_kf_args(u, args, qp_idx,
std::make_index_sequence<mfem::tuple_size<kernel_arg_ts>::value> {});
process_kf_args(v, shadow_args, qp_idx,
std::make_index_sequence<mfem::tuple_size<kernel_arg_ts>::value> {});
process_kf_result(f_qp,
mfem::get<0>(fwddiff_apply_enzyme(kf, args, shadow_args, mfem::tuple<> {})));
}
#endif // MFEM_USE_ENZYME
} // namespace mfem
+187
View File
@@ -0,0 +1,187 @@
#pragma once
#include "dfem_util.hpp"
#include "dfem_qfunction.hpp"
namespace mfem
{
MFEM_HOST_DEVICE
template <typename T0, typename T1, typename T2>
void process_kf_arg(const T0 &, const T1 &, T2 &)
{
static_assert(always_false<T0, T1, T2>,
"process_kf_arg not implemented for arg type");
}
template <typename T>
MFEM_HOST_DEVICE
void process_kf_arg(
const DeviceTensor<1, T> &u,
const DeviceTensor<1, T> &v,
T &arg)
{
arg = u(0);
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
internal::tensor<internal::dual<T, T>, n, m> &arg)
{
for (int i = 0; i < m; i++)
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * m) + j);
}
}
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
internal::dual<T, T> &arg)
{
arg.value = u(0);
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
const DeviceTensor<1> &v,
internal::dual<T, T> &arg)
{
arg.value = u(0);
arg.gradient = v(0);
}
template <typename T, int n>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
const DeviceTensor<1> &v,
internal::tensor<internal::dual<T, T>, n> &arg)
{
for (int i = 0; i < n; i++)
{
arg(i).value = u(i);
arg(i).gradient = v(i);
}
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<1> &u,
const DeviceTensor<1> &v,
internal::tensor<internal::dual<T, T>, n, m> &arg)
{
for (int i = 0; i < m; i++)
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * m) + j);
arg(j, i).gradient = v((i * m) + j);
}
}
}
template <typename T, int n>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const internal::tensor<internal::dual<T, T>, n> &x)
{
for (size_t i = 0; i < n; i++)
{
r(i) = x(i).value;
}
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE inline
void process_kf_result(
DeviceTensor<1, T> &r,
const internal::tensor<internal::dual<T, T>, n, m> &x)
{
for (size_t i = 0; i < n; i++)
{
for (size_t j = 0; j < m; j++)
{
r(i + n * j) = x(i, j).value;
}
}
}
template <typename arg_type>
MFEM_HOST_DEVICE inline
void process_kf_arg(
const DeviceTensor<2> &u,
const DeviceTensor<2> &v,
arg_type &arg,
const int &qp)
{
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
const auto v_qp = Reshape(&v(0, qp), v.GetShape()[0]);
process_kf_arg(u_qp, v_qp, arg);
}
template <size_t num_args, typename kf_args, std::size_t... Is>
MFEM_HOST_DEVICE inline
void process_kf_args(
const std::array<DeviceTensor<2>, num_args> &u,
const std::array<DeviceTensor<2>, num_args> &v,
kf_args &args,
const int &qp,
std::index_sequence<Is...>)
{
(process_kf_arg(u[Is], v[Is], mfem::get<Is>(args), qp), ...);
}
template <typename T, int n, int m>
MFEM_HOST_DEVICE inline
void process_derivative_from_native_dual(
DeviceTensor<1, T> &r,
const internal::tensor<internal::dual<T, T>, n, m> &x)
{
for (size_t i = 0; i < n; i++)
{
for (size_t j = 0; j < m; j++)
{
r(i + n * j) = x(i, j).gradient;
}
}
}
template <typename T, int n>
MFEM_HOST_DEVICE inline
void process_derivative_from_native_dual(
DeviceTensor<1, T> &r,
const internal::tensor<internal::dual<T, T>, n> &x)
{
for (size_t i = 0; i < n; i++)
{
r(i) = x(i).gradient;
}
}
template <typename kf_t, typename kernel_arg_ts, size_t num_args>
MFEM_HOST_DEVICE inline
void apply_kernel_native_dual(
DeviceTensor<1, double> &f_qp,
const kf_t &kf,
kernel_arg_ts &args,
const std::array<DeviceTensor<2>, num_args> &u,
const std::array<DeviceTensor<2>, num_args> &v,
const int &qp_idx)
{
process_kf_args(u, v, args, qp_idx,
std::make_index_sequence<mfem::tuple_size<kernel_arg_ts>::value> {});
auto r = mfem::get<0>(mfem::apply(kf, args));
process_derivative_from_native_dual(f_qp, r);
}
} // namespace mfem
+530
View File
@@ -0,0 +1,530 @@
#pragma once
#include <mfem.hpp>
#include <utility>
#include "dfem_interpolate.hpp"
#include "dfem_integrate.hpp"
#include "dfem_qfunction.hpp"
#include "dfem_qfunction_dual.hpp"
#include "examples/dfem/dfem_util.hpp"
namespace mfem
{
class DerivativeOperator : public Operator
{
using derivative_action_t =
std::function<void(std::vector<Vector> &, const Vector &, Vector &)>;
using restriction_callback_t =
std::function<void(std::vector<Vector> &,
const std::vector<Vector> &,
std::vector<Vector> &)>;
public:
DerivativeOperator(
const std::vector<derivative_action_t> &derivative_actions,
const FieldDescriptor &direction,
const std::vector<Vector *> &solutions_l,
const std::vector<Vector *> &parameters_l,
const std::vector<restriction_callback_t> &restriction_callbacks,
const std::function<void(Vector &, Vector &)> prolongation_transpose) :
derivative_actions(derivative_actions),
direction(direction),
restriction_callbacks(restriction_callbacks),
derivative_action_l(GetVSize(direction)),
prolongation_transpose(prolongation_transpose)
{
MFEM_ASSERT(derivative_actions.size() == restriction_callbacks.size(),
"internal error");
derivative_action_l = 0.0;
this->solutions_l.resize(solutions_l.size());
this->parameters_l.resize(parameters_l.size());
for (int i = 0; i < solutions_l.size(); i++)
{
this->solutions_l[i] = *solutions_l[i];
}
for (int i = 0; i < parameters_l.size(); i++)
{
this->parameters_l[i] = *parameters_l[i];
}
fields_e.resize(solutions_l.size() + parameters_l.size());
}
void Mult(const Vector &x, Vector &y) const override
{
direction_t = x;
direction_t.SetSubVector(ess_tdof_list, 0.0);
prolongation(direction, direction_t, direction_l);
for (int i = 0; i < derivative_actions.size(); i++)
{
restriction_callbacks[i](solutions_l, parameters_l, fields_e);
derivative_actions[i](fields_e, direction_l, derivative_action_l);
}
prolongation_transpose(derivative_action_l, y);
y.SetSubVector(ess_tdof_list, 0.0);
};
private:
std::vector<derivative_action_t> derivative_actions;
mutable std::vector<Vector> solutions_l;
std::vector<Vector> parameters_l;
FieldDescriptor direction;
mutable Vector direction_t;
mutable Vector direction_e;
mutable Vector direction_l;
mutable Vector derivative_action_e;
mutable Vector derivative_action_l;
mutable std::vector<Vector> fields_e;
Array<int> ess_tdof_list;
std::vector<restriction_callback_t> restriction_callbacks;
std::function<void(Vector &, Vector &)> prolongation_transpose;
};
class DifferentiableOperator : public Operator
{
using action_t =
std::function<void(std::vector<Vector> &, const std::vector<Vector> &, Vector &)>;
using derivative_action_t =
std::function<void(std::vector<Vector> &, const Vector &, Vector &)>;
using restriction_callback_t =
std::function<void(std::vector<Vector> &,
const std::vector<Vector> &,
std::vector<Vector> &)>;
public:
DifferentiableOperator(
const std::vector<FieldDescriptor> &solutions,
const std::vector<FieldDescriptor> &parameters,
const ParMesh &mesh);
void Mult(const Vector &x, Vector &y) const override
{
MFEM_ASSERT(!action_callbacks.empty(), "no integrators have been set");
prolongation(solutions, x, solutions_l);
for (auto &action : action_callbacks)
{
action(solutions_l, parameters_l, residual_l);
}
prolongation_transpose(residual_l, y);
y.SetSubVector(ess_tdof_list, 0.0);
}
template <
typename func_t,
typename... input_ts,
typename... output_ts,
typename derivative_indices_t>
void AddDomainIntegrator(
func_t qfunc,
mfem::tuple<input_ts...> inputs,
mfem::tuple<output_ts...> outputs,
const IntegrationRule &integration_rule,
const derivative_indices_t derivative_indices = {});
void SetParameters(std::vector<Vector *> p) const;
std::shared_ptr<DerivativeOperator> GetDerivative(
size_t derivative_idx,
std::vector<Vector *> solutions_l,
std::vector<Vector *> parameters_l)
{
MFEM_ASSERT(derivative_action_callbacks.find(derivative_idx) !=
derivative_action_callbacks.end(),
"no derivative action has been found for index " << derivative_idx);
return std::make_shared<DerivativeOperator>(
derivative_action_callbacks[derivative_idx],
fields[derivative_idx],
solutions_l,
parameters_l,
restriction_callbacks,
prolongation_transpose);
}
private:
const ParMesh &mesh;
std::vector<action_t> action_callbacks;
std::map<size_t, std::vector<derivative_action_t>> derivative_action_callbacks;
std::vector<FieldDescriptor> solutions;
std::vector<FieldDescriptor> parameters;
// solutions and parameters
std::vector<FieldDescriptor> fields;
Array<int> ess_tdof_list;
mutable std::vector<Vector> solutions_l;
mutable std::vector<Vector> parameters_l;
mutable Vector residual_l;
mutable std::vector<Vector> fields_e;
mutable Vector residual_e;
std::function<void(Vector &, Vector &)> prolongation_transpose;
std::vector<restriction_callback_t> restriction_callbacks;
};
void DifferentiableOperator::SetParameters(std::vector<Vector *> p) const
{
MFEM_ASSERT(parameters.size() == p.size(),
"number of parameters doesn't match descriptors");
for (int i = 0; i < parameters.size(); i++)
{
p[i]->Read();
parameters_l[i] = *p[i];
}
}
DifferentiableOperator::DifferentiableOperator(
const std::vector<FieldDescriptor> &solutions,
const std::vector<FieldDescriptor> &parameters,
const ParMesh &mesh) :
mesh(mesh),
solutions(solutions),
parameters(parameters)
{
fields.resize(solutions.size() + parameters.size());
fields_e.resize(fields.size());
solutions_l.resize(solutions.size());
parameters_l.resize(parameters.size());
for (int i = 0; i < solutions.size(); i++)
{
fields[i] = solutions[i];
}
for (int i = 0; i < parameters.size(); i++)
{
fields[i + solutions.size()] = parameters[i];
}
}
template <
typename func_t,
typename... input_ts,
typename... output_ts,
typename derivative_indices_t = std::make_index_sequence<0>>
void DifferentiableOperator::AddDomainIntegrator(
func_t qfunc,
mfem::tuple<input_ts...> inputs,
mfem::tuple<output_ts...> outputs,
const IntegrationRule &integration_rule,
const derivative_indices_t derivative_indices)
{
using entity_t = Entity::Element;
static constexpr size_t num_inputs =
mfem::tuple_size<decltype(inputs)>::value;
static constexpr size_t num_outputs =
mfem::tuple_size<decltype(outputs)>::value;
using qf_param_ts = typename create_function_signature<
decltype(&func_t::operator())>::type::parameter_ts;
using qf_output_t = typename create_function_signature<
decltype(&func_t::operator())>::type::return_t;
// Consistency checks
if constexpr (num_outputs > 1)
{
static_assert(always_false<func_t>,
"more than one output per kernel is not supported right now");
}
constexpr size_t num_qfinputs = mfem::tuple_size<qf_param_ts>::value;
static_assert(num_qfinputs == num_inputs,
"kernel function inputs and descriptor inputs have to match");
constexpr size_t num_qf_outputs = mfem::tuple_size<qf_output_t>::value;
static_assert(num_qf_outputs == num_qf_outputs,
"kernel function outputs and descriptor outputs have to match");
constexpr auto field_tuple = std::tuple_cat(std::tuple<input_ts...> {},
std::tuple<output_ts...> {});
constexpr auto filtered_field_tuple = filter_fields(field_tuple);
constexpr size_t num_fields = count_unique_field_ids(filtered_field_tuple);
constexpr auto dependency_map = make_dependency_map(mfem::tuple<input_ts...> {});
// Create the action callback
auto input_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
inputs,
std::make_index_sequence<num_inputs> {});
auto output_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
outputs,
std::make_index_sequence<num_outputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = output_to_field[hardcoded_output_idx];
ElementDofOrdering element_dof_ordering = ElementDofOrdering::LEXICOGRAPHIC;
DofToQuad::Mode doftoquad_mode = DofToQuad::Mode::TENSOR;
const Operator *R = get_restriction<entity_t>(fields[test_space_field_idx],
element_dof_ordering);
// The explicit captures are necessary to avoid dependency on
// the specific instance of this class (this pointer).
auto restriction_callback =
[=, solutions = this->solutions, parameters = this->parameters]
(std::vector<Vector> &solutions_l,
const std::vector<Vector> &parameters_l,
std::vector<Vector> &fields_e)
{
restriction<entity_t>(solutions, solutions_l, fields_e,
element_dof_ordering);
restriction<entity_t>(parameters, parameters_l, fields_e,
element_dof_ordering,
solutions.size());
};
restriction_callbacks.push_back(restriction_callback);
auto output_fop = mfem::get<hardcoded_output_idx>(outputs);
if constexpr (is_none_fop<decltype(output_fop)>::value)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
y = r_local;
};
}
// else if constexpr (std::is_same_v<decltype(output_fop), One>)
// {
// prolongation_transpose = [&](Vector &r_local, Vector &y)
// {
// double local_sum = r_local.Sum();
// MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
// op.mesh.GetComm());
// MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
// };
// }
else
{
auto P = get_prolongation(fields[test_space_field_idx]);
prolongation_transpose = [P](const Vector &r_local, Vector &y)
{
P->MultTranspose(r_local, y);
};
}
const int num_elements = GetNumEntities<Entity::Element>(mesh);
const int num_entities = GetNumEntities<entity_t>(mesh);
const int num_qp = integration_rule.GetNPoints();
size_t residual_lsize = GetVSize(fields[test_space_field_idx]);
// if constexpr (std::is_same_v<decltype(output_fop), One>)
// {
// this->width = 1;
// }
// else
{
width = residual_lsize;
}
residual_l.SetSize(residual_lsize);
std::vector<const DofToQuad*> dtq;
for (const auto &field : fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(
field,
integration_rule,
doftoquad_mode));
}
const int q1d = (int)floor(pow(num_qp, 1.0/mesh.Dimension()) + 0.5);
residual_e.SetSize(R->Height());
const int residual_size_on_qp =
GetSizeOnQP<entity_t>(mfem::get<hardcoded_output_idx>(outputs),
fields[test_space_field_idx]);
auto input_dtq_maps =
create_dtq_maps<entity_t>(inputs, dtq, input_to_field);
auto output_dtq_maps =
create_dtq_maps<entity_t>(outputs, dtq, output_to_field);
const int test_vdim = mfem::get<hardcoded_output_idx>(outputs).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(inputs).size_on_qp /
mfem::get<hardcoded_output_idx>(outputs).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(outputs).vdim /
num_entities;
auto ir_weights = Reshape(integration_rule.GetWeights().Read(), num_qp);
auto input_size_on_qp =
get_input_size_on_qp(inputs, std::make_index_sequence<num_inputs> {});
auto action_shmem_info =
get_shmem_info<entity_t, num_fields, num_inputs, num_outputs>
(input_dtq_maps, output_dtq_maps, fields, num_entities, inputs, num_qp,
input_size_on_qp, residual_size_on_qp);
Vector shmem_cache(action_shmem_info.total_size);
// print_shared_memory_info(action_shmem_info);
action_callbacks.push_back(
[=](std::vector<Vector> &solutions_l,
const std::vector<Vector> &parameters_l,
Vector &residual_l) mutable
{
restriction_callback(solutions_l, parameters_l, fields_e);
residual_e = 0.0;
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof, num_entities);
auto wrapped_fields_e = wrap_fields(fields_e,
action_shmem_info.field_sizes,
num_entities);
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, input_shmem,
residual_shmem, scratch_shmem] =
unpack_shmem(shmem, action_shmem_info, input_dtq_maps, output_dtq_maps,
wrapped_fields_e, num_qp, e);
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_to_field, inputs, ir_weights,
scratch_shmem);
call_qfunction<TensorProduct, qf_param_ts>(
qfunc, input_shmem, residual_shmem,
residual_size_on_qp, num_qp, q1d);
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(outputs),
output_dtq_shmem[hardcoded_output_idx],
scratch_shmem);
}, num_entities, q1d, q1d, q1d, action_shmem_info.total_size, shmem_cache.ReadWrite());
if constexpr (is_none_fop<decltype(output_fop)>::value)
{
residual_l = residual_e;
}
else
{
R->MultTranspose(residual_e, residual_l);
}
});
for_constexpr([&](auto derivative_idx)
{
// bool is_dependent = false;
// for_constexpr<num_inputs>([&](auto input_idx)
// {
// constexpr auto input_is_dependent_on_field_idx =
// std::get<derivative_idx>(std::get<input_idx>(dependency_map));
// if constexpr (input_is_dependent_on_field_idx == 1)
// {
// is_dependent = true;
// }
// });
// if (!is_dependent)
// {
// derivative_action_callbacks[derivative_idx].push_back(
// [=](const Vector &direction_l, Vector &y) mutable
// {
// y += 0.0;
// });
// return;
// }
auto direction = fields[derivative_idx];
size_t derivative_action_l_size = GetVSize(direction);
const int da_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(outputs),
fields[test_space_field_idx]);
auto shmem_info =
get_shmem_info<entity_t, num_fields, num_inputs, num_outputs>
(input_dtq_maps, output_dtq_maps, fields, num_entities, inputs, num_qp,
input_size_on_qp, residual_size_on_qp, derivative_idx);
Vector shmem_cache(shmem_info.total_size);
// print_shared_memory_info(shmem_info);
Vector direction_e;
Vector derivative_action_e(R->Height());
derivative_action_e = 0.0;
auto input_is_dependent = get_array_from_tuple(std::get<derivative_idx>
(dependency_map));
derivative_action_callbacks[derivative_idx].push_back(
[=](std::vector<Vector> &fields_e, const Vector &direction_l,
Vector &derivative_action_l) mutable
{
restriction<entity_t>(direction, direction_l, direction_e, element_dof_ordering);
auto ye = Reshape(derivative_action_e.ReadWrite(), num_test_dof, test_vdim, num_entities);
auto wrapped_fields_e = wrap_fields(fields_e, shmem_info.field_sizes, num_entities);
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(), shmem_info.direction_size, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
{
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, direction_shmem,
input_shmem, shadow_shmem, residual_shmem, scratch_shmem] =
unpack_shmem(shmem, shmem_info, input_dtq_maps,
output_dtq_maps, wrapped_fields_e, wrapped_direction_e, num_qp, e);
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, input_to_field, inputs, ir_weights,
scratch_shmem);
zero_all(shadow_shmem);
map_direction_to_quadrature_data_conditional<TensorProduct>(
shadow_shmem, direction_shmem, input_dtq_shmem, inputs, ir_weights,
scratch_shmem, input_is_dependent);
call_qfunction_derivative_action<TensorProduct, qf_param_ts>(
qfunc, input_shmem, shadow_shmem, residual_shmem,
da_size_on_qp, num_qp, q1d);
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(outputs),
output_dtq_shmem[hardcoded_output_idx],
scratch_shmem);
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
R->MultTranspose(derivative_action_e, derivative_action_l);
});
}, derivative_indices);
}
} // namespace mfem
// #include "dfem_refactor_action.hpp"
// #include "dfem_refactor_derivatives.hpp"
+232
View File
@@ -0,0 +1,232 @@
#pragma once
#include "dfem_refactor.hpp"
namespace mfem
{
template <typename element_operator_t, size_t num_fields>
void DifferentiableOperator::instantiate_action(
element_operator_t element_operator, action_t &action)
{
using entity_t = typename element_operator_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
element_operator.inputs,
std::make_index_sequence<element_operator.num_inputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
element_operator.outputs,
std::make_index_sequence<element_operator.num_outputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(element_operator.outputs);
const int num_elements = GetNumEntities<Entity::Element>(mesh);
const int num_entities = GetNumEntities<entity_t>(mesh);
const int num_qp = integration_rule.GetNPoints();
this->width = GetTrueVSize(fields[test_space_field_idx]);
size_t residual_lsize = GetVSize(fields[test_space_field_idx]);
// if constexpr (std::is_same_v<decltype(output_fop), One>)
// {
// this->width = 1;
// }
// else
{
this->width = residual_lsize;
}
residual_l.SetSize(residual_lsize);
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(
field,
integration_rule,
doftoquad_mode));
}
const int q1d = (int)floor(pow(num_qp, 1.0/mesh.Dimension()) + 0.5);
residual_e.SetSize(R->Height());
const int residual_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(element_operator.outputs),
fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(element_operator.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(element_operator.outputs, dtq,
koutput_to_field);
// auto input_fops = create_bare_fops(element_operator.inputs);
// auto output_fops = create_bare_fops(element_operator.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>
(element_operator.outputs).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(element_operator.inputs).size_on_qp /
mfem::get<hardcoded_output_idx>(element_operator.outputs).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(element_operator.outputs).vdim /
num_entities;
auto ir_weights = Reshape(integration_rule.GetWeights().Read(), num_qp);
auto input_size_on_qp = get_input_size_on_qp(
element_operator.inputs,
std::make_index_sequence<element_operator.num_inputs> {});
auto shmem_info =
get_shmem_info<entity_t, num_fields, element_operator.num_inputs, element_operator.num_outputs>
(input_dtq_maps,
output_dtq_maps,
fields,
num_entities,
element_operator.inputs,
num_qp,
input_size_on_qp,
residual_size_on_qp);
Vector shmem_cache(shmem_info.total_size);
print_shared_memory_info(shmem_info);
action = [=](const Vector &x, Vector &y) mutable
{
prolongation(solutions, x, solutions_l);
restriction<entity_t>(solutions, solutions_l, this->fields_e,
element_dof_ordering);
restriction<entity_t>(parameters, parameters_l, this->fields_e,
element_dof_ordering,
solutions.size());
residual_e = 0.0;
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof,
num_entities);
auto wrapped_fields_e = wrap_fields(this->fields_e,
shmem_info.field_sizes,
num_entities);
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
// printf("\ne: %d\n", e);
// tic();
auto input_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
shmem_info.input_dtq_sizes,
input_dtq_maps);
auto output_dtq_shmem = load_dtq_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
shmem_info.output_dtq_sizes,
output_dtq_maps);
auto fields_shmem = load_field_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::FIELD],
shmem_info.field_sizes,
kinput_to_field,
element_operator.inputs,
wrapped_fields_e,
e,
std::make_index_sequence<element_operator.num_inputs> {});
// These functions don't copy, they simply create a `DeviceTensor` object
// that points to correct chunks of the shared memory pool.
auto input_shmem = load_input_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::INPUT],
shmem_info.input_sizes,
num_qp);
auto residual_shmem = load_residual_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::OUTPUT],
shmem_info.residual_size,
num_qp);
auto scratch_mem = load_scratch_mem(
shmem,
shmem_info.offsets[SharedMemory::Index::TEMP],
shmem_info.temp_sizes);
MFEM_SYNC_THREAD;
// // printf("shmem load elapsed: %.1fus\n", toc() * 1e6);
// // tic();
map_fields_to_quadrature_data<TensorProduct>(
input_shmem, fields_shmem, input_dtq_shmem, element_operator.inputs, ir_weights,
scratch_mem,
std::make_index_sequence<element_operator.num_inputs> {});
// printf("interpolate elapsed: %.1fus\n", toc() * 1e6);
// // tic();
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
auto qf_args = decay_tuple<typename element_operator_t::qf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), residual_size_on_qp);
apply_kernel(r, element_operator.qfunc, qf_args, input_shmem, q);
}
}
}
MFEM_SYNC_THREAD;
// // printf("qf elapsed: %.1fus\n", toc() * 1e6);
// // tic();
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
mfem::get<0>(element_operator.outputs),
output_dtq_shmem[hardcoded_output_idx],
scratch_mem);
// printf("integrate elapsed: %.1fus\n", toc() * 1e6);
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
if constexpr (std::is_same_v<decltype(output_fop), None<>>)
{
residual_l = y;
}
else
{
R->MultTranspose(residual_e, residual_l);
}
if constexpr (std::is_same_v<decltype(output_fop), None<>>)
{
y = residual_l;
}
// else if constexpr (std::is_same_v<decltype(output_fop), One>)
// {
// double local_sum = residual_l.Sum();
// MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM, mesh.GetComm());
// MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
// }
else
{
get_prolongation(fields[test_space_field_idx])->MultTranspose(residual_l, y);
}
};
}
}
+132
View File
@@ -0,0 +1,132 @@
#pragma once
#include "dfem_refactor.hpp"
template<typename T, T... Ints>
void print_sequence(std::integer_sequence<T, Ints...>)
{
((std::cout << Ints << " "), ...);
std::cout << std::endl;
}
namespace mfem
{
template <
typename element_operator_t,
size_t num_solutions,
size_t num_parameters,
size_t derivative_idx>
DerivativeOperator::DerivativeOperator(
element_operator_t element_operator,
const std::array<FieldDescriptor, num_solutions> &solutions,
const std::array<FieldDescriptor, num_parameters> &parameters,
const std::vector<FieldDescriptor> &fields,
ParMesh &mesh,
const IntegrationRule &integration_rule,
const ElementDofOrdering &element_dof_ordering,
const DofToQuad::Mode &doftoquad_mode,
std::integral_constant<size_t, derivative_idx>)
{
direction = fields[derivative_idx];
size_t derivative_action_l_size = 0;
for (auto &s : solutions)
{
derivative_action_l_size += GetVSize(s);
this->width += GetTrueVSize(s);
}
this->height = derivative_action_l_size;
derivative_action_l.SetSize(derivative_action_l_size);
constexpr size_t num_fields = num_solutions + num_parameters;
using entity_t = typename element_operator_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
element_operator.inputs,
std::make_index_sequence<element_operator.num_inputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(
fields,
element_operator.outputs,
std::make_index_sequence<element_operator.num_outputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(element_operator.outputs);
const int num_elements = GetNumEntities<Entity::Element>(mesh);
const int num_entities = GetNumEntities<entity_t>(mesh);
const int num_qp = integration_rule.GetNPoints();
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(
field,
integration_rule,
doftoquad_mode));
}
const int q1d = (int)floor(pow(num_qp, 1.0/mesh.Dimension()) + 0.5);
derivative_action_e.SetSize(R->Height());
const int da_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(element_operator.outputs),
fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(element_operator.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(element_operator.outputs, dtq,
koutput_to_field);
const int test_vdim = mfem::get<hardcoded_output_idx>
(element_operator.outputs).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(element_operator.inputs).size_on_qp /
mfem::get<hardcoded_output_idx>(element_operator.outputs).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(element_operator.outputs).vdim /
num_entities;
auto ir_weights = Reshape(integration_rule.GetWeights().Read(), num_qp);
auto input_size_on_qp = get_input_size_on_qp(
element_operator.inputs,
std::make_index_sequence<element_operator.num_inputs> {});
auto input_is_dependent = std::get<derivative_idx>
(element_operator.dependency_map);
constexpr bool with_derivatives = true;
auto shmem_info =
get_shmem_info<entity_t, num_fields, element_operator.num_inputs, element_operator.num_outputs>
(input_dtq_maps,
output_dtq_maps,
fields,
num_entities,
element_operator.inputs,
num_qp,
input_size_on_qp,
da_size_on_qp,
derivative_idx);
Vector shmem_cache(shmem_info.total_size);
print_shared_memory_info(shmem_info);
action_callback = [=](const Vector &x, Vector &y) mutable
{
restriction<entity_t>(direction, direction_l, direction_e,
element_dof_ordering);
};
}
} // namespace mfem
+116
View File
@@ -0,0 +1,116 @@
#pragma once
#include <mfem.hpp>
class SharedMemoryManager
{
private:
struct MemoryBlock
{
char* ptr;
int size;
bool used;
};
MFEM_HOST_DEVICE static const int MAX_BLOCKS = 16;
MFEM_HOST_DEVICE static MemoryBlock blocks[MAX_BLOCKS];
MFEM_HOST_DEVICE static int num_blocks;
MFEM_HOST_DEVICE static char* base_ptr;
public:
MFEM_HOST_DEVICE static void init(void* shmem, int total_size)
{
base_ptr = static_cast<char*>(shmem);
num_blocks = 1;
blocks[0] = {base_ptr, total_size, false};
}
template<typename T>
MFEM_HOST_DEVICE static T* reserve(int n)
{
int size_bytes = n * sizeof(T);
for (int i = 0; i < num_blocks; ++i)
{
if (!blocks[i].used && blocks[i].size >= size_bytes)
{
blocks[i].used = true;
if (blocks[i].size > size_bytes)
{
// Split block
if (num_blocks < MAX_BLOCKS)
{
blocks[num_blocks] = {blocks[i].ptr + size_bytes, blocks[i].size - size_bytes, false};
++num_blocks;
blocks[i].size = size_bytes;
}
}
return reinterpret_cast<T*>(blocks[i].ptr);
}
}
return nullptr; // Allocation failed
}
MFEM_HOST_DEVICE static void release(void* ptr)
{
for (int i = 0; i < num_blocks; ++i)
{
if (blocks[i].ptr == ptr)
{
blocks[i].used = false;
return;
}
}
}
MFEM_HOST_DEVICE static void release_and_try_merge(void* ptr)
{
for (int i = 0; i < num_blocks; ++i)
{
if (blocks[i].ptr == ptr)
{
blocks[i].used = false;
merge_adjacent_free_blocks();
return;
}
}
}
private:
MFEM_HOST_DEVICE static void merge_adjacent_free_blocks()
{
// Simple bubble sort for simplicity (can be optimized)
for (int i = 0; i < num_blocks - 1; ++i)
{
for (int j = 0; j < num_blocks - i - 1; ++j)
{
if (blocks[j].ptr > blocks[j + 1].ptr)
{
MemoryBlock temp = blocks[j];
blocks[j] = blocks[j + 1];
blocks[j + 1] = temp;
}
}
}
for (int i = 0; i < num_blocks - 1; ++i)
{
if (!blocks[i].used && !blocks[i + 1].used)
{
blocks[i].size += blocks[i + 1].size;
for (int j = i + 1; j < num_blocks - 1; ++j)
{
blocks[j] = blocks[j + 1];
}
--num_blocks;
--i;
}
}
}
};
MFEM_HOST_DEVICE SharedMemoryManager::MemoryBlock
SharedMemoryManager::blocks[SharedMemoryManager::MAX_BLOCKS];
MFEM_HOST_DEVICE int SharedMemoryManager::num_blocks;
MFEM_HOST_DEVICE char* SharedMemoryManager::base_ptr;
+39
View File
@@ -0,0 +1,39 @@
#pragma once
#include "dfem_refactor.hpp"
#define DFEM_TEST_MAIN(function) \
int main(int argc, char* argv[]) \
{ \
Mpi::Init(); \
\
const char* device_config = "cpu"; \
const char* mesh_file = "../data/ref-square.mesh"; \
int polynomial_order = 1; \
int ir_order = 2; \
int refinements = 0; \
\
OptionsParser args(argc, argv); \
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use."); \
args.AddOption(&polynomial_order, "-o", "--order", ""); \
args.AddOption(&refinements, "-r", "--r", ""); \
args.AddOption(&ir_order, "-iro", "--iro", ""); \
args.AddOption(&device_config, "-d", "--device", \
"Device configuration string, see Device::Configure()."); \
args.ParseCheck(); \
\
Device device(device_config); \
if (Mpi::Root() == 0) \
{ \
device.Print(); \
} \
\
out << std::setprecision(12); \
\
int ret; \
\
ret = function(mesh_file, refinements, polynomial_order); \
out << #function; \
ret ? out << " FAILURE\n" : out << " OK\n"; \
\
return ret; \
}\
File diff suppressed because it is too large Load Diff
+130
View File
@@ -0,0 +1,130 @@
// SPDX-ArtifactOfProjectName: noisy
// SPDX-ArtifactOfProjectHomePage: https://github.com/VincentZalzal/noisy
// SPDX-FileCopyrightText: Copyright 2024 Vincent Zalzal
// SPDX-License-Identifier: MIT
#pragma once
#include <iomanip>
#include <iostream>
namespace vz {
struct Counters {
unsigned m_def_ctor = 0;
unsigned m_copy_ctor = 0;
unsigned m_move_ctor = 0;
unsigned m_copy_assign = 0;
unsigned m_move_assign = 0;
unsigned m_dtor = 0;
void reset() {
*this = {};
}
bool leaks() const {
return m_def_ctor + m_copy_ctor + m_move_ctor != m_dtor;
}
friend std::ostream& operator<<(std::ostream& os, const Counters& c) {
stream_counter(os, "Default constructor count: ", c.m_def_ctor );
stream_counter(os, "Copy constructor count: ", c.m_copy_ctor );
stream_counter(os, "Move constructor count: ", c.m_move_ctor );
stream_counter(os, "Copy assignment count: ", c.m_copy_assign);
stream_counter(os, "Move assignment count: ", c.m_move_assign);
stream_counter(os, "Destructor count: ", c.m_dtor );
return os;
}
friend bool operator==(const Counters& lhs, const Counters& rhs) {
return
lhs.m_def_ctor == rhs.m_def_ctor &&
lhs.m_copy_ctor == rhs.m_copy_ctor &&
lhs.m_move_ctor == rhs.m_move_ctor &&
lhs.m_copy_assign == rhs.m_copy_assign &&
lhs.m_move_assign == rhs.m_move_assign &&
lhs.m_dtor == rhs.m_dtor ;
}
friend bool operator!=(const Counters& lhs, const Counters& rhs) { return !(lhs == rhs); }
private:
static void stream_counter(std::ostream& os, const char* msg, unsigned value) {
if (value != 0)
os << msg << std::setw(2) << value << '\n';
}
};
namespace detail {
struct Globals {
~Globals() {
if (m_verbose)
std::cout << "\n===== Noisy counters =====\n" << m_counters;
}
Counters m_counters;
unsigned m_next_id = 0;
bool m_verbose = true;
};
}
class Noisy {
private:
static detail::Globals& globals() {
static detail::Globals s_globals;
return s_globals;
}
public:
static Counters& counters() { return globals().m_counters; }
static void set_verbose(bool verbose) { globals().m_verbose = verbose; }
Noisy() {
if (globals().m_verbose)
std::cout << *this << ": default constructor\n";
globals().m_counters.m_def_ctor++;
}
Noisy(const Noisy& other) {
if (globals().m_verbose)
std::cout << *this << ": copy constructor from " << other << '\n';
globals().m_counters.m_copy_ctor++;
}
Noisy(Noisy&& other) noexcept {
if (globals().m_verbose)
std::cout << *this << ": move constructor from " << other << '\n';
globals().m_counters.m_move_ctor++;
}
~Noisy() {
if (globals().m_verbose)
std::cout << *this << ": destructor\n";
globals().m_counters.m_dtor++;
}
Noisy& operator=(const Noisy& other) {
if (globals().m_verbose)
std::cout << *this << ": copy assignment from " << other << '\n';
globals().m_counters.m_copy_assign++;
return *this;
}
Noisy& operator=(Noisy&& other) noexcept {
if (globals().m_verbose)
std::cout << *this << ": move assignment from " << other << '\n';
globals().m_counters.m_move_assign++;
return *this;
}
unsigned id() const { return m_id; }
friend std::ostream& operator<<(std::ostream& os, const Noisy& noisy) { return os << "Noisy(" << std::setw(2) << noisy.m_id << ')'; }
private:
unsigned m_id = globals().m_next_id++;
};
}
@@ -0,0 +1,188 @@
{
using entity_t = typename kernel_t::entity_t;
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
constexpr int hardcoded_output_idx = 0;
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
element_dof_ordering);
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
const int num_entities = GetNumEntities<entity_t>(op.mesh);
const int num_qp = op.integration_rule.GetNPoints();
// assume only a single element type for now
std::vector<const DofToQuad*> dtq;
for (const auto &field : op.fields)
{
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
doftoquad_mode));
}
const int q1d = dtq[0]->nqpt;
derivative_action_e.SetSize(R->Height());
const int da_size_on_qp = GetSizeOnQP<entity_t>(
mfem::get<hardcoded_output_idx>(kernel.outputs),
op.fields[test_space_field_idx]);
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
kinput_to_field);
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
koutput_to_field);
auto input_fops = create_bare_fops(kernel.inputs);
auto output_fops = create_bare_fops(kernel.outputs);
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int test_op_dim =
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
mfem::get<hardcoded_output_idx>(output_fops).vdim;
const int num_test_dof = R->Height() /
mfem::get<hardcoded_output_idx>(output_fops).vdim /
num_entities;
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
num_qp);
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
std::make_index_sequence<kernel.num_kinputs> {});
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
output_dtq_maps,
op.fields,
num_entities,
kernel.inputs,
num_qp,
input_size_on_qp,
da_size_on_qp);
Vector shmem_cache(shmem_info.total_size);
func = [=](Vector &ye_mem) mutable
{
restriction<entity_t>(direction, direction_l, direction_e,
op.element_dof_ordering, derivative_idx);
// Check which qf inputs are dependent on the dependent variable
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
bool no_qfinput_is_dependent = true;
for (int i = 0; i < kinput_is_dependent.size(); i++)
{
if (kinput_to_field[i] == derivative_idx)
{
no_qfinput_is_dependent = false;
kinput_is_dependent[i] = true;
// out << "function input " << i << " is dependent on "
// << op.fields[kinput_to_field[i]].field_label << "\n";
}
else
{
kinput_is_dependent[i] = false;
}
}
if (no_qfinput_is_dependent)
{
return;
}
// auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
// auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
// DeviceTensor<1, const double> integration_weights(
// this->op.integration_rule.GetWeights().Read(), num_qp);
// Vector zero;
// GeometricFactorMaps geometric_factors
// {
// DeviceTensor<3, const double>(zero.Read(), 0, 0, 0)
// };
// // Fields interpolated to the quadrature points in the order of
// // kernel function arguments
// auto input_qp = map_inputs_to_memory(input_qp_mem, num_qp,
// kernel.inputs,
// std::make_index_sequence<kernel.num_kinputs> {});
// auto directions_qp = map_inputs_to_memory(directions_qp_mem, num_qp,
// kernel.inputs,
// std::make_index_sequence<kernel.num_kinputs> {});
// constexpr int fixed_output_idx = 0;
// auto Bv = output_dtq_maps[fixed_output_idx];
// auto [num_test_qp, test_op_dim, num_test_dof] = Bv.GetShape();
// const int test_vdim = mfem::get<0>(kernel.outputs).vdim;
// DeviceTensor<3> ye = Reshape(ye_mem.ReadWrite(), num_test_dof, test_vdim, num_entities);
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
{
// map_fields_to_quadrature_data(
// input_qp, e, this->fields_e,
// kinput_to_field, input_dtq_maps,
// integration_weights, geometric_factors, kernel.inputs,
// std::make_index_sequence<kernel.num_kinputs> {});
// map_fields_to_quadrature_data_conditional(
// directions_qp, e,
// directions_e, kinput_to_field,
// input_dtq_maps,
// integration_weights,
// geometric_factors,
// kinput_is_dependent,
// kernel.inputs,
// std::make_index_sequence<kernel.num_kinputs> {});
// for (int qp = 0; qp < num_qp; qp++)
// {
// auto f_qp = apply_kernel_fwddiff_enzyme(
// kernel.func,
// kernel_args,
// input_qp,
// kernel_shadow_args,
// directions_qp,
// qp);
// auto r_qp = Reshape(&da_qp(0, qp, e), da_size_on_qp);
// for (int i = 0; i < da_size_on_qp; i++)
// {
// r_qp(i) = f_qp(i);
// }
// }
// DeviceTensor<3> fhat = Reshape(&da_qp(0, 0, e), test_vdim, test_op_dim, num_qp);
// DeviceTensor<2> y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
// map_quadrature_data_to_fields(y, fhat,
// output_fop,
// output_dtq_maps[hardcoded_output_idx]);
}, num_entities, q1d, q1d, 1, shmem_info.total_size, shmem_cache.GetData());
R->MultTranspose(ye_mem, derivative_action_l);
};
if constexpr (std::is_same_v<decltype(output_fop), One>)
{
prolongation_transpose = [&](Vector &r_local, Vector &y)
{
double local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
op.mesh.GetComm());
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
};
}
else
{
auto P = get_prolongation(op.fields[test_space_field_idx]);
prolongation_transpose = [P](Vector &r_l, Vector &y)
{
P->MultTranspose(r_l, y);
};
}
+49
View File
@@ -0,0 +1,49 @@
* Calculate shared memory requirements
* Interpolation and integration
---
* If grad involved, need B and G
* Fit largest field, depends on polynomial order (#dofs)
-> vdim is irrelevant
* Temporaries for each sum
- DDQ (d1d x d1d x q1d) x 2 -> DDQ0, DDQ1
- DQQ (d1d x q1d x q1d) x 3 -> DQQ0, DQQ1, DQQ2
- QQQ (q1d x q1d x q1d) x 3 -> QQQ0, QQQ1, QQQ2
We need the following combinations at the same time
(1) DDQ0 + DDQ1 + DQQ0 + DQQ1 + DQQ2
(2) DQQ0 + DQQ1 + DQQ2 + QQQ0 + QQQ1 + QQQ2
(3) QQQ0 + QQQ1 + QQQ2 + QQD0 + QQD1 + QQD2
(4) QQD0 + QQD1 + QQD2 + QDD0 + QDD1 + QDD2
Allocate largest memory footprint from 2, 3 or 4 and
add memory footprint of fields and B/G.
Annotations with NR and R mean "not reusable" and
"reusable", respectively. This means the memory location is
reused for _all_ e.g. interpolation of a value etc.
----
For the action of nonlinear diffusion in 2D we have
(rho * |u|^2 \nabla u, \nabla v)
* Load
RHO (D x D) | R (after interpolation)
U (D x D x VDIM) | R (after interpolation)
B (Q x D) | NR
G (Q x D) | NR
* Interpolate Value
Temporary (Q x D) | R
R (Q x Q) | NR
U (Q x Q x VDIM) | NR
* Interpolate Grad
Temporaries (Q x D) + (Q x D) | R
U (Q x Q x DIM x VDIM) | NR
Quadrature point function
-> purely thread local
* Integrate Grad
R | temp from Interpolation
R | U from Load
+845
View File
@@ -0,0 +1,845 @@
// This is serac's tuple implementation
#pragma once
#include "general/backends.hpp"
#include <utility>
#include <mfem.hpp>
#include <tuple>
namespace mfem
{
/**
* @tparam T the types stored in the tuple
* @brief This is a class that mimics most of std::tuple's interface,
* except that it is usable in CUDA kernels and admits some arithmetic operator overloads.
*
* see https://en.cppreference.com/w/cpp/utility/tuple for more information about std::tuple
*/
template <typename... T>
struct tuple
{
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
*/
template <typename T0>
struct tuple<T0>
{
T0 v0; ///< The first member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
*/
template <typename T0, typename T1>
struct tuple<T0, T1>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
*/
template <typename T0, typename T1, typename T2>
struct tuple<T0, T1, T2>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
* @tparam T3 The fourth type stored in the tuple
*/
template <typename T0, typename T1, typename T2, typename T3>
struct tuple<T0, T1, T2, T3>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
* @tparam T3 The fourth type stored in the tuple
* @tparam T4 The fifth type stored in the tuple
*/
template <typename T0, typename T1, typename T2, typename T3, typename T4>
struct tuple<T0, T1, T2, T3, T4>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
T4 v4; ///< The fifth member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
* @tparam T3 The fourth type stored in the tuple
* @tparam T4 The fifth type stored in the tuple
* @tparam T5 The sixth type stored in the tuple
*/
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5>
struct tuple<T0, T1, T2, T3, T4, T5>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
T4 v4; ///< The fifth member of the tuple
T5 v5; ///< The sixth member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
* @tparam T3 The fourth type stored in the tuple
* @tparam T4 The fifth type stored in the tuple
* @tparam T5 The sixth type stored in the tuple
* @tparam T6 The seventh type stored in the tuple
*/
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6>
struct tuple<T0, T1, T2, T3, T4, T5, T6>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
T4 v4; ///< The fifth member of the tuple
T5 v5; ///< The sixth member of the tuple
T6 v6; ///< The seventh member of the tuple
};
/**
* @brief Type that mimics std::tuple
*
* @tparam T0 The first type stored in the tuple
* @tparam T1 The second type stored in the tuple
* @tparam T2 The third type stored in the tuple
* @tparam T3 The fourth type stored in the tuple
* @tparam T4 The fifth type stored in the tuple
* @tparam T5 The sixth type stored in the tuple
* @tparam T6 The seventh type stored in the tuple
* @tparam T7 The eighth type stored in the tuple
*/
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7>
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
T4 v4; ///< The fifth member of the tuple
T5 v5; ///< The sixth member of the tuple
T6 v6; ///< The seventh member of the tuple
T7 v7; ///< The eighth member of the tuple
};
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7, typename T8>
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7, T8>
{
T0 v0; ///< The first member of the tuple
T1 v1; ///< The second member of the tuple
T2 v2; ///< The third member of the tuple
T3 v3; ///< The fourth member of the tuple
T4 v4; ///< The fifth member of the tuple
T5 v5; ///< The sixth member of the tuple
T6 v6; ///< The seventh member of the tuple
T7 v7; ///< The eighth member of the tuple
T8 v8;
};
/**
* @brief Class template argument deduction rule for tuples
* @tparam T The variadic template parameter for tuple types
*/
template <typename... T>
MFEM_HOST_DEVICE
tuple(T...) -> tuple<T...>;
/**
* @brief helper function for combining a list of values into a tuple
* @tparam T types of the values to be tuple-d
* @param args the actual values to be put into a tuple
*/
template <typename... T>
MFEM_HOST_DEVICE tuple<T...> make_tuple(const T&... args)
{
return tuple<T...> {args...};
}
template <class... Types>
struct tuple_size
{
};
template <class... Types>
struct tuple_size<mfem::tuple<Types...>> :
std::integral_constant<std::size_t, sizeof...(Types)>
{
};
/**
* @tparam i the tuple index to access
* @tparam T the types stored in the tuple
* @brief return a reference to the ith tuple entry
*/
template <int i, typename... T>
MFEM_HOST_DEVICE constexpr auto& get(tuple<T...>& values)
{
static_assert(i < sizeof...(T), "");
if constexpr (i == 0)
{
return values.v0;
}
if constexpr (i == 1)
{
return values.v1;
}
if constexpr (i == 2)
{
return values.v2;
}
if constexpr (i == 3)
{
return values.v3;
}
if constexpr (i == 4)
{
return values.v4;
}
if constexpr (i == 5)
{
return values.v5;
}
if constexpr (i == 6)
{
return values.v6;
}
if constexpr (i == 7)
{
return values.v7;
}
if constexpr (i == 8)
{
return values.v8;
}
}
/**
* @tparam i the tuple index to access
* @tparam T the types stored in the tuple
* @brief return a copy of the ith tuple entry
*/
template <int i, typename... T>
MFEM_HOST_DEVICE constexpr const auto& get(const tuple<T...>& values)
{
static_assert(i < sizeof...(T), "");
if constexpr (i == 0)
{
return values.v0;
}
if constexpr (i == 1)
{
return values.v1;
}
if constexpr (i == 2)
{
return values.v2;
}
if constexpr (i == 3)
{
return values.v3;
}
if constexpr (i == 4)
{
return values.v4;
}
if constexpr (i == 5)
{
return values.v5;
}
if constexpr (i == 6)
{
return values.v6;
}
if constexpr (i == 7)
{
return values.v7;
}
if constexpr (i == 8)
{
return values.v8;
}
}
/**
* @brief a function intended to be used for extracting the ith type from a tuple.
*
* @note type<i>(my_tuple) returns a value, whereas get<i>(my_tuple) returns a reference
*
* @tparam i the index of the tuple to query
* @tparam T the types stored in the tuple
* @param values the tuple of values
* @return a copy of the ith entry of the input
*/
template <int i, typename... T>
MFEM_HOST_DEVICE constexpr auto type(const tuple<T...>& values)
{
static_assert(i < sizeof...(T), "");
if constexpr (i == 0)
{
return values.v0;
}
if constexpr (i == 1)
{
return values.v1;
}
if constexpr (i == 2)
{
return values.v2;
}
if constexpr (i == 3)
{
return values.v3;
}
if constexpr (i == 4)
{
return values.v4;
}
if constexpr (i == 5)
{
return values.v5;
}
if constexpr (i == 6)
{
return values.v6;
}
if constexpr (i == 7)
{
return values.v7;
}
if constexpr (i == 8)
{
return values.v8;
}
}
/**
* @brief A helper function for the + operator of tuples
*
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param y tuple of values
* @return the returned tuple sum
*/
template <typename... S, typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto plus_helper(const tuple<S...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) + get<i>(y)...};
}
/**
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @param x a tuple of values
* @param y a tuple of values
* @brief return a tuple of values defined by elementwise sum of x and y
*/
template <typename... S, typename... T>
MFEM_HOST_DEVICE constexpr auto operator+(const tuple<S...>& x,
const tuple<T...>& y)
{
static_assert(sizeof...(S) == sizeof...(T));
return plus_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
}
/**
* @brief A helper function for the += operator of tuples
*
* @tparam T the types stored in the tuples x and y
* @tparam i integer sequence used to index the tuples
* @param x tuple of values to be incremented
* @param y tuple of increment values
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr void plus_equals_helper(tuple<T...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
((get<i>(x) += get<i>(y)), ...);
}
/**
* @tparam T the types stored in the tuples x and y
* @param x a tuple of values
* @param y a tuple of values
* @brief add values contained in y, to the tuple x
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
const tuple<T...>& y)
{
return plus_equals_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @brief A helper function for the -= operator of tuples
*
* @tparam T the types stored in the tuples x and y
* @tparam i integer sequence used to index the tuples
* @param x tuple of values to be subracted from
* @param y tuple of values to subtract from x
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr void minus_equals_helper(tuple<T...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
((get<i>(x) -= get<i>(y)), ...);
}
/**
* @tparam T the types stored in the tuples x and y
* @param x a tuple of values
* @param y a tuple of values
* @brief add values contained in y, to the tuple x
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator-=(tuple<T...>& x,
const tuple<T...>& y)
{
return minus_equals_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @brief A helper function for the - operator of tuples
*
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param y tuple of values
* @return the returned tuple difference
*/
template <typename... S, typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto minus_helper(const tuple<S...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) - get<i>(y)...};
}
/**
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @param x a tuple of values
* @param y a tuple of values
* @brief return a tuple of values defined by elementwise difference of x and y
*/
template <typename... S, typename... T>
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<S...>& x,
const tuple<T...>& y)
{
static_assert(sizeof...(S) == sizeof...(T));
return minus_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
}
/**
* @brief A helper function for the - operator of tuples
*
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @return the returned tuple difference
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto unary_minus_helper(const tuple<T...>& x,
std::integer_sequence<int, i...>)
{
return tuple{-get<i>(x)...};
}
/**
* @tparam T the types stored in the tuple y
* @param x a tuple of values
* @brief return a tuple of values defined by applying the unary minus operator to each element of x
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<T...>& x)
{
return unary_minus_helper(x,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @brief A helper function for the / operator of tuples
*
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param y tuple of values
* @return the returned tuple ratio
*/
template <typename... S, typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<S...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) / get<i>(y)...};
}
/**
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @param x a tuple of values
* @param y a tuple of values
* @brief return a tuple of values defined by elementwise division of x by y
*/
template <typename... S, typename... T>
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<S...>& x,
const tuple<T...>& y)
{
static_assert(sizeof...(S) == sizeof...(T));
return div_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
}
/**
* @brief A helper function for the / operator of tuples
*
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a the constant numerator
* @return the returned tuple ratio
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto div_helper(const double a,
const tuple<T...>& x, std::integer_sequence<int, i...>)
{
return tuple{a / get<i>(x)...};
}
/**
* @brief A helper function for the / operator of tuples
*
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a the constant denomenator
* @return the returned tuple ratio
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<T...>& x,
const double a, std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) / a...};
}
/**
* @tparam T the types stored in the tuple x
* @param a the numerator
* @param x a tuple of denominator values
* @brief return a tuple of values defined by division of a by the elements of x
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator/(const double a, const tuple<T...>& x)
{
return div_helper(a, x,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @tparam T the types stored in the tuple y
* @param x a tuple of numerator values
* @param a a denominator
* @brief return a tuple of values defined by elementwise division of x by a
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<T...>& x, const double a)
{
return div_helper(x, a,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @brief A helper function for the * operator of tuples
*
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param y tuple of values
* @return the returned tuple product
*/
template <typename... S, typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<S...>& x,
const tuple<T...>& y,
std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) * get<i>(y)...};
}
/**
* @tparam S the types stored in the tuple x
* @tparam T the types stored in the tuple y
* @param x a tuple of values
* @param y a tuple of values
* @brief return a tuple of values defined by elementwise multiplication of x and y
*/
template <typename... S, typename... T>
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<S...>& x,
const tuple<T...>& y)
{
static_assert(sizeof...(S) == sizeof...(T));
return mult_helper(x, y,
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
}
/**
* @brief A helper function for the * operator of tuples
*
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a a constant multiplier
* @return the returned tuple product
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto mult_helper(const double a,
const tuple<T...>& x, std::integer_sequence<int, i...>)
{
return tuple{a * get<i>(x)...};
}
/**
* @brief A helper function for the * operator of tuples
*
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a a constant multiplier
* @return the returned tuple product
*/
template <typename... T, int... i>
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<T...>& x,
const double a, std::integer_sequence<int, i...>)
{
return tuple{get<i>(x) * a...};
}
/**
* @tparam T the types stored in the tuple
* @param a a scaling factor
* @param x the tuple object
* @brief multiply each component of x by the value a on the left
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator*(const double a, const tuple<T...>& x)
{
return mult_helper(a, x,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @tparam T the types stored in the tuple
* @param x the tuple object
* @param a a scaling factor
* @brief multiply each component of x by the value a on the right
*/
template <typename... T>
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const double a)
{
return mult_helper(x, a,
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @tparam T the types stored in the tuple
* @tparam i a list of indices used to acces each element of the tuple
* @param out the ostream to write the output to
* @param A the tuple of values
* @brief helper used to implement printing a tuple of values
*/
template <typename... T, std::size_t... i>
auto& print_helper(std::ostream& out, const mfem::tuple<T...>& A,
std::integer_sequence<size_t, i...>)
{
out << "tuple{";
(..., (out << (i == 0 ? "" : ", ") << mfem::get<i>(A)));
out << "}";
return out;
}
/**
* @tparam T the types stored in the tuple
* @param out the ostream to write the output to
* @param A the tuple of values
* @brief print a tuple of values
*/
template <typename... T>
auto& operator<<(std::ostream& out, const mfem::tuple<T...>& A)
{
return print_helper(out, A, std::make_integer_sequence<size_t, sizeof...(T)>());
}
/**
* @brief A helper to apply a lambda to a tuple
*
* @tparam lambda The functor type
* @tparam T The tuple types
* @tparam i The integer sequence to i
* @param f The functor to apply to the tuple
* @param args The input tuple
* @return The functor output
*/
template <typename lambda, typename... T, int... i>
MFEM_HOST_DEVICE auto apply_helper(lambda f, tuple<T...>& args,
std::integer_sequence<int, i...>)
{
return f(get<i>(args)...);
}
/**
* @tparam lambda a callable type
* @tparam T the types of arguments to be passed in to f
* @param f the callable object
* @param args a tuple of arguments
* @brief a way of passing an n-tuple to a function that expects n separate arguments
*
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
*/
template <typename lambda, typename... T>
MFEM_HOST_DEVICE auto apply(lambda f, tuple<T...>& args)
{
return apply_helper(f, std::move(args),
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @overload
*/
template <typename lambda, typename... T, int... i>
MFEM_HOST_DEVICE auto apply_helper(lambda f, const tuple<T...>& args,
std::integer_sequence<int, i...>)
{
return f(get<i>(args)...);
}
/**
* @tparam lambda a callable type
* @tparam T the types of arguments to be passed in to f
* @param f the callable object
* @param args a tuple of arguments
* @brief a way of passing an n-tuple to a function that expects n separate arguments
*
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
*/
template <typename lambda, typename... T>
MFEM_HOST_DEVICE auto apply(lambda f, const tuple<T...>& args)
{
return apply_helper(f, std::move(args),
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
}
/**
* @brief a struct used to determine the type at index I of a tuple
*
* @note see: https://en.cppreference.com/w/cpp/utility/tuple/tuple_element
*
* @tparam I the index of the desired type
* @tparam T a tuple of different types
*/
template <size_t I, class T>
struct tuple_element;
// recursive case
/// @overload
template <size_t I, class Head, class... Tail>
struct tuple_element<I, tuple<Head, Tail...>> : tuple_element<I - 1,
tuple<Tail...>>
{
};
// base case
/// @overload
template <class Head, class... Tail>
struct tuple_element<0, tuple<Head, Tail...>>
{
using type = Head; ///< the type at the specified index
};
/**
* @brief Trait for checking if a type is a @p mfem::tuple
*/
template <typename T>
struct is_tuple : std::false_type
{
};
/// @overload
template <typename... T>
struct is_tuple<mfem::tuple<T...>> : std::true_type
{
};
/**
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
*/
template <typename T>
struct is_tuple_of_tuples : std::false_type
{
};
/**
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
*/
template <typename... T>
struct is_tuple_of_tuples<mfem::tuple<T...>>
{
static constexpr bool value = (is_tuple<T>::value &&
...); ///< true/false result of type check
};
} // namespace mfem
+123
View File
@@ -0,0 +1,123 @@
#include "dfem/dfem_refactor.hpp"
#include "fem/bilininteg.hpp"
#include "fem/coefficient.hpp"
#include "linalg/auxiliary.hpp"
#include "linalg/hypre.hpp"
using namespace mfem;
using mfem::internal::tensor;
int main(int argc, char *argv[])
{
Mpi::Init();
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
const char *mesh_file = "../data/ref-square.mesh";
int polynomial_order = 1;
int ir_order = 2;
int refinements = 1;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
L2_FECollection fec(polynomial_order, dim, BasisType::GaussLobatto);
ParFiniteElementSpace fes(&mesh, &fec);
const IntegrationRule &ir = IntRules.Get(fes.GetFE(0)->GetGeomType(),
ir_order * fec.GetOrder());
const IntegrationRule &ir_face = IntRules.Get(
fes.GetTraceElement(0, fes.GetMesh()->GetFaceGeometry(0))->GetGeomType(),
ir_order * fec.GetOrder());
ParGridFunction u(&fes);
// // -\nabla \cdot (\nabla u + p * I) -> (\nabla u + p * I, \nabla v)
// auto advection_kernel = [](const tensor<double, 2> &dudxi,
// const tensor<double, 2, 2> &J,
// const double &w)
// {
// constexpr tensor<double, 2> b{1.0, 1.0};
// return std::tuple{dot(b, dudxi * inv(J)) * det(J) * w};
// };
// std::tuple argument_operators_0{Gradient{"quantity"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
// std::tuple output_operator_0{Value{"quantity"}};
// ElementOperator op_0{advection_kernel, argument_operators_0, output_operator_0};
// std::array solutions{FieldDescriptor{&fes, "quantity"}};
// std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
// DifferentiableOperator advection_op{solutions, parameters, std::tuple{op_0}, mesh, ir};
// auto adv_du = advection_op.template GetDerivativeWrt<0>({&u}, {mesh_nodes});
// HypreParMatrix A;
// adv_du->Assemble(A);
// std::ofstream mmatofs("dfem_mat.dat");
// A.PrintMatlab(mmatofs);
// mmatofs.close();
auto trace_kernel = [](const double &uL, const double &uR, const double &J,
const double &w)
{
return std::tuple{1.0 / J * w};
};
std::tuple argument_operators_0
{
FaceValueLeft{"quantity"},
FaceValueRight{"quantity"},
Gradient{"coordinates"},
Weight{"integration_weights"}
};
std::tuple output_operator_0{Value{"quantity"}};
FaceElementOperator op_0{trace_kernel, argument_operators_0, output_operator_0};
std::array solutions{FieldDescriptor{&fes, "quantity"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator trace_op{solutions, parameters, std::tuple{op_0}, mesh, ir_face};
auto vector_func = [](const Vector &, Vector &u)
{
u = 1.0;
};
VectorFunctionCoefficient vel_coeff(dim, vector_func);
ParBilinearForm adv_form(&fes);
constexpr double alpha = 1.0;
auto integ = new ConvectionIntegrator(vel_coeff, alpha);
integ->SetIntRule(&ir);
adv_form.AddInteriorFaceIntegrator(
new NonconservativeDGTraceIntegrator(vel_coeff, alpha));
// adv_form.AddDomainIntegrator(integ);
adv_form.Assemble();
adv_form.Finalize();
auto K = adv_form.ParallelAssemble();
std::ofstream kmatofs("mfem_mat.dat");
K->PrintMatlab(kmatofs);
kmatofs.close();
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << u << std::flush;
return 0;
}
+150
View File
@@ -0,0 +1,150 @@
#include "dfem.hpp"
int main(int argc, char *argv[])
{
Mpi::Init();
std::cout << std::setprecision(9);
const char *mesh_file = "../data/star.mesh";
int polynomial_order = 1;
int refinements = 0;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&polynomial_order, "-o", "--order", "");
args.AddOption(&refinements, "-r", "--r", "");
args.ParseCheck();
Mesh mesh_serial(mesh_file, 1, 1);
mesh_serial.SetCurvature(1);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
const int dim = mesh_serial.Dimension();
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh_serial.Clear();
constexpr int vdim = 2;
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_tdof_list;
Array<int> ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
ParGridFunction u(&h1fes);
auto exact_solution = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = x*x + y;
u(1) = x + 0.5*y*y;
};
VectorFunctionCoefficient exact_solution_coeff(dim, exact_solution);
auto elasticity_kernel = [](tensor<double, 2, 2> &dudxi,
tensor<double, 2, 2> &J,
double &w)
{
using mfem::internal::tensor;
using mfem::internal::IsotropicIdentity;
double lambda, mu;
{
lambda = 1.0;
mu = 1.0;
}
static constexpr auto I = IsotropicIdentity<2>();
auto eps = sym(dudxi * inv(J));
auto JxW = transpose(inv(J)) * det(J) * w;
auto r = (lambda * tr(eps) * I + 2.0 * mu * eps) * JxW;
return r;
};
tensor<double, 2, 2> dudxi, s_dudxi, J;
double w = 1.0;
enzyme::get<0>
(enzyme::autodiff<enzyme::Forward,
enzyme::DuplicatedNoNeed<tensor<double, 2, 2>>>
(+elasticity_kernel,
enzyme::Duplicated<tensor<double, 2, 2> *>(&dudxi, &s_dudxi),
enzyme::Const<tensor<double, 2, 2>*>(&J),
enzyme::Const<double*>(&w)));
// std::tuple input_descriptors = {Gradient{"displacement"}, Gradient{"coordinates"}, Weight{"integration_weight"}};
// std::tuple output_descriptors = {Gradient{"displacement"}};
// ElementOperator qf {elasticity_kernel, input_descriptors, output_descriptors};
// ElementOperator forcing_qf
// {
// [](tensor<double, 2> x, tensor<double, 2, 2> J, double w)
// {
// double lambda, mu;
// {
// lambda = 1.0;
// mu = 1.0;
// }
// auto f = x;
// f(0) = 4.0*mu + 2.0*lambda;
// f(1) = 2.0*mu + lambda;
// return f * det(J) * w;
// },
// // inputs
// std::tuple{
// Value{"coordinates"},
// Gradient{"coordinates"},
// Weight{"integration_weight"}},
// // outputs
// std::tuple{
// Value{"displacement"}}
// };
// std::vector<Field> solutions{{&u, "displacement"}};
// std::vector<Field> parameters{{mesh.GetNodes(), "coordinates"}};
// std::vector<Field> dependent_fields{{&u, "displacement"}};
// DifferentiableForm dop(solutions, parameters, dependent_fields, mesh);
// dop.AddElementOperator<AD::Enzyme>(qf, ir);
// dop.AddElementOperator<AD::None>(forcing_qf, ir);
// dop.SetEssentialTrueDofs(ess_tdof_list);
// GMRESSolver gmres(MPI_COMM_WORLD);
// gmres.SetRelTol(1e-12);
// gmres.SetMaxIter(5000);
// gmres.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
// NewtonSolver newton(MPI_COMM_WORLD);
// newton.SetSolver(gmres);
// newton.SetOperator(dop);
// newton.SetRelTol(1e-12);
// newton.SetMaxIter(100);
// newton.SetPrintLevel(1);
// u = 1e-6;
// u.ProjectBdrCoefficient(exact_solution_coeff, ess_bdr);
// Vector x;
// u.GetTrueDofs(x);
// Vector zero;
// newton.Mult(zero, x);
// u.Distribute(x);
// std::cout << "|u-u_ex|_L2 = " << u.ComputeL2Error(exact_solution_coeff) << "\n";
return 0;
}
File diff suppressed because it is too large Load Diff
+115
View File
@@ -0,0 +1,115 @@
#include <tuple>
#include <type_traits>
#include <iostream>
#include <enzyme/enzyme>
template <typename T>
constexpr auto get_type_name() -> std::string_view
{
#if defined(__clang__)
constexpr auto prefix = std::string_view {"[T = "};
constexpr auto suffix = "]";
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
#elif defined(__GNUC__)
constexpr auto prefix = std::string_view {"with T = "};
constexpr auto suffix = "; ";
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
#elif defined(_MSC_VER)
constexpr auto prefix = std::string_view {"get_type_name<"};
constexpr auto suffix = ">(void)";
constexpr auto function = std::string_view{__FUNCSIG__};
#else
#error Unsupported compiler
#endif
const auto start = function.find(prefix) + prefix.size();
const auto end = function.find(suffix);
const auto size = end - start;
return function.substr(start, size);
}
template <typename ... Ts>
constexpr auto decay_types(std::tuple<Ts...> const &)
-> std::tuple<std::remove_cv_t<std::remove_reference_t<Ts>>...>;
template <typename T>
using decay_tuple = decltype(decay_types(std::declval<T>()));
template <class F> struct FunctionSignature;
template <typename output_t, typename... input_ts>
struct FunctionSignature<output_t(input_ts...)>
{
using return_t = output_t;
using parameter_ts = std::tuple<input_ts...>;
};
template <class T> struct create_function_signature;
template <typename output_t, typename T, typename... input_ts>
struct create_function_signature<output_t (T::*)(input_ts...) const>
{
using type = FunctionSignature<output_t(input_ts...)>;
};
template <typename arg_ts, std::size_t... Is>
auto create_enzyme_args(arg_ts &args,
arg_ts &shadow_args,
std::index_sequence<Is...>)
{
((std::cout << std::get<Is>(shadow_args) << "\n"), ...);
return std::tuple<enzyme::Duplicated<decltype(std::get<Is>(args))>...>
{
{ std::get<Is>(args), std::get<Is>(shadow_args) }...
};
}
template <typename kernel_t, typename arg_ts>
auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args, arg_ts &&shadow_args)
{
auto arg_indices =
std::make_index_sequence<std::tuple_size_v<std::remove_reference_t<arg_ts>>> {};
auto enzyme_args = create_enzyme_args(args, shadow_args, arg_indices);
using kf_return_t = typename create_function_signature<
decltype(&kernel_t::operator())>::type::return_t;
std::cout << "args is " << get_type_name<decltype(args)>() << "\n\n";
std::cout << "enzyme_args type is " << get_type_name<decltype(enzyme_args)>() <<
"\n\n";
std::cout << "return type is " << get_type_name<decltype(kf_return_t{})>() <<
"\n\n";
return std::apply([&](auto &&...args)
{
return enzyme::get<0>(
enzyme::autodiff<enzyme::Forward>
(+kernel, args...));
},
enzyme_args);
}
int main()
{
auto func = [](const double &x)
{
return x*x;
};
using kf_param_ts = typename create_function_signature<
decltype(&decltype(func)::operator())>::type::parameter_ts;
using kf_output_t = typename create_function_signature<
decltype(&decltype(func)::operator())>::type::return_t;
auto kernel_args = decay_tuple<kf_param_ts> {};
auto kernel_shadow_args = decay_tuple<kf_param_ts> {};
std::get<0>(kernel_args) = 3;
std::get<0>(kernel_shadow_args) = 1;
const auto res = fwddiff_apply_enzyme(func, kernel_args, kernel_shadow_args);
std::cout << res << " == 6\n";
return 0;
}
+114
View File
@@ -0,0 +1,114 @@
#include "dfem.hpp"
int main(int argc, char *argv[])
{
Mpi::Init();
std::cout << std::setprecision(9);
const char *mesh_file = "../data/star.mesh";
int polynomial_order = 1;
int refinements = 0;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&polynomial_order, "-o", "--order", "");
args.AddOption(&refinements, "-r", "--r", "");
args.ParseCheck();
Mesh mesh_serial(mesh_file, 1, 1);
mesh_serial.SetCurvature(1);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
const int dim = mesh_serial.Dimension();
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh_serial.Clear();
constexpr int vdim = 2;
// test_partial_assembly_setup_qf(mesh, 1, polynomial_order);
// exit(0);
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_tdof_list;
Array<int> ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
ParGridFunction u(&h1fes);
ParGridFunction g(&h1fes);
ParGridFunction rho(&h1fes);
auto exact_solution = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = x*x + y;
u(1) = x + 0.5*y*y;
};
VectorFunctionCoefficient exact_solution_coeff(dim, exact_solution);
auto objective = [](tensor<double, 2> u, double rho,
tensor<double, 2, 2> J,
double w)
{
return sqnorm(u) * det(J) * w;
};
std::tuple inputs{Value{"displacement"}, Value{"density"}, Gradient{"coordinates"}, Weight{"integration_weight"}};
std::tuple outputs{ One{"integral"} };
ElementOperator objective_eop { objective, inputs, outputs };
std::vector<Field> solution_fields{{&u, "displacement"}};
std::vector<Field> parameter_fields{{mesh.GetNodes(), "coordinates"}, {&rho, "density"}};
std::vector<Field> dependent_variables{{&u, "displacement"}};
DifferentiableForm dop(solution_fields, parameter_fields, dependent_variables,
mesh);
dop.AddElementOperator(objective_eop, ir);
u.ProjectCoefficient(exact_solution_coeff);
Vector zero;
Vector y(1);
Vector utdof;
u.GetTrueDofs(utdof);
dop.Mult(utdof, y);
// finite difference test
Vector dgdu(u.Size());
Vector fx(y);
out << "g: ";
print_vector(fx);
out << "\n";
for (int i = 0; i < u.Size(); i++)
{
double h = 1e-6;
u(i) += h;
dop.Mult(u, y);
u(i) -= h;
y -= fx;
y /= h;
dgdu(i) = y(0);
}
out << "dgdu: ";
print_vector(dgdu);
// Vector dgdu = dop.GetGradientWrt({&u, "displacement"});
return 0;
}
+138
View File
@@ -0,0 +1,138 @@
#include "dfem.hpp"
int main(int argc, char *argv[])
{
Mpi::Init();
std::cout << std::setprecision(9);
const char *mesh_file = "../data/star.mesh";
int polynomial_order = 1;
int refinements = 0;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&polynomial_order, "-o", "--order", "");
args.AddOption(&refinements, "-r", "--r", "");
args.ParseCheck();
Mesh mesh_serial(mesh_file, 1, 1);
mesh_serial.SetCurvature(1);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
const int dim = mesh_serial.Dimension();
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh_serial.Clear();
constexpr int vdim = 1;
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_tdof_list;
Array<int> ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
ParGridFunction u(&h1fes);
auto exact_solution = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
// PRESENT
return pow(x,2) + 0.5*x*pow(y,2);
};
FunctionCoefficient exact_solution_coeff(exact_solution);
auto plaplacian = [](double u,
tensor<double, 2> dudxi,
tensor<double, 2, 2> J,
double w)
{
using mfem::internal::tensor;
auto dudx = dudxi * inv(J);
auto JxW = transpose(inv(J)) * det(J) * w;
// PRESENT: Implement (1+u^2) * ∇u
return (1.0 + u*u) * dudx * JxW;
};
// PRESENT: Implement descriptors
std::tuple input_descriptors = {Value{"potential"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
// PRESENT: Implement descriptors
std::tuple output_descriptors = {Gradient{"potential"}};
ElementOperator qf {plaplacian, input_descriptors, output_descriptors};
ElementOperator forcing_qf
{
[](tensor<double, 2> coords, tensor<double, 2, 2> J, double w)
{
int p = 2;
double x = coords(0);
double y = coords(1);
// *INDENT-OFF*
double mathematica_please_help_me = 2.*pow(x,2)*pow(y,2)*(pow(x,2) + 0.5*x*pow(y,2)) + 2*pow(2*x + 0.5*pow(y,2),2)*(pow(x,2) + 0.5*x*pow(y,2)) + 2*(1 + pow(pow(x,2) + 0.5*x*pow(y,2),2)) + 1.*x*(1 + pow(pow(x,2) + 0.5*x*pow(y,2),2));
return mathematica_please_help_me * det(J) * w;
// *INDENT-ON*
},
// inputs
std::tuple{
Value{"coordinates"},
Gradient{"coordinates"},
Weight{"integration_weight"}},
// outputs
std::tuple{
Value{"potential"}}
};
std::tuple list_of_qfs{qf_1, qf_2, qf_n};
std::vector<Field> solutions{{&u, "potential"}};
std::vector<Field> parameters{{mesh.GetNodes(), "coordinates"}};
DifferentiableForm dop(solutions, parameters, mesh);
dop.SetEssentialTrueDofs(ess_tdof_list);
auto R = dop.GetResidual(list_of_qfs, ir);
auto Jacobian_aka_dRdu = dop.GetDerivative<0>(list_of_qfs, ir);
// R(u) = (\grad u, \grad v) + (f, v)
// dop.AddElementOperator<AD::Enzyme>(qf, ir);
// dop.AddElementOperator<AD::None>(forcing_qf, ir);
GMRESSolver gmres(MPI_COMM_WORLD);
gmres.SetRelTol(1e-12);
gmres.SetMaxIter(5000);
gmres.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
NewtonSolver newton(MPI_COMM_WORLD);
newton.SetSolver(gmres);
newton.SetOperator(dop);
newton.SetRelTol(1e-12);
newton.SetMaxIter(100);
newton.SetPrintLevel(1);
u = 1e-6;
u.ProjectBdrCoefficient(exact_solution_coeff, ess_bdr);
Vector x;
u.GetTrueDofs(x);
Vector zero;
newton.Mult(zero, x);
u.Distribute(x);
std::cout << "|u-u_ex|_L2 = " << u.ComputeL2Error(exact_solution_coeff) << "\n";
return 0;
}
+192
View File
@@ -0,0 +1,192 @@
#include "dfem/dfem_refactor.hpp"
#include "linalg/hypre.hpp"
using namespace mfem;
using mfem::internal::tensor;
template <typename diffusion_t, typename force_t>
class DiffusionOperator : public Operator
{
template <typename diffusion_du_t>
class DiffusionJacobianOperator : public Operator
{
public:
DiffusionJacobianOperator(const DiffusionOperator *diffusion,
std::shared_ptr<diffusion_du_t> diff_du) :
Operator(diffusion->Height()), s(diffusion)
{
diff_du->Assemble(A);
A.EliminateBC(s->ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
}
void Mult(const Vector &x, Vector &y) const override
{
A.Mult(x, y);
}
const DiffusionOperator *s;
HypreParMatrix A;
};
public:
DiffusionOperator(diffusion_t &diffusion, force_t &force,
Array<int> &ess_tdofs) :
Operator(diffusion.Height()), diffusion(diffusion),
force(force), ess_tdofs(ess_tdofs), f(force.Height()) {}
void SetParameters(ParGridFunction &mesh_nodes)
{
diffusion.SetParameters({&mesh_nodes});
force.SetParameters({&mesh_nodes});
Vector zero;
this->mesh_nodes.SetSpace(mesh_nodes.ParFESpace());
this->mesh_nodes = mesh_nodes;
}
void Mult(const Vector &x, Vector &r) const override
{
diffusion.Mult(x, r);
force.Mult(x, f);
r -= f;
r.SetSubVector(ess_tdofs, 0.0);
}
Operator &GetGradient(const Vector &x) const override
{
ParGridFunction u(const_cast<ParFiniteElementSpace *>
(*std::get_if<const ParFiniteElementSpace *>
(&diffusion.solutions[0].data)));
u.SetFromTrueDofs(x);
auto dfdu = diffusion.template GetDerivativeWrt<0>({&u}, {&mesh_nodes});
dfdu->Assemble(A);
A.EliminateBC(ess_tdofs, DiagonalPolicy::DIAG_ONE);
return A;
// delete jacobian_operator;
// jacobian_operator = new
// DiffusionJacobianOperator<typename std::remove_pointer<decltype(dfdu.get())>::type>
// (this, dfdu);
// return *jacobian_operator;
}
diffusion_t &diffusion;
force_t &force;
const Array<int> ess_tdofs;
mutable Vector f;
mutable ParGridFunction mesh_nodes;
mutable Operator *jacobian_operator = nullptr;
mutable HypreParMatrix A;
};
int main(int argc, char *argv[])
{
Mpi::Init();
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
const char *mesh_file = "../data/ref-square.mesh";
int polynomial_order = 2;
int ir_order = 2;
int refinements = 4;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection potential_fec(polynomial_order, dim);
ParFiniteElementSpace potential_fes(&mesh, &potential_fec);
const IntegrationRule &potential_ir =
IntRules.Get(potential_fes.GetFE(0)->GetGeomType(),
ir_order * potential_fec.GetOrder());
Array<int> bdr_attr_is_ess(mesh.bdr_attributes.Max());
bdr_attr_is_ess = 1;
Array<int> ess_tdofs;
potential_fes.GetEssentialTrueDofs(bdr_attr_is_ess, ess_tdofs);
ParGridFunction u(&potential_fes);
u = 0.0;
auto diffusion_kernel = [](const internal::dual<double, double> &u,
const tensor<internal::dual<double, double>, 2> &dudxi,
const tensor<double, 2, 2> &J,
const double &w)
{
auto invJ = inv(J);
auto dudx = dudxi * invJ;
return std::tuple{(1.0 + u * u) * dudx * det(J) * w * transpose(invJ)};
};
std::tuple argument_operators_0{Value{"potential"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
std::tuple output_operator_0{Gradient{"potential"}};
ElementOperator op_0{diffusion_kernel, argument_operators_0, output_operator_0};
auto force_kernel = [](const tensor<double, 2, 2> &J,
const double &w)
{
return std::tuple{1.0 * det(J) * w};
};
std::tuple argument_operators_1{Gradient{"coordinates"}, Weight{"integration_weights"}};
std::tuple output_operator_1{Value{"potential"}};
ElementOperator op_1{force_kernel, argument_operators_1, output_operator_1};
std::array solutions{FieldDescriptor{&potential_fes, "potential"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator diffusion_op{solutions, parameters, std::tuple{op_0}, mesh, potential_ir};
DifferentiableOperator force_op{solutions, parameters, std::tuple{op_1}, mesh, potential_ir};
DiffusionOperator diffusion(diffusion_op, force_op, ess_tdofs);
diffusion.SetParameters({*mesh_nodes});
HypreBoomerAMG amg;
amg.SetPrintLevel(0);
CGSolver solver(MPI_COMM_WORLD);
solver.SetAbsTol(1e-12);
solver.SetRelTol(1e-12);
solver.SetMaxIter(500);
solver.SetPrintLevel(2);
solver.SetPreconditioner(amg);
NewtonSolver newton(MPI_COMM_WORLD);
newton.SetOperator(diffusion);
newton.SetSolver(solver);
newton.SetRelTol(1e-8);
newton.SetMaxIter(10);
newton.SetPrintLevel(1);
Vector zero;
Vector x(potential_fes.GetTrueVSize());
u.ParallelProject(x);
newton.Mult(zero, x);
u.SetFromTrueDofs(x);
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << u << std::flush;
return 0;
}
+102
View File
@@ -0,0 +1,102 @@
#include "mfem.hpp"
#include "dfem/dfem_refactor.hpp"
using namespace mfem;
auto main(int argc, char *argv[]) -> int
{
Mpi::Init();
std::cout << std::setprecision(9);
const char *mesh_file = "../data/star.mesh";
int polynomial_order = 1;
int refinements = 0;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&polynomial_order, "-o", "--order", "");
args.AddOption(&refinements, "-r", "--r", "");
args.ParseCheck();
Mesh mesh_serial(mesh_file, 1, 1);
mesh_serial.SetCurvature(1);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
const int dim = mesh_serial.Dimension();
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
constexpr int vdim = 1;
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_tdof_list;
Array<int> ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
ParGridFunction u(&h1fes);
auto exact_solution = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
return 2.345 + x + y;
};
FunctionCoefficient exact_solution_coeff(exact_solution);
u.ProjectCoefficient(exact_solution_coeff);
auto domain_qf = [](const double &u,
const tensor<double, 2, 2> &J,
const double &w)
{
out << u << "\n" << J << "\n" << w << "\n\n";
return std::tuple{u * det(J) * w};
};
std::tuple input_descriptors = {Value{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
std::tuple output_descriptors = {Value{"potential"}};
ElementOperator eop{domain_qf, input_descriptors, output_descriptors};
auto ops = std::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop{solutions, parameters, ops, mesh, ir};
Vector x(h1fes.GetTrueVSize()), y(h1fes.GetTrueVSize());
u.GetTrueDofs(x);
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
// Derivative wrt "potential", indicated by the index 0 of the set {solutions} \cup {parameters}
auto dFd0 = dop.GetDerivativeWrt<0>({&u}, {mesh_nodes});
dFd0->Mult(x, y);
Vector dFd0_vec;
dFd0->Assemble(dFd0_vec);
// Derivative wrt "coordinates", indicated by the index 1 of the set {solutions} \cup {parameters}
auto dFd1 = dop.GetDerivativeWrt<1>({&u}, {mesh_nodes});
dFd1->Mult(x, y);
return 0;
}
+302
View File
@@ -0,0 +1,302 @@
#include "dfem/dfem.hpp"
using namespace mfem;
using mfem::internal::tensor;
template <typename momentum_t, typename mass_conservation_t>
class NavierStokesOperator : public Operator
{
template <typename momentum_du_t, typename momentum_dp_t>
class NavierStokesJacobianOperator : public Operator
{
public:
NavierStokesJacobianOperator(const NavierStokesOperator *ns,
std::shared_ptr<momentum_du_t> mom_du,
std::shared_ptr<momentum_dp_t> mom_dp) :
Operator(ns->Height()), ns(ns), block_op(ns->block_offsets)
{
mom_du->Assemble(A);
A.EliminateBC(ns->vel_ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
mom_dp->Assemble(D);
D.EliminateRows(ns->vel_ess_tdofs);
Dt = new TransposeOperator(D);
block_op.SetBlock(0, 0, &A);
block_op.SetBlock(0, 1, &D);
block_op.SetBlock(1, 0, Dt);
// std::ofstream amatofs("dfem_mat.dat");
// block_op.PrintMatlab(amatofs);
// amatofs.close();
}
void Mult(const Vector &x, Vector &y) const override
{
block_op.Mult(x, y);
}
~NavierStokesJacobianOperator()
{
delete Dt;
}
const NavierStokesOperator *ns = nullptr;
HypreParMatrix A, D;
TransposeOperator *Dt = nullptr;
BlockOperator block_op;
};
public:
NavierStokesOperator(momentum_t &momentum,
mass_conservation_t &mass_conservation,
Array<int> &offsets, Array<int> &vel_ess_tdofs) :
Operator(offsets.Last()), momentum(momentum),
mass_conservation(mass_conservation),
block_offsets(offsets), vel_ess_tdofs(vel_ess_tdofs) {}
void SetParameters(ParGridFunction &mesh_nodes)
{
momentum.SetParameters({&mesh_nodes});
mass_conservation.SetParameters({&mesh_nodes});
this->mesh_nodes.SetSpace(mesh_nodes.ParFESpace());
this->mesh_nodes = mesh_nodes;
}
void Mult(const Vector &x, Vector &r) const override
{
Vector ru(r.ReadWrite() + block_offsets[0],
block_offsets[1] - block_offsets[0]);
Vector rp(r.ReadWrite() + block_offsets[1],
block_offsets[2] - block_offsets[1]);
momentum.Mult(x, ru);
mass_conservation.Mult(x, rp);
ru.SetSubVector(vel_ess_tdofs, 0.0);
}
Operator &GetGradient(const Vector &x) const override
{
xtmp = x;
BlockVector xb(xtmp.ReadWrite(), block_offsets);
ParGridFunction u(const_cast<ParFiniteElementSpace *>
(*std::get_if<const ParFiniteElementSpace *>
(&momentum.solutions[0].data)));
ParGridFunction p(const_cast<ParFiniteElementSpace *>
(*std::get_if<const ParFiniteElementSpace *>
(&momentum.solutions[1].data)));
u.SetFromTrueDofs(xb.GetBlock(0));
p.SetFromTrueDofs(xb.GetBlock(1));
auto mom_du = momentum.template GetDerivativeWrt<0>({&u, &p}, {&mesh_nodes});
auto mom_dp = momentum.template GetDerivativeWrt<1>({&u, &p}, {&mesh_nodes});
delete jacobian_operator;
jacobian_operator = new NavierStokesJacobianOperator<
typename std::remove_pointer<decltype(mom_du.get())>::type,
typename std::remove_pointer<decltype(mom_dp.get())>::type>(this, mom_du,
mom_dp);
return *jacobian_operator;
}
momentum_t &momentum;
mass_conservation_t &mass_conservation;
const Array<int> block_offsets;
const Array<int> vel_ess_tdofs;
mutable Vector xtmp;
mutable ParGridFunction mesh_nodes;
mutable Operator *jacobian_operator = nullptr;
};
double reynolds = 10.0;
int main(int argc, char *argv[])
{
constexpr int dim = 3;
constexpr int vdim = dim;
Mpi::Init();
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
const char *mesh_file = "../data/ref-cube.mesh";
int polynomial_order = 2;
int ir_order = 2;
int refinements = 2;
OptionsParser args(argc, argv);
args.AddOption(&refinements, "-r", "--refinements", "");
args.AddOption(&reynolds, "-rey", "--reynolds", "");
args.ParseCheck();
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection velocity_fec(polynomial_order, dim);
ParFiniteElementSpace velocity_fes(&mesh, &velocity_fec, dim);
H1_FECollection pressure_fec(polynomial_order - 1, dim);
ParFiniteElementSpace pressure_fes(&mesh, &pressure_fec);
const IntegrationRule &velocity_ir =
IntRules.Get(velocity_fes.GetFE(0)->GetGeomType(),
ir_order * velocity_fec.GetOrder());
const IntegrationRule &pressure_ir =
IntRules.Get(pressure_fes.GetFE(0)->GetGeomType(),
ir_order * pressure_fec.GetOrder());
Array<int> bdr_attr_is_ess(mesh.bdr_attributes.Max());
bdr_attr_is_ess = 1;
Array<int> vel_ess_tdofs;
velocity_fes.GetEssentialTrueDofs(bdr_attr_is_ess, vel_ess_tdofs);
ParGridFunction u(&velocity_fes);
ParGridFunction p(&pressure_fes);
auto u_f = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double z = coords(2);
if (z >= 1.0)
{
u(0) = 1.0;
}
else
{
u(0) = 0.0;
}
u(1) = 0.0;
u(2) = 0.0;
};
auto u_coef = VectorFunctionCoefficient(dim, u_f);
u.ProjectCoefficient(u_coef);
p = 0.0;
// -\nabla \cdot (\nabla u + p * I) -> (\nabla u + p * I, \nabla v)
auto momentum_kernel = [](const tensor<double, dim> &u,
const tensor<double, dim, dim> &dudxi,
const double &p,
const tensor<double, dim, dim> &J,
const double &w)
{
static constexpr auto I = mfem::internal::IsotropicIdentity<dim>();
auto invJ = inv(J);
auto dudx = dudxi * invJ;
double Re = reynolds;
return mfem::tuple{(outer(u, u) - 1.0 / Re * dudx + p * I) * det(J) * w * transpose(invJ)};
};
mfem::tuple argument_operators_0{Value{"velocity"}, Gradient{"velocity"}, Value{"pressure"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator_0{Gradient{"velocity"}};
ElementOperator op_0{momentum_kernel, argument_operators_0, output_operator_0};
// (\nabla \cdot u, q)
auto mass_conservation_kernel = [](const tensor<double, dim, dim> &dudxi,
const tensor<double, dim, dim> &J,
const double &w)
{
return mfem::tuple{tr(dudxi * inv(J)) * det(J) * w};
};
mfem::tuple argument_operators_1{Gradient{"velocity"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator_1{Value{"pressure"}};
ElementOperator op_1{mass_conservation_kernel, argument_operators_1, output_operator_1};
std::array solutions{FieldDescriptor{&velocity_fes, "velocity"}, FieldDescriptor{&pressure_fes, "pressure"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator momentum_op{solutions, parameters, mfem::tuple{op_0}, mesh, velocity_ir};
DifferentiableOperator mass_conservation_op{solutions, parameters, mfem::tuple{op_1}, mesh, pressure_ir};
// Preconditioner form
auto pressure_mass_kernel = [](const double &p,
const tensor<double, dim, dim> &J,
const double &w)
{
return mfem::tuple{p * det(J) * w};
};
mfem::tuple pms_args{Value{"pressure"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple pms_outs{Value{"pressure"}};
ElementOperator pressure_mass{pressure_mass_kernel, pms_args, pms_outs};
std::array pms_sols{FieldDescriptor{&pressure_fes, "pressure"}};
std::array pms_params{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator pressure_mass_op{pms_sols, pms_params, mfem::tuple{pressure_mass}, mesh, pressure_ir};
Array<int> block_offsets(3);
block_offsets[0] = 0;
block_offsets[1] = velocity_fes.GetTrueVSize();
block_offsets[2] = pressure_fes.GetTrueVSize();
block_offsets.PartialSum();
NavierStokesOperator navierstokes(momentum_op, mass_conservation_op,
block_offsets,
vel_ess_tdofs);
BlockVector x(block_offsets), y(block_offsets);
u.ParallelProject(x.GetBlock(0));
// p.ParallelProject(x.GetBlock(1));
navierstokes.SetParameters(*mesh_nodes);
HypreParMatrix A;
momentum_op.template GetDerivativeWrt<0>({&u, &p}, {mesh_nodes})->Assemble(A);
A.EliminateBC(vel_ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
HypreBoomerAMG amg(A);
amg.SetMaxLevels(50);
amg.SetPrintLevel(0);
HypreParMatrix Mp;
pressure_mass_op.template GetDerivativeWrt<0>({&p}, {mesh_nodes})->Assemble(Mp);
HypreDiagScale Mp_inv(Mp);
BlockDiagonalPreconditioner prec(block_offsets);
prec.SetDiagonalBlock(0, &amg);
prec.SetDiagonalBlock(1, &Mp_inv);
GMRESSolver solver(MPI_COMM_WORLD);
solver.SetAbsTol(0.0);
solver.SetRelTol(1e-8);
solver.SetKDim(100);
solver.SetMaxIter(500);
solver.SetPrintLevel(2);
solver.SetPreconditioner(prec);
NewtonSolver newton(MPI_COMM_WORLD);
newton.SetOperator(navierstokes);
newton.SetSolver(solver);
newton.SetRelTol(1e-8);
newton.SetMaxIter(50);
newton.SetPrintLevel(1);
Vector zero;
newton.Mult(zero, x);
u.SetFromTrueDofs(x.GetBlock(0));
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << u << std::flush;
return 0;
}
+174
View File
@@ -0,0 +1,174 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_diffusion(
std::string mesh_file, int refinements, int polynomial_order)
{
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == 2, "incorrect mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
mesh_serial.Clear();
out << "#el: " << mesh.GetNE() << "\n";
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
out << "#qp: " << ir.GetNPoints() << "\n";
ParGridFunction f1_g(&h1fes);
ParGridFunction rho_g(&h1fes);
auto kernel = [] MFEM_HOST_DEVICE(const tensor<double, 2, 2>& J,
const double& w, const tensor<double, 2>& dudxi)
{
auto invJ = inv(J);
return mfem::tuple{dudxi * invJ * transpose(invJ) * det(J) * w};
};
mfem::tuple argument_operators =
{
Gradient{"coordinates"}, Weight{}, Gradient{"potential"}
};
mfem::tuple output_operator = {Gradient{"potential"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector& coords)
{
const double x = coords(0);
const double y = coords(1);
return 2.345 + 0.25 * x * x * y + y * y * x;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
y.HostRead();
ParBilinearForm a(&h1fes);
a.AddDomainIntegrator(new DiffusionIntegrator);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble();
a.Finalize();
Vector y2(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(y2);
print_vector(y);
return 1;
}
// // Test linearization here as well
// auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
// if (dFdu->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
// return 1;
// }
// dFdu->Mult(x, y);
// y.HostRead();
// a.Mult(x, y2);
// y2.HostRead();
// diff = y2;
// diff -= y;
// if (diff.Norml2() > 1e-10)
// {
// print_vector(diff);
// print_vector(y2);
// print_vector(y);
// return 1;
// }
// // fd jacobian test
// {
// double eps = 1.0e-6;
// Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
// v *= eps;
// xpv += v;
// xmv -= v;
// dop.Mult(xpv, fxpv);
// dop.Mult(xmv, fxmv);
// fxpv -= fxmv;
// fxpv /= (2.0*eps);
// fxpv -= y;
// if (fxpv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
// return 1;
// }
// }
// f1_g.ProjectCoefficient(f1_c);
// rho_g.ProjectCoefficient(rho_c);
// auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
// if (dFdrho->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
// return 1;
// }
// dFdrho->Mult(rho_g, y);
// // fd test
// {
// double eps = 1.0e-6;
// Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()),
// frhomv(x.Size()); v *= eps; rhopv += v; rhomv -= v;
// dop.SetParameters({&rhopv, mesh_nodes});
// dop.Mult(x, frhopv);
// dop.SetParameters({&rhomv, mesh_nodes});
// dop.Mult(x, frhomv);
// frhopv -= frhomv;
// frhopv /= (2.0*eps);
// frhopv -= y;
// if (frhopv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
// return 1;
// }
// }
return 0;
}
DFEM_TEST_MAIN(test_diffusion);
+296
View File
@@ -0,0 +1,296 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include "examples/dfem/dfem_parametricspace.hpp"
#include "fem/bilininteg.hpp"
#include "general/tic_toc.hpp"
using namespace mfem;
using mfem::internal::tensor;
using mfem::internal::dual;
int test_diffusion_3d(
std::string mesh_file, int refinements, int polynomial_order)
{
constexpr int num_samples = 10;
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "incorrect mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(polynomial_order);
mesh_serial.Clear();
out << "#el: " << mesh.GetNE() << "\n";
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
0)->GetDim() - 1);
printf("#ndof per el = %d\n", h1fes.GetFE(0)->GetDof());
printf("#nqp = %d\n", ir.GetNPoints());
printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
ParametricSpace qdata_space(dim, dim * dim, ir.GetNPoints(),
dim * dim * ir.GetNPoints() * mesh.GetNE());
ParametricFunction qdata(qdata_space);
ParGridFunction f1_g(&h1fes);
ParGridFunction rho_g(&h1fes);
auto f1 = [](const Vector& coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + x + x*y + 1.25 * z*x;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(h1fes.GetTrueVSize());
{
auto diffusion_mf_kernel =
[] MFEM_HOST_DEVICE (
const tensor<dual<real_t, real_t>, dim>& dudxi,
const tensor<double, dim, dim>& J,
const double& w)
{
auto invJ = inv(J);
return mfem::tuple{dudxi * invJ * transpose(invJ) * det(J) * w};
};
mfem::tuple argument_operators = {Gradient{"potential"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator = {Gradient{"potential"}};
ElementOperator eop = {diffusion_mf_kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
dop.SetParameters({mesh_nodes});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, y);
}
sw.Stop();
printf("dfem mf: %fs\n", sw.RealTime() / num_samples);
y.HostRead();
}
{
auto diffusion_setup_kernel =
[] MFEM_HOST_DEVICE (
const tensor<double, dim, dim>& J,
const double& w)
{
auto invJ = inv(J);
return mfem::tuple{invJ * transpose(invJ) * det(J) * w};
};
mfem::tuple argument_operators = {Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator = {None{"qdata"}};
ElementOperator eop = {diffusion_setup_kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array
{
FieldDescriptor{&mesh_fes, "coordinates"},
FieldDescriptor{&qdata_space, "qdata"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
dop.SetParameters({mesh_nodes, &qdata});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, qdata);
}
sw.Stop();
printf("dfem pa setup: %fs\n", sw.RealTime() / num_samples);
qdata.HostRead();
}
// printf("qdata: ");
// print_vector(qdata);
{
auto diffusion_apply_kernel =
[] MFEM_HOST_DEVICE (
const tensor<dual<real_t, real_t>, dim>& dudxi,
const tensor<double, dim, dim>& qdata)
{
return mfem::tuple{dudxi * qdata};
};
mfem::tuple argument_operators = {Gradient{"potential"}, None{"qdata"}};
mfem::tuple output_operator = {Gradient{"potential"}};
ElementOperator eop = {diffusion_apply_kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&qdata_space, "qdata"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
dop.SetParameters({&qdata});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, y);
}
sw.Stop();
printf("dfem pa apply: %fs\n", sw.RealTime() / num_samples);
y.HostRead();
}
// printf("y: ");
// print_vector(y);
Vector y2(h1fes.TrueVSize());
{
ParBilinearForm a(&h1fes);
auto diff_integ = new DiffusionIntegrator;
diff_integ->SetIntRule(&ir);
a.AddDomainIntegrator(diff_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
OperatorPtr A;
StopWatch sw;
sw.Start();
a.Assemble();
a.Finalize();
Array<int> empty;
a.FormSystemMatrix(empty, A);
sw.Stop();
printf("mfem pa setup: %fs\n", sw.RealTime());
sw.Clear();
sw.Start();
for (int i = 0; i < num_samples; i++)
{
A->Mult(x, y2);
}
sw.Stop();
printf("mfem pa apply: %fs\n", sw.RealTime() / num_samples);
y2.HostRead();
}
// printf("y2: ");
// print_vector(y2);
Vector diff(y2);
diff -= y;
if (diff.Norml2() > 1e-15)
{
// printf("y ");
// print_vector(y);
// printf("y2: ");
// print_vector(y2);
// printf("diff: ");
// print_vector(diff);
return 1;
}
// Test linearization here as well
// auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
// if (dFdu->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
// return 1;
// }
// dFdu->Mult(x, y);
// y.HostRead();
// a.Mult(x, y2);
// y2.HostRead();
// diff = y2;
// diff -= y;
// if (diff.Norml2() > 1e-10)
// {
// print_vector(diff);
// print_vector(y2);
// print_vector(y);
// return 1;
// }
// // fd jacobian test
// {
// double eps = 1.0e-6;
// Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
// v *= eps;
// xpv += v;
// xmv -= v;
// dop.Mult(xpv, fxpv);
// dop.Mult(xmv, fxmv);
// fxpv -= fxmv;
// fxpv /= (2.0*eps);
// fxpv -= y;
// if (fxpv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
// return 1;
// }
// }
// f1_g.ProjectCoefficient(f1_c);
// rho_g.ProjectCoefficient(rho_c);
// auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
// if (dFdrho->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
// return 1;
// }
// dFdrho->Mult(rho_g, y);
// // fd test
// {
// double eps = 1.0e-6;
// Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()),
// frhomv(x.Size()); v *= eps; rhopv += v; rhomv -= v;
// dop.SetParameters({&rhopv, mesh_nodes});
// dop.Mult(x, frhopv);
// dop.SetParameters({&rhomv, mesh_nodes});
// dop.Mult(x, frhomv);
// frhopv -= frhomv;
// frhopv /= (2.0*eps);
// frhopv -= y;
// if (frhopv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
// return 1;
// }
// }
return 0;
}
DFEM_TEST_MAIN(test_diffusion_3d);
@@ -0,0 +1,309 @@
#include "dfem/dfem_test_macro.hpp"
#include "examples/dfem/dfem_fieldoperator.hpp"
#include "examples/dfem/dfem_refactor.hpp"
#include "fem/bilininteg.hpp"
#include "general/tic_toc.hpp"
#include <utility>
using namespace mfem;
using mfem::internal::tensor;
using mfem::internal::dual;
int test_diffusion_3d(
std::string mesh_file, int refinements, int polynomial_order)
{
constexpr int num_samples = 100;
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "incorrect mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(polynomial_order);
mesh_serial.Clear();
out << "#el: " << mesh.GetNE() << "\n";
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
0)->GetDim() - 1);
printf("#ndof per el = %d\n", h1fes.GetFE(0)->GetDof());
printf("#nqp = %d\n", ir.GetNPoints());
printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
ParametricSpace qdata_space(dim, dim * dim, ir.GetNPoints(),
dim * dim * ir.GetNPoints() * mesh.GetNE());
ParametricFunction qdata(qdata_space);
ParGridFunction f1_g(&h1fes);
ParGridFunction rho_g(&h1fes);
auto f1 = [](const Vector& coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + x + x*y + 1.25 * z*x;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(h1fes.GetTrueVSize());
{
std::shared_ptr<DerivativeOperator> dpotential;
{
auto diffusion_mf_kernel =
[] MFEM_HOST_DEVICE (
const tensor<real_t, dim>& dudxi,
const tensor<real_t, dim, dim>& J,
const real_t& w)
{
auto invJ = inv(J);
return mfem::tuple{dudxi * invJ * transpose(invJ) * det(J) * w};
};
constexpr int Potential = 0;
constexpr int Coordinates = 1;
auto input_operators = mfem::tuple{Gradient<Potential>{}, Gradient<Coordinates>{}, Weight{}};
auto output_operator = mfem::tuple{Gradient<Potential>{}};
auto solutions = std::vector{FieldDescriptor{Potential, &h1fes}};
auto parameters = std::vector{FieldDescriptor{Coordinates, &mesh_fes}};
DifferentiableOperator dop(solutions, parameters, mesh);
auto derivatives = std::integer_sequence<size_t, Potential> {};
dop.AddDomainIntegrator(
diffusion_mf_kernel, input_operators, output_operator, ir, derivatives);
dop.SetParameters({mesh_nodes});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, y);
}
sw.Stop();
printf("dfem mf: %fs\n", sw.RealTime() / num_samples);
y.HostRead();
dpotential = dop.GetDerivative(Potential, {&f1_g}, {mesh_nodes});
}
dpotential->Mult(x, y);
}
{
auto diffusion_setup_kernel =
[] MFEM_HOST_DEVICE (
const tensor<double, dim, dim>& J,
const double& w)
{
auto invJ = inv(J);
return mfem::tuple{invJ * transpose(invJ) * det(J) * w};
};
constexpr int Potential = 0;
constexpr int Coordinates = 1;
constexpr int QData = 2;
auto input_operators = mfem::tuple{Gradient<Coordinates>{}, Weight{}};
auto output_operator = mfem::tuple{None<QData>{}};
auto solutions = std::vector{FieldDescriptor{Potential, &h1fes}};
auto parameters = std::vector{FieldDescriptor{Coordinates, &mesh_fes},
FieldDescriptor{QData, &qdata_space}};
DifferentiableOperator dop(solutions, parameters, mesh);
dop.AddDomainIntegrator(
diffusion_setup_kernel, input_operators, output_operator, ir);
dop.SetParameters({mesh_nodes, &qdata});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, qdata);
}
sw.Stop();
printf("dfem pa setup: %fs\n", sw.RealTime() / num_samples);
qdata.HostRead();
}
// printf("qdata: ");
// print_vector(qdata);
{
auto diffusion_apply_kernel =
[] MFEM_HOST_DEVICE (
const tensor<real_t, dim>& dudxi,
const tensor<double, dim, dim>& qdata)
{
return mfem::tuple{dudxi * qdata};
};
constexpr int Potential = 0;
constexpr int QData = 1;
auto input_operators = mfem::tuple{Gradient<Potential>{}, None<QData>{}};
auto output_operator = mfem::tuple{Gradient<Potential>{}};
auto solutions = std::vector{FieldDescriptor{Potential, &h1fes}};
auto parameters = std::vector{FieldDescriptor{QData, &qdata_space}};
DifferentiableOperator dop(solutions, parameters, mesh);
dop.AddDomainIntegrator(
diffusion_apply_kernel, input_operators, output_operator, ir);
dop.SetParameters({&qdata});
StopWatch sw;
sw.Start();
for (int i = 0; i < num_samples; i++)
{
dop.Mult(x, y);
}
sw.Stop();
printf("dfem pa apply: %fs\n", sw.RealTime() / num_samples);
y.HostRead();
}
// printf("y: ");
// print_vector(y);
Vector y2(h1fes.TrueVSize());
{
ParBilinearForm a(&h1fes);
auto diff_integ = new DiffusionIntegrator;
diff_integ->SetIntRule(&ir);
a.AddDomainIntegrator(diff_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
OperatorPtr A;
StopWatch sw;
sw.Start();
a.Assemble();
a.Finalize();
Array<int> empty;
a.FormSystemMatrix(empty, A);
sw.Stop();
printf("mfem pa setup: %fs\n", sw.RealTime());
sw.Clear();
sw.Start();
y2 = 0.0;
for (int i = 0; i < num_samples; i++)
{
A->Mult(x, y2);
}
sw.Stop();
printf("mfem pa apply: %fs\n", sw.RealTime() / num_samples);
y2.HostRead();
}
// printf("y2: ");
// print_vector(y2);
Vector diff(y2);
diff -= y;
if (diff.Norml2() > 1e-15)
{
printf("y: ");
print_vector(y);
printf("y2: ");
print_vector(y2);
printf("diff: ");
print_vector(diff);
return 1;
}
// Test linearization here as well
// auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
// if (dFdu->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
// return 1;
// }
// dFdu->Mult(x, y);
// y.HostRead();
// a.Mult(x, y2);
// y2.HostRead();
// diff = y2;
// diff -= y;
// if (diff.Norml2() > 1e-10)
// {
// print_vector(diff);
// print_vector(y2);
// print_vector(y);
// return 1;
// }
// // fd jacobian test
// {
// double eps = 1.0e-6;
// Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
// v *= eps;
// xpv += v;
// xmv -= v;
// dop.Mult(xpv, fxpv);
// dop.Mult(xmv, fxmv);
// fxpv -= fxmv;
// fxpv /= (2.0*eps);
// fxpv -= y;
// if (fxpv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
// return 1;
// }
// }
// f1_g.ProjectCoefficient(f1_c);
// rho_g.ProjectCoefficient(rho_c);
// auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
// if (dFdrho->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
// return 1;
// }
// dFdrho->Mult(rho_g, y);
// // fd test
// {
// double eps = 1.0e-6;
// Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()),
// frhomv(x.Size()); v *= eps; rhopv += v; rhomv -= v;
// dop.SetParameters({&rhopv, mesh_nodes});
// dop.Mult(x, frhopv);
// dop.SetParameters({&rhomv, mesh_nodes});
// dop.Mult(x, frhomv);
// frhopv -= frhomv;
// frhopv /= (2.0*eps);
// frhopv -= y;
// if (frhopv.Norml2() > eps)
// {
// out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
// return 1;
// }
// }
return 0;
}
DFEM_TEST_MAIN(test_diffusion_3d);
+109
View File
@@ -0,0 +1,109 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_elasticity(std::string mesh_file,
int refinements,
int polynomial_order)
{
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
const int vdim = dim;
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_bdr(mesh.bdr_attributes.Max());
Array<int> ess_tdof;
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 6 * h1fec.GetOrder());
out << "#qp: " << ir.GetNPoints() << "\n";
out << "#dof_el: " << h1fes.GetRestrictionMatrix()->Height() / mesh.GetNE() <<
"\n";
ParGridFunction u(&h1fes);
auto f1 = [](const Vector& coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = 2.345 + 0.25 * x * x * y + y * y * x;
u(1) = 2.345 - 0.25 * x * y * y + y * x * x;
};
VectorFunctionCoefficient u_c(dim, f1);
u.ProjectCoefficient(u_c);
ConstantCoefficient l_coeff(0.5), m_coeff(0.25);
ParBilinearForm A_form(&h1fes);
auto A_integ = new ElasticityIntegrator(l_coeff, m_coeff);
A_integ->SetIntegrationRule(ir);
A_form.AddDomainIntegrator(A_integ);
A_form.Assemble();
A_form.Finalize();
auto elasticity_kernel = [](const tensor<double, 2, 2> &dudxi,
const tensor<double, 2, 2> &J,
const double &w)
{
constexpr double lambda = 0.5;
constexpr double mu = 0.25;
static constexpr auto I = mfem::internal::IsotropicIdentity<2>();
auto invJ = inv(J);
auto eps = sym(dudxi * invJ);
return mfem::tuple{transpose(lambda * tr(eps) * I + 2.0 * mu * eps) * det(J) * w * transpose(invJ)};
};
mfem::tuple argument_operators{Gradient{"displacement"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator{Gradient{"displacement"}};
ElementOperator op{elasticity_kernel, argument_operators, output_operator};
std::array solutions{FieldDescriptor{&h1fes, "displacement"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
Vector x(u), y1(h1fes.GetTrueVSize()),
y2(h1fes.GetTrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y1);
y1.HostRead();
A_form.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y1;
if (diff.Norml2() > 1e-10)
{
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
print_vector(diff);
print_vector(y1);
print_vector(y2);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_elasticity);
@@ -0,0 +1,115 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include "examples/dfem/dfem_parametricspace.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_interpolate_gradient_linear_scalar_3d(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
// const IntegrationRule &ir =
// IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
2 * polynomial_order - 1);
ParGridFunction f1_g(&h1fes);
ParametricSpace pspace(dim, dim, ir.GetNPoints(),
dim * ir.GetNPoints() * mesh.GetNE());
ParametricFunction qdata(pspace);
auto kernel = [](const tensor<double, dim> &dudxi,
const tensor<double, dim, dim> &J)
{
return mfem::tuple{dudxi * inv(J)};
};
mfem::tuple argument_operators = {Gradient{"potential"}, Gradient{"coordinates"}};
mfem::tuple output_operator = {None{"qdata"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array
{
FieldDescriptor{&mesh_fes, "coordinates"},
FieldDescriptor{&pspace, "qdata"}
};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + x * y * z + y * z;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize() * dim);
dop.SetParameters({mesh_nodes, &qdata});
dop.Mult(x, y);
Vector f_test(h1fes.GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC)->Height() * dim);
for (int e = 0; e < mesh.GetNE(); e++)
{
ElementTransformation *T = mesh.GetElementTransformation(e);
for (int qp = 0; qp < ir.GetNPoints(); qp++)
{
const IntegrationPoint &ip = ir.IntPoint(qp);
T->SetIntPoint(&ip);
Vector g(dim);
f1_g.GetGradient(*T, g);
// printf("(%f, %f, %f): (%f, %f, %f)\n", ip.x, ip.y, ip.z, g(0), g(1), g(2));
for (int d = 0; d < dim; d++)
{
int qpo = qp * dim;
int eo = e * (ir.GetNPoints() * dim);
f_test(d + qpo + eo) = g(d);
}
}
}
Vector diff(f_test);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(f_test);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_interpolate_gradient_linear_scalar_3d);
@@ -0,0 +1,91 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_interpolate_linear_scalar(std::string mesh_file,
int refinements,
int polynomial_order)
{
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
ParGridFunction f1_g(&h1fes);
auto kernel = [](const double &u, const tensor<double, 2, 2> &J,
const double &w)
{
return mfem::tuple{u};
};
mfem::tuple argument_operators = {Value{"potential"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator = {None{"potential"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
return 2.345 + x + y;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
Vector f_test(h1fes.GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC)->Height());
for (int e = 0; e < mesh.GetNE(); e++)
{
ElementTransformation *T = mesh.GetElementTransformation(e);
for (int qp = 0; qp < ir.GetNPoints(); qp++)
{
const IntegrationPoint &ip = ir.IntPoint(qp);
T->SetIntPoint(&ip);
f_test((e * ir.GetNPoints()) + qp) = f1_c.Eval(*T, ip);
}
}
Vector diff(f_test);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(f_test);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_interpolate_linear_scalar);
@@ -0,0 +1,93 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_interpolate_linear_scalar_3d(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
ParGridFunction f1_g(&h1fes);
auto kernel = [](const double &u)
{
return mfem::tuple{u};
};
mfem::tuple argument_operators = {Value{"potential"}};
mfem::tuple output_operator = {None{"potential"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + x + y + 1.25 * z;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
Vector f_test(h1fes.GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC)->Height());
for (int e = 0; e < mesh.GetNE(); e++)
{
ElementTransformation *T = mesh.GetElementTransformation(e);
for (int qp = 0; qp < ir.GetNPoints(); qp++)
{
const IntegrationPoint &ip = ir.IntPoint(qp);
T->SetIntPoint(&ip);
f_test((e * ir.GetNPoints()) + qp) = f1_c.Eval(*T, ip);
}
}
Vector diff(f_test);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(f_test);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_interpolate_linear_scalar_3d);
@@ -0,0 +1,100 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_interpolate_linear_vector(std::string mesh_file, int refinements,
int polynomial_order)
{
constexpr int vdim = 2;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
QuadratureSpace qspace(mesh, ir);
QuadratureFunction qf(&qspace, vdim);
ParGridFunction f1_g(&h1fes);
auto kernel = [](const tensor<double, 2> &u)
{
return mfem::tuple{u};
};
mfem::tuple argument_operators = {Value{"potential"}};
mfem::tuple output_operator = {None{"potential"}};
ElementOperator eop{kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = 2.345 + x + y;
u(1) = 12.345 + x + y;
};
VectorFunctionCoefficient f1_c(vdim, f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(f1_g.Size());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
Vector f_test(qf.Size());
for (int e = 0; e < mesh.GetNE(); e++)
{
ElementTransformation *T = mesh.GetElementTransformation(e);
for (int qp = 0; qp < ir.GetNPoints(); qp++)
{
const IntegrationPoint &ip = ir.IntPoint(qp);
T->SetIntPoint(&ip);
Vector f(vdim);
f1_g.GetVectorValue(*T, ip, f);
for (int d = 0; d < vdim; d++)
{
int qpo = qp * vdim;
int eo = e * (ir.GetNPoints() * vdim);
f_test(d + qpo + eo) = f(d);
}
}
}
Vector diff(f_test);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(f_test);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_interpolate_linear_vector);
@@ -0,0 +1,105 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_interpolate_linear_vector_3d(std::string mesh_file, int refinements,
int polynomial_order)
{
constexpr int dim = 3;
constexpr int vdim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
QuadratureSpace qspace(mesh, ir);
QuadratureFunction qf(&qspace, vdim);
ParGridFunction f1_g(&h1fes);
auto kernel = [](const tensor<double, vdim> &u)
{
return mfem::tuple{u};
};
mfem::tuple argument_operators = {Value{"potential"}};
mfem::tuple output_operator = {None{"potential"}};
ElementOperator eop{kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
u(0) = 2.345 + x + y + 3.0 * z;
u(1) = 12.345 + x + y + 2.0 * z;
u(2) = 5.345 + x + y + 1.0 * z;
};
VectorFunctionCoefficient f1_c(vdim, f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(f1_g.Size());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
Vector f_test(qf.Size());
for (int e = 0; e < mesh.GetNE(); e++)
{
ElementTransformation *T = mesh.GetElementTransformation(e);
for (int qp = 0; qp < ir.GetNPoints(); qp++)
{
const IntegrationPoint &ip = ir.IntPoint(qp);
T->SetIntPoint(&ip);
Vector f(vdim);
f1_g.GetVectorValue(*T, ip, f);
for (int d = 0; d < vdim; d++)
{
int qpo = qp * vdim;
int eo = e * (ir.GetNPoints() * vdim);
f_test(d + qpo + eo) = f(d);
}
}
}
Vector diff(f_test);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(f_test);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_interpolate_linear_vector_3d);
+113
View File
@@ -0,0 +1,113 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include "fem/bilininteg.hpp"
#include "fem/normal_deriv_restriction.hpp"
#include <fstream>
using namespace mfem;
using mfem::internal::tensor;
int dfem_test_mass_scalar_2d(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 2;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
// IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
// const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
// 2 * polynomial_order - 1);
printf("#nqp = %d\n", ir.GetNPoints());
printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
ParGridFunction f1_g(&h1fes);
auto kernel = [](const double& u,
const tensor<double, dim> x,
const tensor<double, dim, dim> J,
const double& w)
{
out << x << ": " << u << "\n";
return mfem::tuple{u * w * det(J)};
};
mfem::tuple argument_operators = {Value{"potential"}, Value{"coordinates"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator = {Value{"potential"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
return 2.345 + x + x*y + 1.25 * x;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector f1_g_e(f1_g.Size());
auto R = h1fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
// R->Mult(f1_g, f1_g_e);
auto r_out = std::ofstream("r_mat.mtx");
R->PrintMatlab(r_out);
r_out.close();
print_vector(f1_g);
// print_vector(f1_g_e);
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
ParBilinearForm a(&h1fes);
auto mass_integ = new MassIntegrator;
mass_integ->SetIntRule(&ir);
a.AddDomainIntegrator(mass_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble();
a.Finalize();
Vector y2(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
if (diff.Norml2() > 1e-10)
{
print_vector(diff);
print_vector(y2);
print_vector(y);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(dfem_test_mass_scalar_2d);
+147
View File
@@ -0,0 +1,147 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include "fem/bilininteg.hpp"
#include "fem/fe/fe_base.hpp"
using namespace mfem;
using mfem::internal::tensor;
int dfem_test_mass_scalar_3d(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(polynomial_order);
mesh_serial.Clear();
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
0)->GetDim() - 1);
// IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
// const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
// 2 * polynomial_order - 1);
auto dtq = h1fes.GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
// printf("\n B: ");
// dtq.B.Print(out, dtq.B.Size());
// printf("\n G: ");
// dtq.G.Print(out, dtq.G.Size());
// printf("\n w: ");
// ir.GetWeights().Print(out, ir.GetWeights().Size());
// printf("#ndof per el = %d\n", h1fes.GetFE(0)->GetDof());
// printf("#nqp = %d\n", ir.GetNPoints());
// printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
// printf("nodes: ");
// print_vector(*mesh_nodes);
ParGridFunction f1_g(&h1fes);
auto kernel = [](const double &u,
const tensor<double, dim, dim> &J,
const double &w)
{
return mfem::tuple{u * det(J) * w};
};
mfem::tuple argument_operators = {Value{"potential"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator = {Value{"potential"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector &coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + x + x*y + 1.25 * z*x;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
// printf("\nf1_g: ");
// print_vector(f1_g);
auto R = h1fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
// Vector f1_g_e(R->Height());
// R->Mult(f1_g, f1_g_e);
// printf("\nf1_g_e: ");
// print_vector(f1_g_e);
// auto r_out = std::ofstream("r_mat.mtx");
// R->PrintMatlab(r_out);
// r_out.close();
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
ParBilinearForm a(&h1fes);
auto mass_integ = new MassIntegrator;
mass_integ->SetIntRule(&ir);
a.AddDomainIntegrator(mass_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble();
a.Finalize();
Vector y2(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
if (diff.Norml2() > 1e-15)
{
printf("y ");
print_vector(y);
printf("y2: ");
print_vector(y2);
printf("diff: ");
print_vector(diff);
return 1;
}
Vector y3(h1fes.TrueVSize());
auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
dFdu->Mult(x, y3);
diff = y2;
diff -= y;
if (diff.Norml2() > 1e-15)
{
printf("y2 ");
print_vector(y2);
printf("y3: ");
print_vector(y3);
printf("diff: ");
print_vector(diff);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(dfem_test_mass_scalar_3d);
@@ -0,0 +1,114 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_neo_hookean_elasticity_2d(
std::string mesh_file, int refinements, int polynomial_order)
{
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
MFEM_ASSERT(dim == 2, "This test is for 2D meshes only");
mesh_serial.Clear();
out << "#el: " << mesh.GetNE() << "\n";
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, dim);
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
out << "#qp: " << ir.GetNPoints() << "\n";
ParGridFunction u_g(&h1fes);
auto kernel = [] MFEM_HOST_DEVICE(const tensor<double, 2, 2>& J,
const double& w,
const tensor<double, 2, 2>& dudxi)
{
// Neo-Hookean parameters
const double lambda = 1.0;
const double mu = 0.5;
static constexpr auto I = mfem::internal::IsotropicIdentity<2>();
auto F = I + (dudxi * inv(J));
auto E = 0.5 * (transpose(F) * F - I);
auto invF = inv(F);
// 2D plane strain formulation
auto P = mu * (F - transpose(invF)) + lambda * log(det(F)) * transpose(invF);
return mfem::tuple{P * det(J) * w};
};
mfem::tuple argument_operators = {Gradient{"coordinates"}, Weight{},
Gradient{"displacement"}
};
mfem::tuple output_operator = {Gradient{"displacement"}};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array{FieldDescriptor{&h1fes, "displacement"}};
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto displacement = [](const Vector& coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = 0.1 * x * y;
u(1) = 0.1 * y * x;
};
VectorFunctionCoefficient disp_coeff(2, displacement);
u_g.ProjectCoefficient(disp_coeff);
Vector x(u_g), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
y.HostRead();
// Test linearization
auto dFdu = dop.GetDerivativeWrt<0>({&u_g}, {mesh_nodes});
dFdu->Mult(x, y);
// Finite difference Jacobian test
{
double eps = 1.0e-6;
Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
v *= eps;
xpv += v;
xmv -= v;
dop.Mult(xpv, fxpv);
dop.Mult(xmv, fxmv);
fxpv -= fxmv;
fxpv /= (2.0*eps);
fxpv -= y;
if (fxpv.Norml2() > eps)
{
out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
return 1;
}
}
return 0;
}
DFEM_TEST_MAIN(test_neo_hookean_elasticity_2d);
@@ -0,0 +1,169 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_nonlinear_diffusion(
std::string mesh_file, int refinements, int polynomial_order)
{
constexpr int dim = 3;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
mesh_serial.Clear();
out << "#el: " << mesh.GetNE() << "\n";
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec);
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
0)->GetDim() - 1);
out << "#qp: " << ir.GetNPoints() << "\n";
ParGridFunction f1_g(&h1fes);
bool inactive_derivative = false;
auto kernel = [] MFEM_HOST_DEVICE(
const tensor<double, dim, dim>& J,
const double& w,
const tensor<double, dim>& dudxi,
const double& u)
{
auto invJ = inv(J);
return mfem::tuple{(u * u) * dudxi * invJ * transpose(invJ) * det(J) * w};
};
mfem::tuple argument_operators =
{
Gradient{"coordinates"},
Weight{},
Gradient{"potential"},
Value{"potential"}
};
mfem::tuple output_operator =
{
Gradient{"potential"}
};
ElementOperator eop = {kernel, argument_operators, output_operator};
auto ops = mfem::tuple{eop};
auto solutions = std::array
{
FieldDescriptor{&h1fes, "potential"}
};
auto parameters = std::array
{
FieldDescriptor{&mesh_fes, "coordinates"}
};
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
auto f1 = [](const Vector& coords)
{
const double x = coords(0);
const double y = coords(1);
const double z = coords(2);
return 2.345 + 0.25 * x * x * y + y * y * x + z;
};
FunctionCoefficient f1_c(f1);
f1_g.ProjectCoefficient(f1_c);
Vector x(f1_g), y(h1fes.TrueVSize());
dop.SetParameters({mesh_nodes});
dop.Mult(x, y);
y.HostRead();
ParBilinearForm a(&h1fes);
GridFunctionCoefficient f1gc(&f1_g);
TransformedCoefficient tf_c(&f1gc, [](double f) { return f * f; });
a.AddDomainIntegrator(new DiffusionIntegrator(tf_c));
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble();
a.Finalize();
Vector y2(h1fes.TrueVSize()), diff(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
diff = y2;
diff -= y;
if (diff.Norml2() > 1e-10)
{
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
print_vector(diff);
print_vector(y);
print_vector(y2);
return 1;
}
// Test linearization here as well
auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
dFdu->Mult(x, y);
// fd jacobian test
{
double eps = 1.0e-6;
Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
v *= eps;
xpv += v;
xmv -= v;
dop.Mult(xpv, fxpv);
dop.Mult(xmv, fxmv);
fxpv -= fxmv;
fxpv /= (2.0*eps);
fxpv -= y;
if (fxpv.Norml2() > eps)
{
out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
return 1;
}
}
// ParBilinearForm da(&h1fes);
// TransformedCoefficient dtf_c(&f1gc, [](double f) { return 2.0 * f; });
// da.AddDomainIntegrator(new DiffusionIntegrator(dtf_c));
// da.SetAssemblyLevel(AssemblyLevel::PARTIAL);
// da.Assemble();
// da.Finalize();
// if (dFdu->Height() != h1fes.GetTrueVSize())
// {
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
// return 1;
// }
// dFdu->Mult(x, y);
// print_vector(y);
// da.Mult(x, y2);
// print_vector(y2);
// y2 -= y;
// out << "||dFdu x - A x||_l2 = " << y2.Norml2() << "\n";
// if (y2.Norml2() > 1e-10)
// {
// out << "||dFdu u^* - ex||_l2 = " << y2.Norml2() << "\n";
// }
return 0;
}
DFEM_TEST_MAIN(test_nonlinear_diffusion);
@@ -0,0 +1,267 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include <fstream>
using namespace mfem;
using mfem::internal::tensor;
using mfem::internal::dual;
class FDJacobian : public Operator
{
public:
FDJacobian(const Operator &op, const Vector &x) :
Operator(op.Height()),
op(op),
x(x)
{
f.SetSize(Height());
xpev.SetSize(Height());
op.Mult(x, f);
xnorm = x.Norml2();
}
void Mult(const Vector &v, Vector &y) const override
{
x.HostRead();
// See [1] for choice of eps.
//
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
// finite difference matrix-vector products in Newton-Krylov solvers for
// implicit climate dynamics with spectral elements. Procedia Computer
// Science, 51, pp.2036-2045.
real_t eps = lambda * (lambda + xnorm / v.Norml2());
for (int i = 0; i < x.Size(); i++)
{
xpev(i) = x(i) + eps * v(i);
}
// y = f(x + eps * v)
op.Mult(xpev, y);
// y = (f(x + eps * v) - f(x)) / eps
for (int i = 0; i < x.Size(); i++)
{
y(i) = (y(i) - f(i)) / eps;
}
}
virtual MemoryClass GetMemoryClass() const override
{
return Device::GetDeviceMemoryClass();
}
private:
const Operator &op;
Vector x, f;
mutable Vector xpev;
real_t lambda = 1.0e-6;
real_t xnorm;
};
template <typename elasticity_t>
class ElasticityOperator : public Operator
{
template <typename elasticity_du_t>
class ElasticityJacobianOperator : public Operator
{
public:
ElasticityJacobianOperator(const ElasticityOperator *elasticity,
std::shared_ptr<elasticity_du_t> dRdu) :
Operator(elasticity->Height()),
elasticity(elasticity),
dRdu(dRdu),
x_ess(dRdu->Height())
{
}
void Mult(const Vector &x, Vector &y) const override
{
x_ess = x;
x_ess.SetSubVector(elasticity->ess_tdofs, 0.0);
dRdu->Mult(x_ess, y);
for (int i = 0; i < elasticity->ess_tdofs.Size(); i++)
{
y[elasticity->ess_tdofs[i]] = x[elasticity->ess_tdofs[i]];
}
}
const ElasticityOperator *elasticity = nullptr;
std::shared_ptr<elasticity_du_t> dRdu;
mutable Vector x_ess;
};
public:
ElasticityOperator(ParFiniteElementSpace &fes, elasticity_t &elasticity,
Array<int> &ess_tdofs) :
Operator(fes.GetTrueVSize()),
fes(fes),
elasticity(elasticity),
ess_tdofs(ess_tdofs) {}
void Mult(const Vector &x, Vector &r) const override
{
elasticity.Mult(x, r);
r.SetSubVector(ess_tdofs, 0.0);
}
Operator &GetGradient(const Vector &x) const override
{
ParGridFunction u(const_cast<ParFiniteElementSpace *>
(*std::get_if<const ParFiniteElementSpace *>
(&elasticity.solutions[0].data)));
u.SetFromTrueDofs(x);
auto dRdu = elasticity.template GetDerivativeWrt<0>({&u}, {mesh_nodes});
jacobian.reset(
new ElasticityJacobianOperator<
typename std::remove_pointer<decltype(dRdu.get())>::type> (this, dRdu));
// jacobian.reset(new FDJacobian(*this, x));
return *jacobian;
}
void SetParameters(ParGridFunction &mesh_nodes)
{
elasticity.SetParameters({&mesh_nodes});
this->mesh_nodes = &mesh_nodes;
}
ParFiniteElementSpace &fes;
elasticity_t &elasticity;
Array<int> ess_tdofs;
mutable ParGridFunction *mesh_nodes = nullptr;
mutable std::shared_ptr<Operator> jacobian;
};
int test_nonlinear_elasticity_3d(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 3;
constexpr int vdim = dim;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(polynomial_order);
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_tdof_list, ess_bdr(mesh.bdr_attributes.Max());
ess_bdr = 0;
ess_bdr[0] = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
const IntegrationRule& ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
0)->GetDim() - 1);
out << "#qp: " << ir.GetNPoints() << "\n";
out << "#dof: " << h1fes.GetNDofs() << "\n";
ParGridFunction u(&h1fes);
auto elasticity_kernel = [] MFEM_HOST_DEVICE
(const tensor<dual<real_t, real_t>, dim, dim> &dudxi,
const tensor<real_t, dim, dim> &J,
const real_t &w)
{
// shear modulus
real_t D1{0.1e6};
// bulk modulus
real_t C1{1.0e6};
constexpr auto I = mfem::internal::IsotropicIdentity<dim>();
auto invJ = inv(J);
auto dudx = dudxi * invJ;
auto F = det(I + dudx);
auto p = -2.0 * D1 * F * (F - 1);
auto devB = dev(dudx + transpose(dudx) + dot(dudx, transpose(dudx)));
auto sigma = -(p / F) * I + 2.0 * (C1 / pow(F, 5.0 / 3.0)) * devB;
return mfem::tuple{sigma * det(J) * w * transpose(invJ)};
};
mfem::tuple argument_operators{Gradient{"displacement"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator{Gradient{"displacement"}};
// B^T D(B0*dudxi, B1*J, B2*w)
ElementOperator op(elasticity_kernel, argument_operators, output_operator, ir);
std::array solutions{FieldDescriptor{&h1fes, "displacement"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop(solutions, parameters, mfem::tuple{op}, mesh,
AutoDiff::NativeDualNumber{});
ElasticityOperator elasticity(h1fes, dop, ess_tdof_list);
VectorArrayCoefficient f(dim);
for (int i = 0; i < dim-1; i++)
{
f.Set(i, new ConstantCoefficient(0.0));
}
{
Vector pull_force(mesh.bdr_attributes.Max());
pull_force = 0.0;
pull_force(1) = -1.0e-2;
f.Set(dim-1, new PWConstCoefficient(pull_force));
}
ParLinearForm b(&h1fes);
b.AddBoundaryIntegrator(new VectorBoundaryLFIntegrator(f));
b.UseFastAssembly(true);
b.Assemble();
auto B = b.ParallelAssemble();
Vector X = u.GetTrueVector();
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-8);
cg.SetMaxIter(1000);
cg.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
NewtonSolver newton(MPI_COMM_WORLD);
newton.SetSolver(cg);
newton.SetOperator(elasticity);
newton.SetRelTol(1e-6);
newton.SetMaxIter(100);
// newton.SetAdaptiveLinRtol();
newton.SetPrintLevel(IterativeSolver::PrintLevel().Iterations());
elasticity.SetParameters(*mesh_nodes);
// Vector zero;
newton.Mult(*B, X);
u.SetFromTrueDofs(X);
ParaViewDataCollection paraview_dc("dfem", &mesh);
paraview_dc.SetPrefixPath("ParaView");
paraview_dc.SetLevelsOfDetail(polynomial_order);
paraview_dc.SetDataFormat(VTKFormat::BINARY);
paraview_dc.SetHighOrderOutput(true);
paraview_dc.SetCycle(0);
paraview_dc.SetTime(0.0);
paraview_dc.RegisterField("displacement", &u);
paraview_dc.Save();
return 0;
}
DFEM_TEST_MAIN(test_nonlinear_elasticity_3d);
+82
View File
@@ -0,0 +1,82 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
#include "fem/coefficient.hpp"
#include "fem/pgridfunc.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_ordering(std::string mesh_file,
int refinements,
int polynomial_order)
{
constexpr int dim = 2;
constexpr int vdim = dim;
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(polynomial_order);
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
const IntegrationRule &ir =
IntRules.Get(mesh_fes.GetFE(0)->GetGeomType(),
2 * mesh_fes.FEColl()->GetOrder() - 1);
for (int q = 0; q < ir.GetNPoints(); q++)
{
out << "(" << ir.IntPoint(q).x << ", " << ir.IntPoint(q).y << ")\n";
}
ParGridFunction u(&mesh_fes);
auto f = [](const Vector &coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = x*x*y + 1.0;
u(1) = y*y*x*x + 2.0;
};
VectorFunctionCoefficient uc(dim, f);
u.ProjectCoefficient(uc);
auto kernel = [](const tensor<double, dim> &xi,
const tensor<double, vdim, dim> &J,
const tensor<double, dim> &u,
const tensor<double, vdim, dim> &dudxi)
{
out << "xi: " << xi << "\n";
out << "J: " << J << "\n";
out << "u: " << u << "\n";
out << "dudxi: " << dudxi << "\n\n";
return mfem::tuple{J};
};
mfem::tuple argument_operators{Value{"coordinates"}, Gradient{"coordinates"}, Value{"potential"}, Gradient{"potential"}};
mfem::tuple output_operator{Gradient{"potential"}};
ElementOperator op{kernel, argument_operators, output_operator};
std::array solutions{FieldDescriptor{&mesh_fes, "potential"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
Vector y(u);
dop.SetParameters({mesh_nodes});
dop.Mult(u, y);
print_vector(y);
return 0;
}
DFEM_TEST_MAIN(test_ordering);
+102
View File
@@ -0,0 +1,102 @@
#include "dfem/dfem.hpp"
#include "dfem/dfem_test_macro.hpp"
using namespace mfem;
using mfem::internal::tensor;
int test_vector_diffusion(std::string mesh_file,
int refinements,
int polynomial_order)
{
Mesh mesh_serial = Mesh(mesh_file);
for (int i = 0; i < refinements; i++)
{
mesh_serial.UniformRefinement();
}
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
mesh.SetCurvature(1);
const int dim = mesh.Dimension();
const int vdim = dim;
mesh_serial.Clear();
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
H1_FECollection h1fec(polynomial_order, dim);
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
Array<int> ess_bdr(mesh.bdr_attributes.Max());
Array<int> ess_tdof;
ess_bdr = 1;
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof);
const IntegrationRule &ir =
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() - 1);
ParGridFunction u(&h1fes);
auto f1 = [](const Vector& coords, Vector &u)
{
const double x = coords(0);
const double y = coords(1);
u(0) = 2.345 + 0.25 * x * x * y + y * y * x;
u(1) = 2.345 - 0.25 * x * y * y + y * x * x;
};
VectorFunctionCoefficient u_c(dim, f1);
u.ProjectCoefficient(u_c);
auto vector_diffusion_kernel = [](const tensor<double, 2> &xi,
const tensor<double, 2, 2> &dudxi,
const tensor<double, 2, 2> &J,
const double &w)
{
out << "xi: " << xi << "\n";
out << "dudxi: " << dudxi << "\n";
return mfem::tuple{dudxi * inv(J) * det(J) * w * transpose(inv(J))};
// return mfem::tuple{dudxi};
};
mfem::tuple argument_operators{Value{"coordinates"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{}};
mfem::tuple output_operator{Gradient{"potential"}};
ElementOperator op{vector_diffusion_kernel, argument_operators, output_operator};
std::array solutions{FieldDescriptor{&h1fes, "potential"}};
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
Vector x(u), y1(h1fes.GetTrueVSize()),
y2(h1fes.GetTrueVSize());
ParBilinearForm A_form(&h1fes);
auto A_integ = new VectorDiffusionIntegrator(vdim);
A_integ->SetIntegrationRule(ir);
A_form.AddDomainIntegrator(A_integ);
A_form.Assemble();
A_form.Finalize();
dop.SetParameters({mesh_nodes});
dop.Mult(x, y1);
y1.HostRead();
A_form.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y1;
if (diff.Norml2() > 1e-10)
{
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
print_vector(diff);
print_vector(y1);
print_vector(y2);
return 1;
}
return 0;
}
DFEM_TEST_MAIN(test_vector_diffusion);
+122
View File
@@ -0,0 +1,122 @@
#include <tuple>
#include <type_traits>
#include <iostream>
#include <enzyme/enzyme>
template <typename T>
constexpr auto get_type_name() -> std::string_view
{
#if defined(__clang__)
constexpr auto prefix = std::string_view {"[T = "};
constexpr auto suffix = "]";
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
#elif defined(__GNUC__)
constexpr auto prefix = std::string_view {"with T = "};
constexpr auto suffix = "; ";
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
#elif defined(_MSC_VER)
constexpr auto prefix = std::string_view {"get_type_name<"};
constexpr auto suffix = ">(void)";
constexpr auto function = std::string_view{__FUNCSIG__};
#else
#error Unsupported compiler
#endif
const auto start = function.find(prefix) + prefix.size();
const auto end = function.find(suffix);
const auto size = end - start;
return function.substr(start, size);
}
template <typename ... Ts>
constexpr auto decay_types(std::tuple<Ts...> const &)
-> std::tuple<std::remove_cv_t<std::remove_reference_t<Ts>>...>;
template <typename T>
using decay_tuple = decltype(decay_types(std::declval<T>()));
template <class F> struct FunctionSignature;
template <typename output_t, typename... input_ts>
struct FunctionSignature<output_t(input_ts...)>
{
using return_t = output_t;
using parameter_ts = std::tuple<input_ts...>;
};
template <class T> struct create_function_signature;
template <typename output_t, typename T, typename... input_ts>
struct create_function_signature<output_t (T::*)(input_ts...) const>
{
using type = FunctionSignature<output_t(input_ts...)>;
};
template <typename arg_ts, std::size_t... Is>
auto create_enzyme_args(arg_ts &args,
arg_ts &shadow_args,
std::index_sequence<Is...>)
{
// (std::cout << ... << std::get<Is>(shadow_args));
return std::tuple<enzyme::Duplicated<decltype(std::get<Is>(args))>...>
{
{ std::get<Is>(args), std::get<Is>(shadow_args) }...
};
}
template <typename kernel_t, typename arg_ts>
auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args, arg_ts &&shadow_args)
{
auto arg_indices =
std::make_index_sequence<std::tuple_size_v<std::remove_reference_t<arg_ts>>> {};
auto enzyme_args = create_enzyme_args(args, shadow_args, arg_indices);
// using kf_return_t = typename create_function_signature<
// decltype(&kernel_t::operator())>::type::return_t;
std::cout << "\n";
std::cout << "args is " << get_type_name<decltype(args)>() << "\n\n";
std::cout << "enzyme_args type is " << get_type_name<decltype(enzyme_args)>() <<
"\n\n";
// std::cout << "return type is " << get_type_name<decltype(kf_return_t{})>() <<
// "\n\n";
std::cout << "args " << std::get<0>(args) << "\n";
std::cout << "shadow args " << std::get<0>(shadow_args) << "\n";
return std::apply([&](auto &&...args)
{
// std::cout << enzyme::autodiff<enzyme::Forward>(+kernel, args...) << "\n";
return enzyme::get<0>
(enzyme::autodiff<enzyme::Forward>(+kernel, args...));
},
enzyme_args);
}
int main()
{
auto func = [](const double &x, double &y)
{
std::cout << "func( x = " << x << " )\n";
return x*x;
};
using kf_param_ts = typename create_function_signature<
decltype(&decltype(func)::operator())>::type::parameter_ts;
using kf_output_t = typename create_function_signature<
decltype(&decltype(func)::operator())>::type::return_t;
auto kernel_args = decay_tuple<kf_param_ts> {};
auto kernel_shadow_args = decay_tuple<kf_param_ts> {};
std::get<0>(kernel_args) = 3;
std::get<0>(kernel_shadow_args) = 1;
auto dx = fwddiff_apply_enzyme(func, kernel_args, kernel_shadow_args);
std::cout << "dfdx = " << dx << "\n";
return 0;
}
+16 -20
View File
@@ -44,7 +44,7 @@ protected:
BilinearForm *M;
BilinearForm *K;
SparseMatrix Mmat, Kmat, Kmat0;
SparseMatrix Mmat, Kmat;
SparseMatrix *T; // T = M + dt K
real_t current_dt;
@@ -83,25 +83,24 @@ WaveOperator::WaveOperator(FiniteElementSpace &f,
: SecondOrderTimeDependentOperator(f.GetTrueVSize(), (real_t) 0.0),
fespace(f), M(NULL), K(NULL), T(NULL), current_dt(0.0), z(height)
{
const real_t rel_tol = 1e-8;
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
// Assemble Laplace matrix
c2 = new ConstantCoefficient(speed*speed);
K = new BilinearForm(&fespace);
K->AddDomainIntegrator(new DiffusionIntegrator(*c2));
K->Assemble();
Array<int> dummy;
K->FormSystemMatrix(dummy, Kmat0);
K->FormSystemMatrix(ess_tdof_list, Kmat);
// Assemble Mass matrix
M = new BilinearForm(&fespace);
M->AddDomainIntegrator(new MassIntegrator());
M->Assemble();
// Apply Bcs
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
K->FormSystemMatrix(ess_tdof_list, Kmat);
M->FormSystemMatrix(ess_tdof_list, Mmat);
// Configure preconditioner
const real_t rel_tol = 1e-8;
M_solver.iterative_mode = false;
M_solver.SetRelTol(rel_tol);
M_solver.SetAbsTol(0.0);
@@ -110,14 +109,13 @@ WaveOperator::WaveOperator(FiniteElementSpace &f,
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(Mmat);
// Configure solver
T_solver.iterative_mode = false;
T_solver.SetRelTol(rel_tol);
T_solver.SetAbsTol(0.0);
T_solver.SetMaxIter(100);
T_solver.SetPrintLevel(0);
T_solver.SetPreconditioner(T_prec);
T = NULL;
}
void WaveOperator::Mult(const Vector &u, const Vector &du_dt,
@@ -126,9 +124,11 @@ void WaveOperator::Mult(const Vector &u, const Vector &du_dt,
// Compute:
// d2udt2 = M^{-1}*-K(u)
// for d2udt2
Kmat.Mult(u, z);
K->FullMult(u, z);
z.Neg(); // z = -z
z.SetSubVector(ess_tdof_list, 0.0);
M_solver.Mult(z, d2udt2);
d2udt2.SetSubVector(ess_tdof_list, 0.0);
}
void WaveOperator::ImplicitSolve(const real_t fac0, const real_t fac1,
@@ -142,14 +142,11 @@ void WaveOperator::ImplicitSolve(const real_t fac0, const real_t fac1,
T = Add(1.0, Mmat, fac0, Kmat);
T_solver.SetOperator(*T);
}
Kmat0.Mult(u, z);
K->FullMult(u, z);
z.Neg();
for (int i = 0; i < ess_tdof_list.Size(); i++)
{
z[ess_tdof_list[i]] = 0.0;
}
z.SetSubVector(ess_tdof_list, 0.0);
T_solver.Mult(z, d2udt2);
d2udt2.SetSubVector(ess_tdof_list, 0.0);
}
void WaveOperator::SetParameters(const Vector &u)
@@ -314,7 +311,6 @@ int main(int argc, char *argv[])
ess_bdr = 0;
}
}
WaveOperator oper(fespace, ess_bdr, speed);
u_gf.SetFromTrueDofs(u);
+2
View File
@@ -67,6 +67,8 @@ public:
ZCoefficient(int vdim, GridFunction &psi_, real_t alpha_ = 1.0)
: VectorCoefficient(vdim), psi(&psi_), alpha(alpha_) { }
using VectorCoefficient::Eval;
virtual void Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip);
void SetAlpha(real_t alpha_) { alpha = alpha_; }
+2
View File
@@ -67,6 +67,8 @@ public:
ZCoefficient(int vdim, ParGridFunction &psi_, real_t alpha_ = 1.0)
: VectorCoefficient(vdim), psi(&psi_), alpha(alpha_) { }
using VectorCoefficient::Eval;
virtual void Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip);
void SetAlpha(real_t alpha_) { alpha = alpha_; }
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/ginkgo/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+1
View File
@@ -96,6 +96,7 @@ public:
{
Vector w_glob(width);
pfes.Dof_TrueDof_Matrix()->MultTranspose(w, w_glob);
w_glob.HostReadWrite(); // read+write -> can use w_glob(i) (non-const)
for (int i = 0; i < width; i++) { grad(0, i) = w_glob(i); }
}
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/hiop/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ..
MFEM_BUILD_DIR ?= ..
MFEM_INSTALL_DIR ?= ../mfem
SRC = $(if $(MFEM_DIR:..=),$(MFEM_DIR)/examples/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/moonolith/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/petsc/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
-1
View File
@@ -66,7 +66,6 @@ int main(int argc, char *argv[])
{
// 1. Initialize MPI (required by PUMI) and HYPRE.
Mpi::Init(argc, argv);
int num_procs = Mpi::WorldSize();
int myid = Mpi::WorldRank();
Hypre::Init();
-2
View File
@@ -80,8 +80,6 @@ int main(int argc, char *argv[])
{
// 1. Initialize MPI (required by PUMI) and HYPRE.
Mpi::Init(argc, argv);
int num_proc = Mpi::WorldSize();
int myId = Mpi::WorldRank();
Hypre::Init();
// 2. Parse command-line options.
+3 -4
View File
@@ -12,11 +12,10 @@
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
MFEM_INSTALL_DIR ?= ../../mfem
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/pumi/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
$(wildcard $(MFEM_INSTALL_DIR)/share/mfem/config.mk))
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
+16 -3
View File
@@ -31,11 +31,21 @@ include_directories(BEFORE ${PROJECT_BINARY_DIR})
add_custom_target(test_sundials
${CMAKE_CTEST_COMMAND} -R sundials USES_TERMINAL)
# Add one executable per cpp file, adding "sundials_" as prefix. Sets
# "test_sundials" as a target that depends on the given examples.
# Add one executable per cpp file, adding "sundials_" as prefix so the CMake
# target is unique from those in the non-SUNDIALS examples. Also sets
# "test_sundials" as a target that depends on the given SUNDIALS examples.
set(PFX sundials_)
add_mfem_examples(SUNDIALS_EXAMPLES_SRCS ${PFX} "" test_sundials)
# Remove "sundials_" prefix from exectuable name for consistency with GNU build
# system.
foreach(SRC_FILE ${SUNDIALS_EXAMPLES_SRCS})
get_filename_component(SRC_FILENAME ${SRC_FILE} NAME)
string(REPLACE ".cpp" "" TARGET_NAME "${PFX}${SRC_FILENAME}")
string(REPLACE ${PFX} "" EXE_NAME ${TARGET_NAME})
set_target_properties(${TARGET_NAME} PROPERTIES OUTPUT_NAME ${EXE_NAME})
endforeach()
# Testing.
# The SUNDIALS tests can be run separately using the target "test_sundials"
# which builds the examples and runs:
@@ -51,7 +61,10 @@ if (MFEM_ENABLE_TESTING)
set(EX10_COMMON_OPTS -m ../../data/beam-quad.mesh -o 2 -s 5 -dt 0.15 -tf 6 -vs 10)
set(EX10_TEST_OPTS ${EX10_COMMON_OPTS} -r 2)
set(EX10P_TEST_OPTS ${EX10_COMMON_OPTS} -rp 1)
# Example 16: use the default options
# Example 16: test ARKODE with implicit time stepping using mass form
set(EX16_COMMON_OPTS -s 15)
set(EX16_TEST_OPTS ${EX16_COMMON_OPTS})
set(EX16P_TEST_OPTS ${EX16_COMMON_OPTS})
# Add the tests: one test per source file.
foreach(SRC_FILE ${SUNDIALS_EXAMPLES_SRCS})
+3 -1
View File
@@ -1,7 +1,9 @@
// MFEM Example 10
// SUNDIALS Modification
//
// Compile with: make ex10
// Compile with:
// make ex10 (GNU make)
// make sundials_ex10 (CMake)
//
// Sample runs:
// ex10 -m ../../data/beam-quad.mesh -r 2 -o 2 -s 12 -dt 0.15 -vs 10
+3 -1
View File
@@ -1,7 +1,9 @@
// MFEM Example 10 - Parallel Version
// SUNDIALS Modification
//
// Compile with: make ex10p
// Compile with:
// make ex10p (GNU make)
// make sundials_ex10p (CMake)
//
// Sample runs:
// mpirun -np 4 ex10p -m ../../data/beam-quad.mesh -rp 1 -o 2 -s 12 -dt 0.15 -vs 10
+256 -163
View File
@@ -1,15 +1,21 @@
// MFEM Example 16
// SUNDIALS Modification
//
// Compile with: make ex16
// Compile with:
// make ex16 (GNU make)
// make sundials_ex16 (CMake)
//
// Sample runs: ex16
// ex16 -m ../../data/inline-tri.mesh
// ex16 -m ../../data/disc-nurbs.mesh -tf 2
// ex16 -s 12 -a 0.0 -k 1.0
// ex16 -s 15 -a 0.0 -k 1.0
// ex16 -s 8 -a 1.0 -k 0.0 -dt 1e-4 -tf 5e-2 -vs 25
// ex16 -s 11 -a 1.0 -k 0.0 -dt 1e-4 -tf 5e-2 -vs 25
// ex16 -s 9 -a 0.5 -k 0.5 -o 4 -dt 1e-4 -tf 2e-2 -vs 25
// ex16 -s 12 -a 0.5 -k 0.5 -o 4 -dt 1e-4 -tf 2e-2 -vs 25
// ex16 -s 10 -dt 1.0e-4 -tf 4.0e-2 -vs 40
// ex16 -s 13 -dt 1.0e-4 -tf 4.0e-2 -vs 40
// ex16 -m ../../data/fichera-q2.mesh
// ex16 -m ../../data/escher.mesh
// ex16 -m ../../data/beam-tet.mesh -tf 10 -dt 0.1
@@ -37,75 +43,102 @@
using namespace std;
using namespace mfem;
/** After spatial discretization, the conduction model can be written as:
/** After spatial discretization, the conduction model is expressed as
*
* du/dt = M^{-1}(-Ku)
* M du/dt = - K(u) u
*
* where u is the vector representing the temperature, M is the mass matrix,
* and K is the diffusion operator with diffusivity depending on u:
* and K(u) is the diffusion operator with diffusivity depending on u:
* (\kappa + \alpha u).
*
* Class ConductionOperator represents the right-hand side of the above ODE.
* Class ConductionOperatorOperator represents the above ODE operator in the
* general form F(u, k, t) = G(u, t) where
*
* 1. F(u, du/dt, t) = du/dt (ODE is expressed in EXPLICIT form)
* G(u, t) = - inv(M) K(u) u
* 2. F(u, du/dt, t) = M du/dt (ODE is expressed in IMPLICIT form)
* G(u, t) = - K(u) u
*/
class ConductionOperator : public TimeDependentOperator
{
protected:
FiniteElementSpace &fespace;
Array<int> ess_tdof_list; // this list remains empty for pure Neumann b.c.
BilinearForm *M;
BilinearForm *K;
BilinearForm M;
SparseMatrix Mmat;
SparseMatrix Mmat, Kmat;
SparseMatrix *T; // T = M + dt K
const real_t alpha, kappa;
std::unique_ptr<BilinearForm> K;
SparseMatrix Kmat;
std::unique_ptr<SparseMatrix> T; // T = M + gam K(u)
CGSolver M_solver; // Krylov solver for inverting the mass matrix M
DSmoother M_prec; // Preconditioner for the mass matrix M
CGSolver T_solver; // Implicit solver for T = M + dt K
CGSolver T_solver; // Implicit solver for T = M + gam K(u)
DSmoother T_prec; // Preconditioner for the implicit solver
double alpha, kappa;
mutable Vector z; // auxiliary vector
public:
ConductionOperator(FiniteElementSpace &f, double alpha, double kappa,
const Vector &u);
virtual void Mult(const Vector &u, Vector &du_dt) const;
ConductionOperator(FiniteElementSpace &f, const real_t alpha,
const real_t kappa, const Vector &u,
const Type &ode_expression_type);
/** Solve the Backward-Euler equation: k = f(u + dt*k, t), for the unknown k.
This is the only requirement for high-order SDIRK implicit integration.*/
virtual void ImplicitSolve(const double dt, const Vector &u, Vector &k);
// Compute K(u_n) for use as an approximation in - K(u) u
void SetConductionTensor(const Vector &u);
/// Custom Jacobian system solver for the SUNDIALS time integrators.
/** For the ODE system represented by ConductionOperator
/** Compute G(u, t) as defined in the IMPLICIT expression form of the ODE
operator, i.e., @a v = - K(u_n) @a u. Note that K(u_n) is an
approximation to K(u). */
void ExplicitMult(const Vector &u, Vector &v) const override;
M du/dt = -K(u),
/** Solve for k in F(u, k, t) = G(u, t) for either EXPLICIT or IMPLICIT
expression forms of the ODE operator, i.e., @a k = - inv(M) K(u_n) @a u.
Note that K(u_n) is an approximation to K(u). */
void Mult(const Vector &u, Vector &k) const override;
this class facilitates the solution of linear systems of the form
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t) for either EXPLICIT
or IMPLICIT expression forms of the ODE operator, i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u . Note that K(u_n) is an
approximation to K(u). */
void ImplicitSolve(const real_t gam, const Vector &u, Vector &k) override;
(M + γK) y = M b,
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F for
either EXPLICIT or IMPLICIT expression forms of the ODE operator, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override;
for given b, u (not used), and γ = GetTimeStep(). */
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing either
1. @a r = G - F = inv(M) f(u) - k (EXPLICIT expression form)
1. @a r = G - F = f(u) - M k (IMPLICIT expression form)
*/
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
/** Setup the system (M + dt K) x = M b. This method is used by the implicit
SUNDIALS solvers. */
virtual int SUNImplicitSetup(const Vector &x, const Vector &fx,
int jok, int *jcur, double gamma);
int SUNMassSetup() override;
/** Solve the system (M + dt K) x = M b. This method is used by the implicit
SUNDIALS solvers. */
virtual int SUNImplicitSolve(const Vector &b, Vector &x, double tol);
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override;
/// Update the diffusion BilinearForm K using the given true-dof vector `u`.
void SetParameters(const Vector &u);
virtual ~ConductionOperator();
int SUNMassMult(const Vector &x, Vector &v) override;
};
double InitialTemperature(const Vector &x);
real_t InitialTemperature(const Vector &x)
{
if (x.Norml2() < 0.5)
{
return 2.0;
}
else
{
return 1.0;
}
}
int main(int argc, char *argv[])
{
@@ -117,16 +150,16 @@ int main(int argc, char *argv[])
int ref_levels = 2;
int order = 2;
int ode_solver_type = 9; // CVODE implicit BDF
double t_final = 0.5;
double dt = 1.0e-2;
double alpha = 1.0e-2;
double kappa = 0.5;
real_t t_final = 0.5;
real_t dt = 1.0e-2;
real_t alpha = 1.0e-2;
real_t kappa = 0.5;
bool visualization = true;
bool visit = false;
int vis_steps = 5;
// Relative and absolute tolerances for CVODE and ARKODE.
const double reltol = 1e-4, abstol = 1e-4;
const real_t reltol = 1e-4, abstol = 1e-4;
int precision = 8;
cout.precision(precision);
@@ -151,7 +184,10 @@ int main(int argc, char *argv[])
"9 - CVODE (implicit BDF),\n\t"
"10 - ARKODE (default explicit),\n\t"
"11 - ARKODE (explicit Fehlberg-6-4-5),\n\t"
"12 - ARKODE (default impicit).");
"12 - ARKODE (default implicit),\n\t"
"13 - ARKODE (default explicit with MFEM mass solve),\n\t"
"14 - ARKODE (explicit Fehlberg-6-4-5 with MFEM mass solve),\n\t"
"15 - ARKODE (default implicit with MFEM mass solve).");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
@@ -174,16 +210,13 @@ int main(int argc, char *argv[])
args.PrintUsage(cout);
return 1;
}
if (ode_solver_type < 1 || ode_solver_type > 12)
{
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
return 3;
}
args.PrintOptions(cout);
bool use_mass_solver = ode_solver_type >= 13;
// 2. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral and hexahedral meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
std::unique_ptr<Mesh> mesh(new Mesh(mesh_file, 1, 1));
int dim = mesh->Dimension();
// 3. Refine the mesh to increase the resolution. In this example we do
@@ -197,7 +230,7 @@ int main(int argc, char *argv[])
// 4. Define the vector finite element space representing the current and the
// initial temperature, u_ref.
H1_FECollection fe_coll(order, dim);
FiniteElementSpace fespace(mesh, &fe_coll);
FiniteElementSpace fespace(mesh.get(), &fe_coll);
int fe_size = fespace.GetTrueVSize();
cout << "Number of temperature unknowns: " << fe_size << endl;
@@ -211,8 +244,17 @@ int main(int argc, char *argv[])
Vector u;
u_gf.GetTrueDofs(u);
// 6. Initialize the conduction operator and the visualization.
ConductionOperator oper(fespace, alpha, kappa, u);
// 6. Initialize the conduction ODE operator and the visualization.
ConductionOperator::Type ode_expression_type;
if (use_mass_solver)
{
ode_expression_type = ConductionOperator::Type::IMPLICIT;
}
else
{
ode_expression_type = ConductionOperator::Type::EXPLICIT;
}
ConductionOperator oper(fespace, alpha, kappa, u, ode_expression_type);
u_gf.SetFromTrueDofs(u);
{
@@ -224,7 +266,7 @@ int main(int argc, char *argv[])
u_gf.Save(osol);
}
VisItDataCollection visit_dc("Example16", mesh);
VisItDataCollection visit_dc("Example16", mesh.get());
visit_dc.RegisterField("temperature", &u_gf);
if (visit)
{
@@ -258,52 +300,75 @@ int main(int argc, char *argv[])
}
// 7. Define the ODE solver used for time integration.
double t = 0.0;
ODESolver *ode_solver = NULL;
CVODESolver *cvode = NULL;
ARKStepSolver *arkode = NULL;
real_t t = 0.0;
std::unique_ptr<ODESolver> ode_solver;
switch (ode_solver_type)
{
// MFEM explicit methods
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(0.5); break; // midpoint method
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 1: ode_solver = std::make_unique<ForwardEulerSolver>(); break;
case 2: ode_solver = std::make_unique<RK2Solver>(0.5); break; // midpoint method
case 3: ode_solver = std::make_unique<RK3SSPSolver>(); break;
case 4: ode_solver = std::make_unique<RK4Solver>(); break;
// MFEM implicit L-stable methods
case 5: ode_solver = new BackwardEulerSolver; break;
case 6: ode_solver = new SDIRK23Solver(2); break;
case 7: ode_solver = new SDIRK33Solver; break;
case 5: ode_solver = std::make_unique<BackwardEulerSolver>(); break;
case 6: ode_solver = std::make_unique<SDIRK23Solver>(2); break;
case 7: ode_solver = std::make_unique<SDIRK33Solver>(); break;
// CVODE
case 8:
cvode = new CVODESolver(CV_ADAMS);
cvode->Init(oper);
cvode->SetSStolerances(reltol, abstol);
cvode->SetMaxStep(dt);
ode_solver = cvode; break;
case 9:
cvode = new CVODESolver(CV_BDF);
{
int cvode_solver_type;
if (ode_solver_type == 8)
{
cvode_solver_type = CV_ADAMS;
}
else
{
cvode_solver_type = CV_BDF;
}
std::unique_ptr<CVODESolver> cvode(new CVODESolver(cvode_solver_type));
cvode->Init(oper);
cvode->SetSStolerances(reltol, abstol);
cvode->SetMaxStep(dt);
ode_solver = cvode; break;
ode_solver = std::move(cvode);
break;
}
// ARKODE
case 10:
case 11:
arkode = new ARKStepSolver(ARKStepSolver::EXPLICIT);
case 12:
case 13:
case 14:
case 15:
{
ARKStepSolver::Type arkode_solver_type;
if (ode_solver_type == 12 || ode_solver_type == 15)
{
arkode_solver_type = ARKStepSolver::IMPLICIT;
}
else
{
arkode_solver_type = ARKStepSolver::EXPLICIT;
}
std::unique_ptr<ARKStepSolver> arkode(
new ARKStepSolver(arkode_solver_type));
arkode->Init(oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
if (ode_solver_type == 11)
if (ode_solver_type == 11 || ode_solver_type == 14)
{
arkode->SetERKTableNum(ARKODE_FEHLBERG_13_7_8);
}
ode_solver = arkode; break;
case 12:
arkode = new ARKStepSolver(ARKStepSolver::IMPLICIT);
arkode->Init(oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
ode_solver = arkode; break;
if (use_mass_solver)
{
arkode->UseMFEMMassLinearSolver(SUNFALSE);
}
ode_solver = std::move(arkode);
break;
}
default:
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
return 3;
}
// Initialize MFEM integrators, SUNDIALS integrators are initialized above
@@ -311,8 +376,14 @@ int main(int argc, char *argv[])
// Since we want to update the diffusion coefficient after every time step,
// we need to use the "one-step" mode of the SUNDIALS solvers.
if (cvode) { cvode->SetStepMode(CV_ONE_STEP); }
if (arkode) { arkode->SetStepMode(ARK_ONE_STEP); }
if (CVODESolver* cvode = dynamic_cast<CVODESolver*>(ode_solver.get()))
{
cvode->SetStepMode(CV_ONE_STEP);
}
else if (ARKStepSolver* arkode = dynamic_cast<ARKStepSolver*>(ode_solver.get()))
{
arkode->SetStepMode(ARK_ONE_STEP);
}
// 8. Perform time-integration (looping over the time iterations, ti, with a
// time-step dt).
@@ -323,7 +394,7 @@ int main(int argc, char *argv[])
bool last_step = false;
for (int ti = 1; !last_step; ti++)
{
double dt_real = min(dt, t_final - t);
real_t dt_real = min(dt, t_final - t);
// Note that since we are using the "one-step" mode of the SUNDIALS
// solvers, they will, generally, step over the final time and will not
@@ -337,8 +408,14 @@ int main(int argc, char *argv[])
if (last_step || (ti % vis_steps) == 0)
{
cout << "step " << ti << ", t = " << t << endl;
if (cvode) { cvode->PrintInfo(); }
if (arkode) { arkode->PrintInfo(); }
if (CVODESolver* cvode = dynamic_cast<CVODESolver*>(ode_solver.get()))
{
cvode->PrintInfo();
}
else if (ARKStepSolver* arkode = dynamic_cast<ARKStepSolver*>(ode_solver.get()))
{
arkode->PrintInfo();
}
u_gf.SetFromTrueDofs(u);
if (visualization)
@@ -353,137 +430,153 @@ int main(int argc, char *argv[])
visit_dc.Save();
}
}
oper.SetParameters(u);
oper.SetConductionTensor(u);
}
tic_toc.Stop();
cout << "Done, " << tic_toc.RealTime() << "s." << endl;
// 9. Save the final solution. This output can be viewed later using GLVis:
// "glvis -m ex16.mesh -g ex16-final.gf".
{
ofstream osol("ex16-final.gf");
osol.precision(precision);
u_gf.Save(osol);
}
// 10. Free the used memory.
delete ode_solver;
delete mesh;
u_gf.Save("ex16-final.gf", precision);
return 0;
}
ConductionOperator::ConductionOperator(FiniteElementSpace &f, double al,
double kap, const Vector &u)
: TimeDependentOperator(f.GetTrueVSize(), 0.0), fespace(f), M(NULL), K(NULL),
T(NULL), z(height)
ConductionOperator::ConductionOperator(FiniteElementSpace &fes,
const real_t alpha, const real_t kappa,
const Vector &u,
const Type &ode_expression_type)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0, ode_expression_type),
fespace(fes), alpha(alpha), kappa(kappa), M(&fespace), z(height)
{
const double rel_tol = 1e-8;
// specify a relative tolerance for all solves with MFEM integrators
const real_t rel_tol = 1e-8;
M = new BilinearForm(&fespace);
M->AddDomainIntegrator(new MassIntegrator());
M->Assemble();
M->FormSystemMatrix(ess_tdof_list, Mmat);
M.AddDomainIntegrator(new MassIntegrator());
M.Assemble();
M.FormSystemMatrix(ess_tdof_list, Mmat);
M_solver.iterative_mode = false;
M_solver.SetRelTol(rel_tol);
M_solver.SetRelTol(rel_tol); // will be overwritten with SUNDIALS integrators
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(50);
M_solver.SetPrintLevel(0);
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(Mmat);
alpha = al;
kappa = kap;
T_solver.iterative_mode = false;
T_solver.SetRelTol(rel_tol);
T_solver.SetRelTol(rel_tol); // will be overwritten with SUNDIALS integrators
T_solver.SetAbsTol(0.0);
T_solver.SetMaxIter(100);
T_solver.SetPrintLevel(0);
T_solver.SetPreconditioner(T_prec);
SetParameters(u);
SetConductionTensor(u);
}
void ConductionOperator::Mult(const Vector &u, Vector &du_dt) const
{
// Compute:
// du_dt = M^{-1}*-K(u)
// for du_dt
Kmat.Mult(u, z);
z.Neg(); // z = -z
M_solver.Mult(z, du_dt);
}
void ConductionOperator::ImplicitSolve(const double dt,
const Vector &u, Vector &du_dt)
{
// Solve the equation:
// du_dt = M^{-1}*[-K(u + dt*du_dt)]
// for du_dt
if (T) { delete T; }
T = Add(1.0, Mmat, dt, Kmat);
T_solver.SetOperator(*T);
Kmat.Mult(u, z);
z.Neg();
T_solver.Mult(z, du_dt);
}
void ConductionOperator::SetParameters(const Vector &u)
void ConductionOperator::SetConductionTensor(const Vector &u)
{
// Compute K(u_n).
GridFunction u_alpha_gf(&fespace);
u_alpha_gf.SetFromTrueDofs(u);
for (int i = 0; i < u_alpha_gf.Size(); i++)
{
u_alpha_gf(i) = kappa + alpha*u_alpha_gf(i);
}
delete K;
K = new BilinearForm(&fespace);
GridFunctionCoefficient u_coeff(&u_alpha_gf);
K = std::make_unique<BilinearForm>(&fespace);
K->AddDomainIntegrator(new DiffusionIntegrator(u_coeff));
K->Assemble();
K->FormSystemMatrix(ess_tdof_list, Kmat);
}
int ConductionOperator::SUNImplicitSetup(const Vector &x,
const Vector &fx, int jok, int *jcur,
double gamma)
void ConductionOperator::ExplicitMult(const Vector &u, Vector &v) const
{
// Setup the ODE Jacobian T = M + gamma K.
if (T) { delete T; }
T = Add(1.0, Mmat, gamma, Kmat);
// Compute - K(u_n) u.
Kmat.Mult(u, v);
v.Neg();
}
void ConductionOperator::Mult(const Vector &u, Vector &k) const
{
// Compute - inv(M) K(u_n) u.
ExplicitMult(u, z);
M_solver.Mult(z, k);
}
void ConductionOperator::ImplicitSolve(const real_t gam, const Vector &u,
Vector &k)
{
// Solve for k in M k = - K(u_n) [u + gam*k].
ExplicitMult(u, z);
T = std::unique_ptr<SparseMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
*jcur = 1;
return (0);
T_solver.Mult(z, k);
}
int ConductionOperator::SUNImplicitSolve(const Vector &b, Vector &x, double tol)
int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
int jok, int *jcur, real_t gam)
{
// Solve the system A x = z => (M - gamma K) x = M b.
Mmat.Mult(b, z);
T_solver.Mult(z, x);
return (0);
// Compute T = M + gamma K(u_n).
T = std::unique_ptr<SparseMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
*jcur = SUNTRUE; // this should eventually only be set true if K(u) is used
return SUNLS_SUCCESS;
}
ConductionOperator::~ConductionOperator()
int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
real_t tol)
{
delete T;
delete M;
delete K;
}
double InitialTemperature(const Vector &x)
{
if (x.Norml2() < 0.5)
// Solve the system [M + gamma K(u_n)] dk = - K(u_n) u - M k.
// What value r is providing depends on the ODE expression form:
// EXPLICIT form: r = -inv(M) K(u_n) u - k
// IMPLICIT form: r = -K(u_n) u - M k
T_solver.SetRelTol(tol);
if (isExplicit())
{
return 2.0;
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
}
else
{
return 1.0;
T_solver.Mult(r, dk);
}
if (T_solver.GetConverged())
{
return SUNLS_SUCCESS;
}
else
{
return SUNLS_CONV_FAIL;
}
}
int ConductionOperator::SUNMassSetup()
{
// Do nothing b/c mass solver was setup in constructor.
return SUNLS_SUCCESS;
}
int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
{
// Solve the system M x = b.
M_solver.SetRelTol(tol);
M_solver.Mult(b, x);
if (M_solver.GetConverged())
{
return SUNLS_SUCCESS;
}
else
{
return SUNLS_CONV_FAIL;
}
}
int ConductionOperator::SUNMassMult(const Vector &x, Vector &v)
{
// Compute M x.
Mmat.Mult(x, v);
return SUNLS_SUCCESS;
}
+285 -188
View File
@@ -1,16 +1,22 @@
// MFEM Example 16 - Parallel Version
// SUNDIALS Modification
//
// Compile with: make ex16p
// Compile with:
// make ex16p (GNU make)
// make sundials_ex16p (CMake)
//
// Sample runs:
// mpirun -np 4 ex16p
// mpirun -np 4 ex16p -m ../../data/inline-tri.mesh
// mpirun -np 4 ex16p -m ../../data/disc-nurbs.mesh -tf 2
// mpirun -np 4 ex16p -s 12 -a 0.0 -k 1.0
// mpirun -np 4 ex16p -s 15 -a 0.0 -k 1.0
// mpirun -np 4 ex16p -s 8 -a 1.0 -k 0.0 -dt 4e-6 -tf 2e-2 -vs 50
// mpirun -np 4 ex16p -s 11 -a 1.0 -k 0.0 -dt 4e-6 -tf 2e-2 -vs 50
// mpirun -np 8 ex16p -s 9 -a 0.5 -k 0.5 -o 4 -dt 8e-6 -tf 2e-2 -vs 50
// mpirun -np 8 ex16p -s 12 -a 0.5 -k 0.5 -o 4 -dt 8e-6 -tf 2e-2 -vs 50
// mpirun -np 4 ex16p -s 10 -dt 2.0e-4 -tf 4.0e-2
// mpirun -np 4 ex16p -s 13 -dt 2.0e-4 -tf 4.0e-2
// mpirun -np 16 ex16p -m ../../data/fichera-q2.mesh
// mpirun -np 16 ex16p -m ../../data/escher-p2.mesh
// mpirun -np 8 ex16p -m ../../data/beam-tet.mesh -tf 10 -dt 0.1
@@ -38,66 +44,102 @@
using namespace std;
using namespace mfem;
/** After spatial discretization, the conduction model can be written as:
/** After spatial discretization, the conduction model is expressed as
*
* du/dt = M^{-1}(-Ku)
* M du/dt = - K(u) u
*
* where u is the vector representing the temperature, M is the mass matrix,
* and K is the diffusion operator with diffusivity depending on u:
* and K(u) is the diffusion operator with diffusivity depending on u:
* (\kappa + \alpha u).
*
* Class ConductionOperator represents the right-hand side of the above ODE.
* Class ConductionOperatorOperator represents the above ODE operator in the
* general form F(u, k, t) = G(u, t) where either
*
* 1. F(u, du/dt, t) = du/dt (ODE is expressed in EXPLICIT form)
* G(u, t) = - inv(M) K(u) u
* 2. F(u, du/dt, t) = M du/dt (ODE is expressed in IMPLICIT form)
* G(u, t) = - K(u) u
*/
class ConductionOperator : public TimeDependentOperator
{
protected:
ParFiniteElementSpace &fespace;
Array<int> ess_tdof_list; // this list remains empty for pure Neumann b.c.
ParBilinearForm *M;
ParBilinearForm *K;
ParBilinearForm M;
HypreParMatrix Mmat;
const real_t alpha, kappa;
std::unique_ptr<BilinearForm> K;
HypreParMatrix Kmat;
HypreParMatrix *T; // T = M + dt K
double current_dt;
CGSolver M_solver; // Krylov solver for inverting the mass matrix M
HypreSmoother M_prec; // Preconditioner for the mass matrix M
std::unique_ptr<HypreParMatrix> T; // T = M + gam K(u)
CGSolver T_solver; // Implicit solver for T = M + dt K
HypreSmoother T_prec; // Preconditioner for the implicit solver
CGSolver M_solver; // Krylov solver for inverting the mass matrix M
HypreSmoother M_prec; // Preconditioner for the mass matrix M
double alpha, kappa;
CGSolver T_solver; // Implicit solver for T = M + gam K(u)
HypreSmoother T_prec; // Preconditioner for the implicit solver
mutable Vector z; // auxiliary vector
public:
ConductionOperator(ParFiniteElementSpace &f, double alpha, double kappa,
const Vector &u);
virtual void Mult(const Vector &u, Vector &du_dt) const;
ConductionOperator(ParFiniteElementSpace &f, const real_t alpha,
const real_t kappa, const Vector &u,
const Type &ode_expression_type);
/** Solve the Backward-Euler equation: k = f(u + dt*k, t), for the unknown k.
This is the only requirement for high-order SDIRK implicit integration.*/
virtual void ImplicitSolve(const double dt, const Vector &u, Vector &k);
// Compute K(u_n) for use as an approximation in - K(u) u
void SetConductionTensor(const Vector &u);
/** Setup the system (M + dt K) x = M b. This method is used by the implicit
SUNDIALS solvers. */
virtual int SUNImplicitSetup(const Vector &x, const Vector &fx,
int jok, int *jcur, double gamma);
/** Compute G(u, t) as defined in the IMPLICIT expression form of the ODE
operator, i.e., @a v = - K(u_n) @a u. Note that K(u_n) is an
approximation to K(u). */
void ExplicitMult(const Vector &u, Vector &v) const override;
/** Solve the system (M + dt K) x = M b. This method is used by the implicit
SUNDIALS solvers. */
virtual int SUNImplicitSolve(const Vector &b, Vector &x, double tol);
/** Solve for k in F(u, k, t) = G(u, t) for either EXPLICIT or IMPLICIT
expression forms of the ODE operator, i.e., @a k = - inv(M) K(u_n) @a u.
Note that K(u_n) is an approximation to K(u). */
void Mult(const Vector &u, Vector &k) const override;
/// Update the diffusion BilinearForm K using the given true-dof vector `u`.
void SetParameters(const Vector &u);
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t) for either EXPLICIT
or IMPLICIT expression forms of the ODE operator, i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u . Note that K(u_n) is an
approximation to K(u). */
void ImplicitSolve(const real_t gam, const Vector &u, Vector &k) override;
virtual ~ConductionOperator();
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F for
either EXPLICIT or IMPLICIT expression forms of the ODE operator, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override;
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing either
1. @a r = G - F = inv(M) f(u) - k (EXPLICIT expression form)
1. @a r = G - F = f(u) - M k (IMPLICIT expression form)
*/
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
int SUNMassSetup() override;
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override;
int SUNMassMult(const Vector &x, Vector &v) override;
};
double InitialTemperature(const Vector &x);
real_t InitialTemperature(const Vector &x)
{
if (x.Norml2() < 0.5)
{
return 2.0;
}
else
{
return 1.0;
}
}
int main(int argc, char *argv[])
{
@@ -114,16 +156,16 @@ int main(int argc, char *argv[])
int par_ref_levels = 1;
int order = 2;
int ode_solver_type = 9; // CVODE implicit BDF
double t_final = 0.5;
double dt = 1.0e-2;
double alpha = 1.0e-2;
double kappa = 0.5;
real_t t_final = 0.5;
real_t dt = 1.0e-2;
real_t alpha = 1.0e-2;
real_t kappa = 0.5;
bool visualization = true;
bool visit = false;
int vis_steps = 5;
// Relative and absolute tolerances for CVODE and ARKODE.
const double reltol = 1e-4, abstol = 1e-4;
const real_t reltol = 1e-4, abstol = 1e-4;
int precision = 8;
cout.precision(precision);
@@ -150,7 +192,10 @@ int main(int argc, char *argv[])
"9 - CVODE (implicit BDF),\n\t"
"10 - ARKODE (default explicit),\n\t"
"11 - ARKODE (explicit Fehlberg-6-4-5),\n\t"
"12 - ARKODE (default impicit).");
"12 - ARKODE (default implicit),\n\t"
"13 - ARKODE (default explicit with MFEM mass solve),\n\t"
"14 - ARKODE (explicit Fehlberg-6-4-5 with MFEM mass solve),\n\t"
"15 - ARKODE (default implicit with MFEM mass solve).");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
@@ -174,40 +219,33 @@ int main(int argc, char *argv[])
return 1;
}
if (myid == 0)
if (Mpi::Root())
{
args.PrintOptions(cout);
}
// check for valid ODE solver option
if (ode_solver_type < 1 || ode_solver_type > 12)
{
if (myid == 0)
{
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
}
return 1;
}
bool use_mass_solver = ode_solver_type >= 13;
// 3. Read the serial mesh from the given mesh file on all processors. We can
// 3. Define a parallel mesh by a partitioning of a serial mesh. Read the
// serial mesh from the given mesh file on all processors. We can
// handle triangular, quadrilateral, tetrahedral and hexahedral meshes
// with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the mesh in serial to increase the resolution. In this example
// we do 'ser_ref_levels' of uniform refinement, where 'ser_ref_levels' is
// a command-line parameter.
for (int lev = 0; lev < ser_ref_levels; lev++)
std::unique_ptr<ParMesh> pmesh;
{
mesh->UniformRefinement();
}
std::unique_ptr<Mesh> mesh(new Mesh(mesh_file, 1, 1));
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
// 4. Refine the mesh in serial to increase the resolution. In this example
// we do 'ser_ref_levels' of uniform refinement, where 'ser_ref_levels' is
// a command-line parameter.
for (int lev = 0; lev < ser_ref_levels; lev++)
{
mesh->UniformRefinement();
}
// 5. Refine this mesh further in parallel to increase the resolution.
// Once the parallel mesh is defined, the serial mesh can be deleted.
pmesh = std::make_unique<ParMesh>(MPI_COMM_WORLD, *mesh);
}
for (int lev = 0; lev < par_ref_levels; lev++)
{
pmesh->UniformRefinement();
@@ -215,8 +253,9 @@ int main(int argc, char *argv[])
// 6. Define the vector finite element space representing the current and the
// initial temperature, u_ref.
int dim = pmesh->Dimension();
H1_FECollection fe_coll(order, dim);
ParFiniteElementSpace fespace(pmesh, &fe_coll);
ParFiniteElementSpace fespace(pmesh.get(), &fe_coll);
int fe_size = fespace.GlobalTrueVSize();
if (myid == 0)
@@ -233,8 +272,17 @@ int main(int argc, char *argv[])
Vector u;
u_gf.GetTrueDofs(u);
// 8. Initialize the conduction operator and the VisIt visualization.
ConductionOperator oper(fespace, alpha, kappa, u);
// 8. Initialize the conduction ODE operator and the visualization.
ConductionOperator::Type ode_expression_type;
if (use_mass_solver)
{
ode_expression_type = ConductionOperator::Type::IMPLICIT;
}
else
{
ode_expression_type = ConductionOperator::Type::EXPLICIT;
}
ConductionOperator oper(fespace, alpha, kappa, u, ode_expression_type);
u_gf.SetFromTrueDofs(u);
{
@@ -249,7 +297,7 @@ int main(int argc, char *argv[])
u_gf.Save(osol);
}
VisItDataCollection visit_dc("Example16-Parallel", pmesh);
VisItDataCollection visit_dc("Example16-Parallel", pmesh.get());
visit_dc.RegisterField("temperature", &u_gf);
if (visit)
{
@@ -293,52 +341,76 @@ int main(int argc, char *argv[])
}
// 9. Define the ODE solver used for time integration.
double t = 0.0;
ODESolver *ode_solver = NULL;
CVODESolver *cvode = NULL;
ARKStepSolver *arkode = NULL;
real_t t = 0.0;
std::unique_ptr<ODESolver> ode_solver;
switch (ode_solver_type)
{
// MFEM explicit methods
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(0.5); break; // midpoint method
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 1: ode_solver = std::make_unique<ForwardEulerSolver>(); break;
case 2: ode_solver = std::make_unique<RK2Solver>(0.5); break; // midpoint method
case 3: ode_solver = std::make_unique<RK3SSPSolver>(); break;
case 4: ode_solver = std::make_unique<RK4Solver>(); break;
// MFEM implicit L-stable methods
case 5: ode_solver = new BackwardEulerSolver; break;
case 6: ode_solver = new SDIRK23Solver(2); break;
case 7: ode_solver = new SDIRK33Solver; break;
case 5: ode_solver = std::make_unique<BackwardEulerSolver>(); break;
case 6: ode_solver = std::make_unique<SDIRK23Solver>(2); break;
case 7: ode_solver = std::make_unique<SDIRK33Solver>(); break;
// CVODE
case 8:
cvode = new CVODESolver(MPI_COMM_WORLD, CV_ADAMS);
cvode->Init(oper);
cvode->SetSStolerances(reltol, abstol);
cvode->SetMaxStep(dt);
ode_solver = cvode; break;
case 9:
cvode = new CVODESolver(MPI_COMM_WORLD, CV_BDF);
{
int cvode_solver_type;
if (ode_solver_type == 8)
{
cvode_solver_type = CV_ADAMS;
}
else
{
cvode_solver_type = CV_BDF;
}
std::unique_ptr<CVODESolver> cvode(
new CVODESolver(MPI_COMM_WORLD, cvode_solver_type));
cvode->Init(oper);
cvode->SetSStolerances(reltol, abstol);
cvode->SetMaxStep(dt);
ode_solver = cvode; break;
ode_solver = std::move(cvode);
break;
}
// ARKODE
case 10:
case 11:
arkode = new ARKStepSolver(MPI_COMM_WORLD, ARKStepSolver::EXPLICIT);
case 12:
case 13:
case 14:
case 15:
{
ARKStepSolver::Type arkode_solver_type;
if (ode_solver_type == 12 || ode_solver_type == 15)
{
arkode_solver_type = ARKStepSolver::IMPLICIT;
}
else
{
arkode_solver_type = ARKStepSolver::EXPLICIT;
}
std::unique_ptr<ARKStepSolver> arkode(
new ARKStepSolver(MPI_COMM_WORLD, arkode_solver_type));
arkode->Init(oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
if (ode_solver_type == 11)
if (ode_solver_type == 11 || ode_solver_type == 14)
{
arkode->SetERKTableNum(ARKODE_FEHLBERG_13_7_8);
}
ode_solver = arkode; break;
case 12:
arkode = new ARKStepSolver(MPI_COMM_WORLD, ARKStepSolver::IMPLICIT);
arkode->Init(oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
ode_solver = arkode; break;
if (use_mass_solver)
{
arkode->UseMFEMMassLinearSolver(SUNFALSE);
}
ode_solver = std::move(arkode);
break;
}
default:
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
return 3;
}
// Initialize MFEM integrators, SUNDIALS integrators are initialized above
@@ -346,12 +418,18 @@ int main(int argc, char *argv[])
// Since we want to update the diffusion coefficient after every time step,
// we need to use the "one-step" mode of the SUNDIALS solvers.
if (cvode) { cvode->SetStepMode(CV_ONE_STEP); }
if (arkode) { arkode->SetStepMode(ARK_ONE_STEP); }
if (CVODESolver* cvode = dynamic_cast<CVODESolver*>(ode_solver.get()))
{
cvode->SetStepMode(CV_ONE_STEP);
}
else if (ARKStepSolver* arkode = dynamic_cast<ARKStepSolver*>(ode_solver.get()))
{
arkode->SetStepMode(ARK_ONE_STEP);
}
// 10. Perform time-integration (looping over the time iterations, ti, with a
// time-step dt).
if (myid == 0)
if (Mpi::Root())
{
cout << "Integrating the ODE ..." << endl;
}
@@ -361,7 +439,7 @@ int main(int argc, char *argv[])
bool last_step = false;
for (int ti = 1; !last_step; ti++)
{
double dt_real = min(dt, t_final - t);
real_t dt_real = min(dt, t_final - t);
// Note that since we are using the "one-step" mode of the SUNDIALS
// solvers, they will, generally, step over the final time and will not
@@ -377,8 +455,14 @@ int main(int argc, char *argv[])
if (myid == 0)
{
cout << "step " << ti << ", t = " << t << endl;
if (cvode) { cvode->PrintInfo(); }
if (arkode) { arkode->PrintInfo(); }
if (CVODESolver* cvode = dynamic_cast<CVODESolver*>(ode_solver.get()))
{
cvode->PrintInfo();
}
else if (ARKStepSolver* arkode = dynamic_cast<ARKStepSolver*>(ode_solver.get()))
{
arkode->PrintInfo();
}
}
u_gf.SetFromTrueDofs(u);
@@ -395,46 +479,38 @@ int main(int argc, char *argv[])
visit_dc.Save();
}
}
oper.SetParameters(u);
oper.SetConductionTensor(u);
}
tic_toc.Stop();
if (myid == 0)
if (Mpi::Root())
{
cout << "Done, " << tic_toc.RealTime() << "s." << endl;
}
// 11. Save the final solution in parallel. This output can be viewed later
// using GLVis: "glvis -np <np> -m ex16-mesh -g ex16-final".
{
ostringstream sol_name;
sol_name << "ex16-final." << setfill('0') << setw(6) << myid;
ofstream osol(sol_name.str().c_str());
osol.precision(precision);
u_gf.Save(osol);
}
// 12. Free the used memory.
delete ode_solver;
delete pmesh;
u_gf.Save("ex16-final", precision);
return 0;
}
ConductionOperator::ConductionOperator(ParFiniteElementSpace &f, double al,
double kap, const Vector &u)
: TimeDependentOperator(f.GetTrueVSize(), 0.0), fespace(f), M(NULL), K(NULL),
T(NULL),
M_solver(f.GetComm()), T_solver(f.GetComm()), z(height)
ConductionOperator::ConductionOperator(ParFiniteElementSpace &fes,
const real_t alpha, const real_t kappa,
const Vector &u,
const Type &ode_expression_type)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0, ode_expression_type),
fespace(fes), alpha(alpha), kappa(kappa), M(&fespace),
M_solver(fes.GetComm()), T_solver(fes.GetComm()), z(height)
{
const double rel_tol = 1e-8;
// specify a relative tolerance for all solves with MFEM integrators
const real_t rel_tol = 1e-8;
M = new ParBilinearForm(&fespace);
M->AddDomainIntegrator(new MassIntegrator());
M->Assemble(0); // keep sparsity pattern of M and K the same
M->FormSystemMatrix(ess_tdof_list, Mmat);
M.AddDomainIntegrator(new MassIntegrator());
M.Assemble(0); // keep zeros to keep sparsity pattern of M and K the same
M.FormSystemMatrix(ess_tdof_list, Mmat);
M_solver.iterative_mode = false;
M_solver.SetRelTol(rel_tol);
M_solver.SetRelTol(rel_tol); // will be overwritten with SUNDIALS integrators
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
@@ -442,97 +518,118 @@ ConductionOperator::ConductionOperator(ParFiniteElementSpace &f, double al,
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(Mmat);
alpha = al;
kappa = kap;
T_solver.iterative_mode = false;
T_solver.SetRelTol(rel_tol);
T_solver.SetRelTol(rel_tol); // will be overwritten with SUNDIALS integrators
T_solver.SetAbsTol(0.0);
T_solver.SetMaxIter(100);
T_solver.SetPrintLevel(0);
T_solver.SetPreconditioner(T_prec);
SetParameters(u);
SetConductionTensor(u);
}
void ConductionOperator::Mult(const Vector &u, Vector &du_dt) const
{
// Compute:
// du_dt = M^{-1}*-K(u)
// for du_dt
Kmat.Mult(u, z);
z.Neg(); // z = -z
M_solver.Mult(z, du_dt);
}
void ConductionOperator::ImplicitSolve(const double dt,
const Vector &u, Vector &du_dt)
{
// Solve the equation:
// du_dt = M^{-1}*[-K(u + dt*du_dt)]
// for du_dt
if (T) { delete T; }
T = Add(1.0, Mmat, dt, Kmat);
T_solver.SetOperator(*T);
Kmat.Mult(u, z);
z.Neg();
T_solver.Mult(z, du_dt);
}
int ConductionOperator::SUNImplicitSetup(const Vector &x,
const Vector &fx, int jok, int *jcur,
double gamma)
{
// Setup the ODE Jacobian T = M + gamma K.
if (T) { delete T; }
T = Add(1.0, Mmat, gamma, Kmat);
T_solver.SetOperator(*T);
*jcur = 1;
return (0);
}
int ConductionOperator::SUNImplicitSolve(const Vector &b, Vector &x, double tol)
{
// Solve the system A x = z => (M - gamma K) x = M b.
Mmat.Mult(b, z);
T_solver.Mult(z, x);
return (0);
}
void ConductionOperator::SetParameters(const Vector &u)
void ConductionOperator::SetConductionTensor(const Vector &u)
{
// Compute K(u_n).
ParGridFunction u_alpha_gf(&fespace);
u_alpha_gf.SetFromTrueDofs(u);
for (int i = 0; i < u_alpha_gf.Size(); i++)
{
u_alpha_gf(i) = kappa + alpha*u_alpha_gf(i);
}
delete K;
K = new ParBilinearForm(&fespace);
GridFunctionCoefficient u_coeff(&u_alpha_gf);
K = std::make_unique<ParBilinearForm>(&fespace);
K->AddDomainIntegrator(new DiffusionIntegrator(u_coeff));
K->Assemble(0); // keep sparsity pattern of M and K the same
K->Assemble(0); // keep zeros to keep sparsity pattern of M and K the same
K->FormSystemMatrix(ess_tdof_list, Kmat);
}
ConductionOperator::~ConductionOperator()
void ConductionOperator::ExplicitMult(const Vector &u, Vector &v) const
{
delete T;
delete M;
delete K;
// Compute - K(u_n) u.
Kmat.Mult(u, v);
v.Neg();
}
double InitialTemperature(const Vector &x)
void ConductionOperator::Mult(const Vector &u, Vector &k) const
{
if (x.Norml2() < 0.5)
// Compute - inv(M) K(u_n) u.
ExplicitMult(u, z);
M_solver.Mult(z, k);
}
void ConductionOperator::ImplicitSolve(const real_t gam, const Vector &u,
Vector &k)
{
// Solve for k in M k = - K(u_n) [u + gam*k].
ExplicitMult(u, z);
T = std::unique_ptr<HypreParMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
T_solver.Mult(z, k);
}
int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
int jok, int *jcur, real_t gam)
{
// Compute T = M + gamma K(u_n).
T = std::unique_ptr<HypreParMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
*jcur = SUNTRUE; // this should eventually only be set true if K(u) is used
return SUNLS_SUCCESS;
}
int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
real_t tol)
{
// Solve the system [M + gamma K(u_n)] dk = - K(u_n) u - M k.
// What value r is providing depends on the ODE expression form:
// EXPLICIT form: r = -inv(M) K(u_n) u - k
// IMPLICIT form: r = -K(u_n) u - M k
T_solver.SetRelTol(tol);
if (isExplicit())
{
return 2.0;
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
}
else
{
return 1.0;
T_solver.Mult(r, dk);
}
if (T_solver.GetConverged())
{
return SUNLS_SUCCESS;
}
else
{
return SUNLS_CONV_FAIL;
}
}
int ConductionOperator::SUNMassSetup()
{
// Do nothing b/c mass solver was setup in constructor.
return SUNLS_SUCCESS;
}
int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
{
// Solve the system M x = b.
M_solver.SetRelTol(tol);
M_solver.Mult(b, x);
if (M_solver.GetConverged())
{
return SUNLS_SUCCESS;
}
else
{
return SUNLS_CONV_FAIL;
}
}
int ConductionOperator::SUNMassMult(const Vector &x, Vector &v)
{
// Compute M x.
Mmat.Mult(x, v);
return SUNLS_SUCCESS;
}
+3 -1
View File
@@ -1,7 +1,9 @@
// MFEM Example 9
// SUNDIALS Modification
//
// Compile with: make ex9
// Compile with:
// make ex9 (GNU make)
// make sundials_ex9 (CMake)
//
// Sample runs:
// ex9 -m ../../data/periodic-segment.mesh -p 0 -r 2 -s 7 -dt 0.005
+3 -1
View File
@@ -1,7 +1,9 @@
// MFEM Example 9 - Parallel Version
// SUNDIALS Modification
//
// Compile with: make ex9p
// Compile with:
// make ex9p (GNU make)
// make sundials_ex9p (CMake)
//
// Sample runs:
// mpirun -np 4 ex9p -m ../../data/periodic-segment.mesh -p 1 -rp 1 -s 7 -dt 0.0025

Some files were not shown because too many files have changed in this diff Show More