Compare commits

...
829 Commits
Author SHA1 Message Date
bslazarov 7dae996358 modified: defaults.cmake
modified:   ../examples/CMakeLists.txt
	new file:   ../fem/adnonlininteg.cpp
	new file:   ../fem/adnonlininteg.hpp
	modified:   ../linalg/CMakeLists.txt
	new file:   ../linalg/taddensemat.hpp
	new file:   ../linalg/tadvector.hpp
2020-02-04 14:01:11 -08:00
Julian Andrej 417d53096f Add missing files to cmake 2020-01-27 12:57:49 -08:00
Tzanio Kolev 4a8ffc1b97 Merge pull request #1258 from mfem/hotfix-mixedbilinearform
Revert name change in MixedBilinearForm
2020-01-27 07:22:50 -08:00
Tzanio Kolev 1209323bb5 Merge pull request #1242 from mfem/load-dc-bugfix-dev
Memory overlap in MPI_Allreduce [load-dc-bugfix-dev]
2020-01-24 17:46:50 -08:00
Tzanio Kolev 9c0260de59 Merge pull request #1252 from mfem/kinsol-par-print
Change KINSOL behavior to only print on root node [kinsol-par-print]
2020-01-24 17:39:30 -08:00
Julian Andrej 78f0fbd2c4 Revert name change in MixedBilinearForm 2020-01-23 16:48:59 -08:00
Tzanio Kolev 4828e7e2a7 Merge pull request #1229 from mfem/binary_file_test-tk
Travis checks, including for the introduction/modification of a binary file [binary_file_test-tk]
2020-01-23 08:34:47 -08:00
Veselin Dobrev 854c364b48 Merge pull request #1159 from mfem/nlpa-dev
Add PA for nonlinear integrators [nlpa-dev]
2020-01-22 19:05:33 -08:00
Tzanio 4b9f8079b1 minor 2020-01-22 13:49:35 -08:00
Adrien M. Bernede d122132480 Removing unecessary stuff 2020-01-22 13:29:55 -08:00
Adrien M. Bernede 05556fbb1f Travis, add os: linux when using apt 2020-01-22 13:02:19 -08:00
Julian Andrej d2219eb63b Refined CHANGELOG 2020-01-22 11:29:34 -08:00
Tzanio Kolev 5870bae946 Merge pull request #1244 from mfem/binary_file_test-tk2
Proposed changes to #1229 [binary_file_test-tk2]
2020-01-22 11:25:42 -08:00
Julian Andrej 3584066337 Merge branch 'master' into nlpa-dev
Conflicts:
	CHANGELOG
	tests/unit/fem/test_pa_kernels.cpp
2020-01-22 10:36:35 -08:00
Adrien M. Bernede ba8a5965b0 Slight reformat of usage 2020-01-22 10:31:39 -08:00
Adrien M. Bernede 70585a11b2 Minor changes 2020-01-22 10:01:26 -08:00
Tzanio Kolev 093f951ada Merge pull request #1164 from mfem/rectop-bc-dev
Divergence and Gradient integrators as well as PA versions
2020-01-21 18:33:19 -08:00
Tzanio 24d6028f79 Updated CHANGELOG. 2020-01-21 18:29:20 -08:00
Tzanio 25e67d568d Renamed areColumnsSorted() to ColumsAreSorted(). Small comments and typos. 2020-01-21 17:49:58 -08:00
Tzanio Kolev 6ae42d2894 Merge pull request #513 from mfem/adams-bashforth-dev
Adding Adams Bashforth time integrator [adams-bashforth-dev]
2020-01-21 16:33:29 -08:00
Tzanio 4e35956daa make style 2020-01-21 14:51:26 -08:00
Julian Andrej bbd769b402 Added SetPrintLevel to serial case 2020-01-21 14:44:51 -08:00
Julian Andrej 647dc96619 Change KINSOL behavior to only print on root node 2020-01-21 14:34:00 -08:00
Tzanio Kolev 114b743f28 Merge pull request #1215 from intact-solutions/intpoint-indices-dev
Add indices to integration points. [intpoint-indices-dev]
2020-01-20 18:36:04 -08:00
Tzanio 9794864901 actual fix 2020-01-20 18:20:58 -08:00
Tzanio 51eb265d69 Addressed https://github.com/mfem/mfem/pull/1244#discussion_r368773359 2020-01-20 17:49:15 -08:00
Tzanio 24fa1ac093 Final changes to branch-history 2020-01-20 17:18:04 -08:00
Tzanio 5047b0b5b2 Fixes 2020-01-20 16:51:09 -08:00
Tzanio ae1aa05446 Fixes 2020-01-20 16:33:09 -08:00
Tzanio ad736f591f Fixes 2020-01-20 15:45:51 -08:00
Tzanio b048215a06 Proposed changes to #1229. 2020-01-20 12:47:09 -08:00
Tzanio e0934177dd minor 2020-01-18 21:55:39 -08:00
Tzanio 4e1a5465b0 proposed fix for regression error 2020-01-18 21:33:12 -08:00
Tzanio 0804ac2675 minor 2020-01-18 21:25:17 -08:00
Tzanio 839a5a8a5a Small fixes 2020-01-18 21:22:34 -08:00
Veselin Dobrev dac7dca262 Merge pull request #1179 from mfem/get-values-dev
Adding get-values miniapp [get-values-dev]
2020-01-17 20:53:31 -08:00
Tzanio 0e2201a0f9 Made the Travis gitignore check more useful by adding a new make target,
test-noclean, that runs the test without cleaning the output (via clean-exec).
2020-01-17 20:20:54 -08:00
Veselin Dobrev 6a5ed97f53 A few tweaks after the merges from master. 2020-01-17 20:15:47 -08:00
Tzanio f2044ab567 Small changes in git-history 2020-01-17 19:46:27 -08:00
Tzanio 40cea58b36 Small changes in git-history 2020-01-17 19:23:22 -08:00
Tzanio bb42b1ad42 Small changes in git-history 2020-01-17 19:21:55 -08:00
Tzanio e11f685d80 Addressing Adrien's review comments 2020-01-17 18:50:14 -08:00
Tzanio 02f03afddb Merge branch 'master' into binary_file_test-tk 2020-01-17 18:34:32 -08:00
Tzanio 5446ce63e3 Merge branch 'master' into rectop-bc-dev
Conflicts:
	linalg/operator.cpp
2020-01-17 18:33:45 -08:00
Tzanio Kolev ceb59fa5a4 Merge pull request #1234 from mfem/complex-fem-memcheck-dev
Fixing issues uncovered by memcheck [complex-fem-memcheck-dev]
2020-01-17 18:25:21 -08:00
Tzanio Kolev 9f04c09197 Merge pull request #1236 from mfem/miniapps-common-shared
miniapps-common fix for shared builds [miniapps-common-shared]
2020-01-17 18:06:49 -08:00
Tzanio Kolev 2c2d902d45 Merge pull request #1239 from mfem/pfespace-fix
GetProlongationMatrix returns IdentityOperator when NRanks ==1 [pfespace-fix]
2020-01-17 18:04:39 -08:00
Tzanio 045fe4b024 Merge branch 'master' into binary_file_test-tk
Conflicts:
	makefile
2020-01-16 19:22:48 -08:00
Tzanio 69e370ea17 Travis-proof make clean 2020-01-16 19:21:24 -08:00
Tzanio 70441f1189 Merge branch 'master' into pfespace-fix 2020-01-16 17:59:57 -08:00
Tzanio b4a789bb34 Better line breaking 2020-01-16 17:41:46 -08:00
Stowell, Mark L c6ca662a1b The bool datatype in smaller than an MPI_INT so this lead to an overlap in the arguments to MPI_Allreduce 2020-01-16 16:51:15 -08:00
Tzanio 873b1009a1 minor 2020-01-16 16:38:59 -08:00
Tzanio 3e2e0de888 minor 2020-01-16 16:30:47 -08:00
Tzanio d3e1db05dc Wrapped long lines 2020-01-16 16:29:59 -08:00
camierjs 20a63b218a GetProlongationMatrix returns IdentityOperator when NRanks ==1
Add IsIdentityProlongation in operator.hpp
2020-01-16 16:11:51 -08:00
Tzanio Kolev b132457da7 Merge pull request #949 from mfem/nc-tet-dev
Non-conforming tetrahedal refinement [nc-tet-dev]
2020-01-16 16:09:42 -08:00
Adrien M. Bernede f88b262f4d Simplifying the test on line changes 2020-01-16 16:08:31 -08:00
Adrien M. Bernede faee003c31 Typos 2020-01-16 15:18:17 -08:00
Adrien M. Bernede 8cc5d8dc91 Replacing googlecode with github 2020-01-16 15:02:22 -08:00
Adrien M. Bernede ae1df3df78 Fix portability for size check 2020-01-16 14:54:18 -08:00
Adrien M. Bernede bb0a5aa7d6 Revert "Added a style-check target to the makefile."
This reverts commit 3fc8ee9ec4.
2020-01-16 13:54:26 -08:00
Adrien M. Bernede cafc2ce55f Revert "Adding context in style results to print directory"
This reverts commit 3e22f0a061.
2020-01-16 13:53:25 -08:00
Adrien M. Bernede 131e755abe Adding size check on files in commit + syntax 2020-01-16 13:48:48 -08:00
Tzanio c1e88ae047 Merge branch 'master' into adams-bashforth-dev
Conflicts:
	CHANGELOG
2020-01-16 12:29:17 -08:00
Julian Andrej a375ac17be Merge branch 'master' into nlpa-dev
Conflicts:
	CHANGELOG
	fem/bilininteg.hpp
	tests/unit/fem/test_pa_kernels.cpp
2020-01-16 10:43:59 -08:00
Tzanio 2ddd0c1777 Fix wording in tests/scripts/README 2020-01-16 09:07:18 -08:00
Tzanio 5d5b4af33b Added a README in tests/scripts.
Expanded the test script description in CHANGELOG.
2020-01-16 09:04:26 -08:00
Tzanio 52f2701899 Merge branch 'master' into binary_file_test-tk
Conflicts:
	CHANGELOG
2020-01-16 08:25:48 -08:00
Tzanio 5a1b515a72 Some small fixes git-history 2020-01-15 18:08:36 -08:00
Stowell, Mark L 9d3482ebf3 Replacing return 0 at end of main 2020-01-15 16:59:11 -08:00
Stowell, Mark L c12f7316d2 Adding new miniapp to cmake and documentation files 2020-01-15 16:56:15 -08:00
Tzanio 1326e8641a make style 2020-01-15 16:09:02 -08:00
Julian Andrej 046b990a7d Merge branch 'master' into rectop-bc-dev
Resolved conflicts:
	CHANGELOG
	fem/bilininteg.hpp
	tests/unit/fem/test_pa_kernels.cpp
2020-01-15 15:18:29 -08:00
camierjs 0e05e18d50 miniapps/electromagnetics and tools makefile fix for shared builds 2020-01-15 13:35:22 -08:00
Tzanio 3eb3354796 Merge branch 'master' into get-values-dev
Conflicts:
	miniapps/tools/makefile
2020-01-15 13:03:36 -08:00
Tzanio 96fd8d05cd Added get-values in .gitignore 2020-01-15 13:01:11 -08:00
Ido Akkerman 74643e614e Fix Changelog error 2020-01-15 17:14:39 +01:00
Ido Akkerman e9c2488fda Modified start procedure for AM 2020-01-15 17:07:34 +01:00
Stowell, Mark L c40d653c18 Removing "return" at end of main so that heap objects will be deleted 2020-01-15 00:45:10 -08:00
Stowell, Mark L 0cfffe7ff7 memory related bug fixes in ex22p 2020-01-14 20:06:16 -08:00
Stowell, Mark L a74458bee8 Deleting preconditioners 2020-01-14 19:59:36 -08:00
Stowell, Mark L a34f8feaa7 Initialize solution vector 2020-01-14 19:53:23 -08:00
Stowell, Mark L 6fc34dd1c3 Need to initialize LinearForm member data in every constructor 2020-01-14 19:49:21 -08:00
Stowell, Mark L 3eb8852e5e Objects create on the heap do not destruct before return statement 2020-01-14 19:48:53 -08:00
Tzanio Kolev 67c195af60 Merge pull request #864 from mfem/lib-extras-dev
Collect miniapps/common/*.o in a library file [lib-extras-dev]
2020-01-14 18:35:40 -08:00
Tzanio 496c26aa80 Final changes 2020-01-14 18:34:49 -08:00
Tzanio ba8781d94c Merge branch 'master' into lib-extras-dev 2020-01-14 18:29:17 -08:00
Veselin Dobrev c6731b7dc2 Merge pull request #1167 from mfem/vecmass-vecdiff-dev
Add vector mass and vector diffusion PA integrators
2020-01-14 16:19:50 -08:00
Ido Akkerman 2aff024d74 Make style 2020-01-14 11:59:56 +01:00
Ido Akkerman 403143f880 Small typo corrected 2020-01-14 11:55:14 +01:00
Ido Akkerman b5e619599c Changing CHANGELOG 2020-01-14 11:27:50 +01:00
Ido Akkerman 7237357daf Changing CHANGELOG 2020-01-14 11:25:28 +01:00
Ido Akkerman c976791989 Changing CHANGELOG 2020-01-14 11:23:34 +01:00
Ido Akkerman 20bda274b1 Rewrite index shift again 2020-01-14 11:16:18 +01:00
Ido Akkerman ba4c6d7c5e Rename time integrators in unit test 2020-01-14 10:55:21 +01:00
Ido Akkerman 6f70a2e5ba Rewrite index shift 2020-01-14 10:54:48 +01:00
Ido Akkerman 88f0058b42 AB and AM rewrite 2020-01-14 10:42:19 +01:00
Ido Akkerman 367a833b20 AB and AM fixes 2020-01-14 10:16:28 +01:00
Ido Akkerman fe0b01c43d Generalized alpha fix 2020-01-14 09:29:33 +01:00
Veselin Dobrev eaea9d53a4 Merge branch 'master' into vecmass-vecdiff-dev
Resolved conflicts:
   CHANGELOG
2020-01-13 21:00:47 -08:00
Veselin Dobrev 424b36d676 Merge pull request #1211 from mfem/artv3/app-fixes
Extensions for app developers
2020-01-13 20:49:24 -08:00
Veselin Dobrev 0724286729 Merge branch 'master' into nc-tet-dev
Resolved conflicts:
   makefile
2020-01-13 20:37:06 -08:00
Veselin Dobrev d507e8824a Fix the 'clean' makefile target to remove the subdirectory
'libmfem.x.x.x.dylib.dSYM' -- it was generating an error.
2020-01-13 20:22:47 -08:00
Veselin Dobrev e1186f4781 Yet another update of a random seed in ex12p.
In ex13p, in one of the sample runs, find 4 eigenvectors (instead of
5) to avoid a failure in LOBPCG.
2020-01-13 20:13:15 -08:00
Tzanio Kolev 16e54161ab Merge pull request #1209 from termi-official/load-dc-fix
Fixed synchronization issue in load-dc miniapp.
2020-01-13 18:21:00 -08:00
Tzanio Kolev 8eeda20454 Merge pull request #1223 from mfem/operator-jacobi-fix
Fix an issue in class OperatorJacobiSmoother [operator-jacobi-fix]
2020-01-13 18:19:02 -08:00
Veselin Dobrev 916a5c43dd Fix the shared library build of libmfem-common on linux. 2020-01-13 14:56:03 -08:00
camierjs 080c7500a9 Keep myid ifdef MFEM_DEBUG or MFEM_USE_MPI 2020-01-13 13:36:56 -08:00
camierjs 9287e51160 Fix nvcc 'using' warnings & test_operatorjacobismoother duplicate refine local variable 2020-01-13 12:29:25 -08:00
Tzanio d64bd211d5 Several changes related to porting the mystash/autotest scripts to Travis
- Ported the code-style, documentation, gitignore and runtest scripts from
  mystash/autotest to the tests/scripts directory.

- Based on Adrien's work, added a new script, git-history that checks for
  binary and large files in the history of the current branch.

- Updated .travis.yml to include an expanded Checks stage with the code-style,
  git-history, documentation and gitignore scripts.
2020-01-12 22:45:55 -08:00
Veselin Dobrev 8e2991114c Update the GNU make build system to not install the mfem-common
library and headers.

Update the makefile in miniapps/common to build the mfem-common library.

Update the makefiles in miniapps/{electromagnetics,tools} to use the
mfem-common library from miniapps/common.
2020-01-10 21:32:55 -08:00
Tzanio Kolev 0f31bba2f8 Merge pull request #866 from mfem/yohann/okina-gpu-libceed
[CEED] libCeed integration [yohann/okina-gpu-libceed]
2020-01-10 17:58:22 -08:00
Tzanio 49c6a564a7 Final changes 2020-01-10 17:41:33 -08:00
Veselin Dobrev 0cdb33fe95 Another update of the random seed in ex12p. 2020-01-10 13:36:01 -08:00
Adrien Bernede 897a90e495 Typo in CHANGELOG 2020-01-10 10:50:22 -08:00
Tzanio 8b7a2ff0a8 Fixed a bug 2020-01-09 20:19:53 -08:00
Veselin Dobrev 0a024af85e Update the random seed for one of the sample runs in ex12p. 2020-01-09 19:10:57 -08:00
Adrien M. Bernede 78a4c0df38 adding astyle checks 2020-01-09 17:27:44 -08:00
Adrien M. Bernede 900cd05f49 Adding context in style results to print directory 2020-01-09 17:27:44 -08:00
Aaron Fisher 0e7d9d6458 Added a style-check target to the makefile. 2020-01-09 16:28:43 -08:00
Veselin Dobrev 94da8d4a90 Fix an issue in class OperatorJacobiSmoother.
Unified the implementation of AssembleDiagonal in one place
in BilinearForm which also works in parallel (using the virtual
call GetProlongationMatrix).
2020-01-09 16:16:30 -08:00
Adrien M. Bernede 36f1310de4 Checking for large commits 2020-01-09 15:07:31 -08:00
Ido Akkerman 1fbc0c1a92 Make vector mutable 2020-01-09 22:26:35 +01:00
Ido Akkerman 5a341ca423 Small type correction 2020-01-09 21:18:23 +01:00
Jakub Červený 52be4032e9 Changed random seed because of different triangular refinement. 2020-01-09 11:58:32 +01:00
Jakub Červený 78b7e65dc9 Fixed coarse/fine checks in RefinementMatrix and RefinementOperator. 2020-01-09 11:14:57 +01:00
Adrien M. Bernede 137f86e92d Testing binary per commit 2020-01-08 18:56:07 -08:00
Tzanio 339f23e8b5 Small fixes and adjustments 2020-01-08 18:51:32 -08:00
Tzanio 239637c415 make style 2020-01-08 17:28:27 -08:00
Tzanio f78bbccbf5 Styling unification between libCEED, Ginkgo, HIOP and GSLIB 2020-01-08 16:57:18 -08:00
Tzanio 4eee1d439e Small edits in CHANGELOG 2020-01-08 16:55:47 -08:00
Tzanio d1b636450e Minor edits in CHANGELOG 2020-01-08 16:48:24 -08:00
Tzanio 5534584ef8 Merge branch 'master' into rectop-bc-dev 2020-01-08 16:44:22 -08:00
Adrien M. Bernede f5bd0bcf01 Re-organize tests 2020-01-08 15:40:27 -08:00
Tzanio c3ac60b647 make style 2020-01-08 12:58:04 -08:00
Tzanio b9d6e8f6e8 minor wording 2020-01-08 12:54:05 -08:00
Tzanio 08c2daf6ce make style 2020-01-08 10:56:28 -08:00
Julian Andrej f30b4d7b61 update CHANGELOG 2020-01-08 10:36:08 -08:00
Julian Andrej 89883e515b add copyright header 2020-01-08 10:30:35 -08:00
Julian Andrej 365871cb25 update CHANGELOG 2020-01-08 10:26:02 -08:00
Julian Andrej c8b30e444f updated CHANGELOG 2020-01-08 10:15:25 -08:00
Yohann Dudouit c7953aa673 Change AssembleDiagonal to work with libCEED.
Merge branch 'master' into yohann/okina-gpu-libceed

Conflicts:
	INSTALL
	makefile
2020-01-08 09:37:28 -08:00
Ido Akkerman 91cf36c4c7 Small changes to ODE, still wrong convergence error 2020-01-08 18:19:53 +01:00
Ido Akkerman e13dc42804 Small corrections to unit test 2020-01-08 18:16:16 +01:00
Jakub Červený aa62349c46 Fixed Mesh::UniformRefinement for an all-tet mesh and the 1D case. 2020-01-08 12:10:43 +01:00
Stowell, Mark L 2c63b4c1f7 Removing more references to old names in CMake files 2020-01-07 20:29:36 -08:00
Stowell, Mark L 508f673787 Updating cmake files with new file names and locations 2020-01-07 20:14:01 -08:00
Stowell, Mark L 3ebda213e5 Changing extras to common in various places 2020-01-07 17:14:35 -08:00
Stowell, Mark L 238a5190f9 Oops 2020-01-07 16:59:10 -08:00
Stowell, Mark L 87fbf32df6 Renaming mfem-common.hpp 2020-01-07 16:58:15 -08:00
Stowell, Mark L 2f0be1a27e Moving and renaming new header file 2020-01-07 16:37:10 -08:00
Adrien M. Bernede 3f217cabbe Update changelog 2020-01-07 15:12:51 -08:00
Adrien M. Bernede eec3ff4c87 Adding binary check to travis 2020-01-07 13:26:55 -08:00
Adrien M. Bernede 949f756f3b Testing binary check 2020-01-07 13:26:55 -08:00
Ido Akkerman a13b78b2da Adding time integrator unit test 2020-01-07 14:54:54 +01:00
Jakub Červený 915b2cdf6f Both ex15 and ex15p now prefer nonconforming refinement. Updated descriptions too. 2020-01-07 14:10:59 +01:00
Jakub Červený 0410bedf66 Explicit initialization of the 'ldof' array. 2020-01-07 13:48:15 +01:00
Jakub Červený 6256f0b15d FESpace::DerefinementMatrix: fixed a case of empty processor. 2020-01-07 13:47:33 +01:00
Jakub Červený da055e0f99 Mesh::UniformRefinement: put back a branch for the case 'ref_algo == 1'. 2020-01-07 13:21:11 +01:00
camierjs 139462db10 Merge branch 'master' into nlpa-dev 2020-01-06 10:49:28 -08:00
Stowell, Mark L 8824a222fd Fixing example run 2020-01-06 10:11:33 -08:00
Tzanio Kolev 248debb3dc Merge pull request #982 from mfem/feature/okina-jacobi
Jacobi preconditioning for partial assembly [feature/okina-jacobi]
2020-01-05 21:51:16 -08:00
Tzanio b0a79d993b Final adjustments 2020-01-05 21:50:25 -08:00
artv3 e7c252daba comment fix: Size() -> TotalSize() 2020-01-03 15:18:49 -08:00
Andrew T. Barker f1b0981d34 Fix dumb mistake in examples. 2020-01-03 14:42:53 -08:00
Andrew T. Barker 9f691f5460 Address issues from reviews by Veselin and Tzanio. 2020-01-03 13:22:11 -08:00
Veselin Dobrev acdf7030ce Merge pull request #1166 from mfem/hilbert-sort-dev
Cheap ordering of elements with Hilbert spatial sort [hilbert-sort-dev]
2020-01-02 18:10:00 -08:00
Tzanio 1fa7d4c7c5 Updated CHANGELOG 2020-01-02 13:05:35 -08:00
Andrew Taber f61fccdfbd Add indices to integration points. 2020-01-02 12:59:21 -08:00
Tzanio 096951e913 Merge branch 'master' into feature/okina-jacobi 2020-01-02 12:54:50 -08:00
Tzanio 80552e0d7c Small edits 2020-01-02 12:54:16 -08:00
Tzanio Kolev 11b4c178a5 Merge pull request #1130 from mfem/stefanozampini/petsc-bugfixesandimprovements
Bug fixes and improvements for SparseMatrix conversion to PetscParMatrix
2020-01-02 11:41:48 -08:00
Stefano ZampiniandTzanio Kolev 08f3c2ff59 implement suggestion
Co-Authored-By: Tzanio Kolev <tzanio@llnl.gov>
2020-01-02 19:34:54 +01:00
Tzanio Kolev 831310ba38 Merge pull request #1198 from mfem/corona
HIP fixes for CEED benchmarks
2020-01-02 10:31:34 -08:00
Tzanio 0be36fd40e Final edits 2020-01-02 10:30:31 -08:00
Tzanio 42701bb380 Merge branch 'master' into nc-tet-dev
Conflicts:
	CHANGELOG
2019-12-30 18:22:13 -08:00
Tzanio Kolev 185039f673 Merge pull request #822 from mfem/findpts-gslib-dev
gslib integration [findpts-gslib-dev]
2019-12-30 09:47:07 -08:00
Tzanio 897a931d88 Removed potentially confusing comment. 2019-12-30 09:36:42 -08:00
Tzanio 4886cb45bd Merge branch 'master' into yohann/okina-gpu-libceed
Conflicts:
	CHANGELOG
	CMakeLists.txt
	makefile
2019-12-29 14:48:44 -08:00
Tzanio 0b6f3b31b4 Merge branch 'master' into findpts-gslib-dev
Conflicts:
	CHANGELOG
	makefile
2019-12-29 14:42:43 -08:00
Tzanio Kolev 9afe4ec37e Merge pull request #584 from mfem/complex-mfem-dev
Support for complex-valued linear systems [complex-mfem-dev]
2019-12-29 14:35:06 -08:00
Tzanio Kolev 9e83da7ef8 Merge pull request #1127 from mfem/ginkgo-interface-dev
Interface to the Ginkgo library [ginkgo-interface-dev]
2019-12-29 14:31:17 -08:00
Tzanio ab21b64864 minor 2019-12-28 15:55:14 -08:00
Tzanio KolevandDavid Medina cf7a33c41f Update linalg/vector.cpp
Co-Authored-By: David Medina <dmed256@gmail.com>
2019-12-28 15:52:54 -08:00
Tzanio a356c75fdb Final adjustments 2019-12-27 18:00:27 -08:00
Stowell, Mark L 2359afc806 Merge branch 'complex-mfem-dev' of github.com:mfem/mfem into complex-mfem-dev 2019-12-27 17:40:14 -08:00
Stowell, Mark L ed8d35c29b Adding auxiliary documentation 2019-12-27 17:35:05 -08:00
Tzanio 56443a4826 Updated CHANGELOG 2019-12-27 13:43:44 -08:00
Tzanio 93bb029ba4 Merge branch 'master' into complex-mfem-dev
Conflicts:
	examples/makefile
2019-12-27 13:35:14 -08:00
Tzanio fda21fe590 Final adjustments 2019-12-27 13:23:01 -08:00
Tzanio a901601ba1 Merge branch 'master' into ginkgo-interface-dev
Conflicts:
	CMakeLists.txt
	linalg/linalg.hpp
	makefile
2019-12-27 12:17:51 -08:00
Tzanio a55a3fac00 Updated CONTRIBUTING.md and documentation 2019-12-27 10:35:45 -08:00
Tzanio d466d0dd7c Merge branch 'master' into findpts-gslib-dev
Conflicts:
	CHANGELOG
	config/config.hpp.in
	config/defaults.mk
	makefile
2019-12-27 09:47:59 -08:00
Tzanio a3180cd9e1 Merge branch 'master' into hilbert-sort-dev
Conflicts:
	CHANGELOG
2019-12-27 09:33:43 -08:00
Tzanio Kolev 1b80b82755 Merge pull request #1165 from mfem/vtk_vtu
Add ParaView DataCollection [vtk_vtu]
2019-12-26 18:35:12 -08:00
Tzanio 76a2579dbf Final adjustments 2019-12-26 18:29:56 -08:00
Tzanio bef68c28e7 Merge branch 'master' into vtk_vtu 2019-12-26 18:10:32 -08:00
Tzanio Kolev 53ec6091a4 Merge pull request #616 from mfem/hiop-dev
Interface to HiOp [hiop-dev]
2019-12-26 18:08:09 -08:00
Tzanio f329c1cfe4 Final adjustments 2019-12-26 18:07:01 -08:00
Tzanio Kolev 29b763110a Merge pull request #1189 from mfem/l2-elem-resitrction-gpu
GPU support for L2ElementRestriction
2019-12-24 15:45:17 -08:00
Tzanio Kolev c11424df05 Merge pull request #1185 from mfem/mesh-partitioning-warning-fix
Tweak the warning messages in Mesh::GeneratePartitioning() [mesh-partitioning-warning-fix]
2019-12-24 15:44:05 -08:00
Tzanio Kolev 923edf8674 Merge pull request #1087 from mfem/prod-iter-fix-dev
Added checks in ProductOperator, TripleProductOperator, and RAPOperat…
2019-12-24 15:43:34 -08:00
Stowell, Mark L cdf787a54f Fixing compiler warnings 2019-12-24 11:02:53 -08:00
Vladimir Tomov bffa692809 Update gslib.hpp
Updated doxygen.
2019-12-23 22:22:18 -08:00
Tzanio cbef4990f2 Properly treat ParaView saves in make distclean and .gitignore 2019-12-23 18:57:29 -08:00
Tzanio Kolev 5c03fad898 Merge pull request #1191 from Rombur/parse_documentation
Add documentation for Parse() function of OptionsParser
2019-12-23 18:31:03 -08:00
Tzanio Kolev 6286299efa Merge pull request #1193 from mfem/pa_coeff_host_setup
PA coeff host write in AssemblePA
2019-12-23 18:29:03 -08:00
artv3 fb1607089b a few extensions for application developers 2019-12-23 14:21:53 -08:00
Tzanio f9efa202c1 Fixed CMake, gitignore, doxygen. Some wordsmithing 2019-12-21 23:54:26 -08:00
Dennis Ogiermann 18b23992f5 Fixed synchronization issue in load-dc miniapp. 2019-12-21 22:15:41 +01:00
Tzanio ccb7183d10 Removing demo code in ex1p.cpp 2019-12-21 12:02:52 -08:00
Jakub Červený c7c338d1e9 Removing demo code in ex1.cpp 2019-12-21 15:55:31 +01:00
Vladimir Tomov 866b447fda Minor 2019-12-20 18:20:54 -08:00
Vladimir Tomov 48a4f10aef Fixed incorrect declarations. 2019-12-20 18:20:07 -08:00
Ketan Mittal e1e2b2835e updated file name change in CHANGELOG 2019-12-20 17:25:13 -08:00
Ketan Mittal 1fadb75a0c minor 2019-12-20 17:23:46 -08:00
camierjs cebb80428d Style; added using to avoid warnings; local max T_MAX_D1D and T_MAX_Q1D to allow CUDA compilation 2019-12-20 17:07:28 -08:00
Vladimir Tomov 731a915ff6 Renamed "compare" to "field_diff". 2019-12-20 17:06:38 -08:00
Tzanio e0f483e20f Adjusted GCC diagnostic pragmas 2019-12-20 16:42:37 -08:00
Ketan Mittal bbe5d94df7 changing default mesh order to 3 2019-12-20 16:32:43 -08:00
Julian Andrej 0933afb3be Added general description of PA kernels 2019-12-20 15:57:39 -08:00
blaz e2957cf6cd modified: .gitignore 2019-12-20 14:15:33 -08:00
Ketan Mittal 90243d9318 minor 2019-12-20 11:21:55 -08:00
Ketan Mittal 7fa8f78d76 resolving more comments 2019-12-20 11:12:12 -08:00
Ketan Mittal 063016c344 gaurding GSLIB miniapp so that it is built only when MFEM_USE_GSLIB=YES 2019-12-20 10:31:44 -08:00
Tzanio f782fbd5b0 make style 2019-12-19 18:57:20 -08:00
Tzanio 07ceb2db8e Use the already exisisting macro MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC 2019-12-19 18:52:56 -08:00
Ketan Mittal 6c5db44182 addressed more comments 2019-12-19 17:14:14 -08:00
Julian Andrej 7175701296 Actually do the correct thing in dimension checking... 2019-12-19 15:13:41 -08:00
Julian Andrej 716d836114 Properly format an if statement 2019-12-19 15:13:03 -08:00
Julian Andrej 7fa65739af Add dimension check to new PA kernels 2019-12-19 15:11:43 -08:00
Veselin Dobrev dc5765cf14 Merge branch 'master' into yohann/okina-gpu-libceed 2019-12-19 13:45:32 -08:00
Ketan Mittal 95426d6f9a resolving some of the comments made by the reviewers. 2019-12-19 11:29:34 -08:00
Andrew T. Barker af2ab488a7 Make OperatorJacobi::Setup() public for GPU compilation/use. 2019-12-19 10:13:20 -08:00
Tzanio e2142dd51f make style 2019-12-19 09:59:41 -08:00
Tzanio 32faaccca4 Remove gslib.h from the mfem namespace 2019-12-19 08:42:37 -08:00
Jakub Červený cdfe60232b Adjusted the #pragma to ignore warnings from gslib.h.
The original version didn't work for me (gcc 5.4.0). Apparently
the unused function warning is special in that it gets processed
in later stages in the compiler. Applying the pragma to the whole
file shouldn't be a big deal hopefully.
2019-12-19 14:43:01 +01:00
Jakub Červený 33bed2956c Mentioned new mesh-explorer options in CHANGELOG. 2019-12-19 11:53:06 +01:00
Tzanio ef3eb57978 No need to include "gslib.hpp" in examples.
Less warnings when building with GSLib (none with GCC)

Other small corrections.
2019-12-18 22:21:26 -08:00
Stowell, Mark L cb1a3c44cb Merge remote-tracking branch 'origin/master' into lib-extras-dev
# Conflicts:
#	CHANGELOG
#	miniapps/electromagnetics/volta_solver.hpp
2019-12-18 20:08:42 -08:00
Stowell, Mark L 9bebe0d7ba Changing namespace name 2019-12-18 19:59:49 -08:00
Stowell, Mark L e754c48a17 Updating "make install" target 2019-12-18 19:58:34 -08:00
Stowell, Mark L 34d08210e2 Updating CHANGELOG 2019-12-18 19:58:13 -08:00
Stowell, Mark L 55e5b9540f Updating gitignore 2019-12-18 19:57:58 -08:00
Tzanio eed87d6778 Merge branch 'master' into findpts-gslib-dev
Conflicts:
	CHANGELOG
2019-12-18 18:42:29 -08:00
Dylan Copeland 8d2e581b83 Adding ifdef to skip operator tests if MFEM_USE_EXCEPTIONS is not defined. 2019-12-18 18:35:31 -08:00
Will Pazner bac14b74c0 Destroy Ceed context in Device destructor
Also inititalize internal::ceed to NULL
2019-12-18 15:38:40 -08:00
Yohann Dudouit da41a0f3db Fix the destructor of CeedData. 2019-12-18 14:37:23 -08:00
Dylan Copeland 27d71ec7b1 Adding unit tests for checking iterative mode in product operators. 2019-12-18 12:25:45 -08:00
Stowell, Mark L d2ecffd6d5 Adding a block diagonal preconditioner in the serial example to improve solver performance 2019-12-18 10:32:11 -08:00
Stowell, Mark L 408ed4ef51 Switching to OperatorHandle::As rather than an explicit dynamic cast 2019-12-18 10:31:26 -08:00
Jakub Červený 6ea578ce9b Added more comments in the Hilbert sort algorithm. 2019-12-18 16:32:03 +01:00
Jakub Červený afa8dbc769 Merge branch 'master' into hilbert-sort-dev 2019-12-18 16:11:38 +01:00
Jakub Červený c52fbf7feb mesh-explorer: added options for mesh reordering and 1D partitioning. 2019-12-18 16:03:45 +01:00
Dylan Copeland 8471693dc9 Merge branch 'master' of https://github.com/mfem/mfem into prod-iter-fix-dev 2019-12-17 16:21:27 -08:00
camierjs fde20508f1 Remove escapes that are hipcc specific 2019-12-17 16:01:20 -08:00
Andrew T. Barker 912e2391c6 BilinearForm::AssembleDiagonal() now runs on AMR meshes and gives wrong answer
The diagonal produced has the correct size but not the correct values, and in
this case the code gives a warning.
2019-12-17 14:32:37 -08:00
Andrew T. Barker 174f93a44e Rename variable in OperatorJacobiSmoother::Setup(). 2019-12-17 13:53:31 -08:00
Andrew T. Barker ffb8179eb4 Address (some of) Yohann's comments in PR review. 2019-12-17 13:50:20 -08:00
Andrew T. Barker 43bdeb989e Rename QD3 -> QD2 (after removing previous QD2) 2019-12-17 13:34:49 -08:00
Andrew T. Barker 12260d84f2 Typos + don't compute unnecessary intermediate quantities. 2019-12-17 13:33:56 -08:00
Jakub Červený 32167a1e4f Dropped the array ParFESpace::gfdofs to simplify ghost DOF calculations
and fix a bug related to mixed faces beyond the ghost layer.
2019-12-17 18:14:54 +01:00
Tzanio 6a2b794583 Use mfem::out instead of std::cout 2019-12-17 09:09:42 -08:00
Pratik Nayak 87887627e7 Fix some faulty documentation. 2019-12-17 10:28:29 +01:00
Tzanio 9d045b019e minor 2019-12-16 21:30:35 -08:00
Veselin Dobrev 6220e788c8 Address reviewer comments. 2019-12-16 18:49:25 -08:00
Tzanio aeae0bfda6 merge fix 2019-12-16 16:03:09 -08:00
Tzanio 3328df08d8 Merge branch 'master' into hiop-dev
Conflicts:
	CHANGELOG
2019-12-16 16:01:25 -08:00
Julian Andrej aedff84fb8 Add missing files to cmake build system 2019-12-16 12:01:59 -08:00
Julian Andrej bfe5fcb0be Addressing comments 2019-12-16 11:56:30 -08:00
Stowell, Mark L 36024e3ad2 Fixing typos 2019-12-16 10:28:31 -08:00
camierjs 861823b6d3 Merge branch 'master' into corona 2019-12-16 10:09:26 -08:00
Tzanio ae4f5805ac Merge branch 'master' into hiop-dev 2019-12-15 20:43:36 -08:00
blaz da6ebcd008 Merge branch 'master' into vtk_vtu 2019-12-15 18:48:53 -08:00
blaz 615334bc56 modified: mesh/mesh.cpp 2019-12-14 10:50:21 -08:00
blaz 4695afe395 modified: examples/ex9.cpp
modified:   examples/ex9p.cpp
	modified:   mesh/mesh.cpp
2019-12-14 10:46:32 -08:00
blaz 71a1c46ec5 modified: ex5.cpp
modified:   ex5p.cpp
	modified:   ex9.cpp
	modified:   ex9p.cpp
2019-12-14 10:08:25 -08:00
blaz 5da38b41e3 modified: datacollection.cpp
modified:   ../mesh/mesh.cpp
2019-12-14 10:06:18 -08:00
Stowell, Mark L 64e2970e50 Changing name and location of libmfem-extras in CMake files for consistency with name and location used in Makefile 2019-12-13 21:19:18 -08:00
Stowell, Mark L 406550b47b Moving libmfem-extras to miniapps subdirectory 2019-12-13 21:17:58 -08:00
Veselin Dobrev 42a68b7d28 In the main makefile, ensure the directory fem/libceed is cleaned. 2019-12-13 19:50:32 -08:00
Julian Andrej affee0e682 Clarify assembly level for nonlinearform 2019-12-13 17:22:19 -08:00
blaz e78c9c5d6c modified: ../examples/ex9p.cpp 2019-12-13 17:14:25 -08:00
Tzanio 5732c7fade Merge branch 'master' into hilbert-sort-dev
Conflicts:
	CHANGELOG
2019-12-13 16:24:39 -08:00
camierjs 60d17267f4 PA coeff host write in AssemblePA 2019-12-13 16:21:26 -08:00
Julian Andrej de3063e63b Revert "Removed unused methods from BilinearForm"
This reverts commit e78c578b7b.
2019-12-13 16:17:56 -08:00
Tzanio c49fde0406 make style 2019-12-13 16:14:45 -08:00
blaz e2091dea83 modified: ex9p.cpp 2019-12-13 16:10:05 -08:00
blaz 468cdda403 modified: ../fem/datacollection.cpp
modified:   ../fem/datacollection.hpp
2019-12-13 16:01:39 -08:00
blaz c445981c23 modified: ex5.cpp
modified:   ex5p.cpp
2019-12-13 16:01:20 -08:00
blaz eacf35d4e4 modified: ../CHANGELOG
modified:   ex9.cpp
2019-12-13 16:00:57 -08:00
Andrew T. Barker 0176f9d50c Merge remote-tracking branch 'origin/master' into feature/okina-jacobi
Conflicts:
	fem/bilininteg_diffusion.cpp
2019-12-13 15:56:57 -08:00
Andrew T. Barker 6cd31f61c5 bilininteg_mass: same loop index fix from last commit 2019-12-13 15:48:50 -08:00
Andrew T. Barker 7d02d12828 bilininteg_diffusion: better naming of loop indices 2019-12-13 15:48:04 -08:00
Andrew T. Barker f6edebdcf9 CHANGELOG: add one-line description of matrix-free Jacobi PR 2019-12-13 15:44:00 -08:00
Aaron Fisher d54cd15799 Fixed the defaults in the GINKGO example and polished up the makefile. 2019-12-13 15:29:51 -08:00
blaz c85347314a modified: datacollection.cpp
modified:   ../mesh/mesh.cpp
2019-12-13 15:19:46 -08:00
Aaron Fisher f2d923a8e2 Added the ginko example to the make system and removed the attempt to link to HIP. 2019-12-13 14:51:46 -08:00
Julian Andrej e78c578b7b Removed unused methods from BilinearForm 2019-12-13 12:15:45 -08:00
Julian Andrej c64832304f Add comment to hypre method. 2019-12-13 11:56:25 -08:00
Julian Andrej 2231d7a894 Update error messsages 2019-12-13 11:54:06 -08:00
Julian Andrej ae0bb38c8c Update tests for renaming and formatting 2019-12-13 11:53:17 -08:00
Julian Andrej 704dfa3dfa Rename Divergence to VectorDivergence 2019-12-13 11:53:00 -08:00
Julian Andrej d4bed93244 Rename "output" to test in prolongation restriction. 2019-12-13 11:52:25 -08:00
Julian Andrej 73747debca Removed test case tag MixedBilinearFormBC. 2019-12-13 11:23:30 -08:00
Julian Andrej adeef50ecd Remove MultPA and switch to AddMultPA 2019-12-13 11:20:36 -08:00
Julian Andrej 8e3c3f8b32 Remove AddMultTransposePA in impl 2019-12-13 10:39:20 -08:00
Julian Andrej b96af29a0e Removed AddMultTransposePA. 2019-12-13 10:33:57 -08:00
Julian Andrej 8d927d7072 Remove missed GPU related line. 2019-12-13 10:28:24 -08:00
Ketan Mittal d924e2e061 run serial example in parallel 2019-12-13 09:17:19 -08:00
Bruno Turcksin 5385bf3aae Add documentation for Parse() function of OptionsParser 2019-12-13 12:00:12 -05:00
Yohann Dudouit 2fcc1c7325 Minor style change. 2019-12-12 15:09:12 -08:00
Yohann Dudouit 50f7ac6f80 Merge branch 'master' into yohann/okina-gpu-libceed
Conflicts:
	fem/bilininteg_diffusion.cpp
	fem/bilininteg_mass.cpp
2019-12-12 15:07:42 -08:00
Ketan Mittal 95f6aacf6c Error message if user runs gslib serial example with parallel build 2019-12-12 14:19:11 -08:00
Will Pazner 7e8d95c4e6 Whitespace 2019-12-12 13:22:56 -08:00
Pratik Nayak 112d01d12c Review update.
+ Some minor fixes.
2019-12-12 22:18:00 +01:00
Pratik Nayak 733b09aeeb Update print_lvl to match existing interface. 2019-12-12 21:45:26 +01:00
camierjs cc27809c0f Merge branch 'master' into corona 2019-12-12 11:22:20 -08:00
Tzanio dd87515407 Merge branch 'master' into findpts-gslib-dev
Conflicts:
	CHANGELOG
2019-12-12 10:47:23 -08:00
Tzanio 75e821c3a0 Merge branch 'master' into hiop-dev
Conflicts:
	CHANGELOG
2019-12-12 10:43:23 -08:00
Yohann Dudouit f599d80dd0 Merge branch 'yohann/okina-gpu-libceed' of github.com:mfem/mfem into yohann/okina-gpu-libceed 2019-12-12 09:42:53 -08:00
Yohann Dudouit d504de8705 Add sample runs for libCEED in ex1, ex1p, ex6, ex6p. 2019-12-12 09:42:17 -08:00
Pratik Nayak ccaba7ee90 Add a custom logger to log all the residual norms.
+ If print_lvl>=2, these are printed out to screen.
2019-12-12 12:27:14 +01:00
Veselin Dobrev 06647f5e5f A small tweak in the script 'config/sample-runs.sh' for
the case when using externally built MFEM.
2019-12-11 21:25:21 -08:00
Veselin Dobrev 012f337818 In the script 'config/sample-runs.sh':
- Add support for libCEED in device mode.
- In device mode, make OCCA, RAJA, libCEED optional depending on
  the existence of their respective default directories.
- In device mode, sample runs using OCCA, RAJA, libCEED are
  filtered out if the respecive library is not enabled.
- The CUDA architecture (used when building MFEM in device mode)
  can now be set using the CUDA_ARCH variable, either through the
  environment or on the command line.
2019-12-11 20:47:10 -08:00
Julian Andrej 049d6ace3e moved VectorConvection kernel to separate file 2019-12-11 16:19:51 -08:00
Julian Andrej c64e533e63 Renamed Setup() to AssemblePA() in NonlinearForm PA methods 2019-12-11 16:18:45 -08:00
Veselin Dobrev 1af6a91008 make style 2019-12-11 16:18:22 -08:00
Vladimir Tomov f2c3c30a47 Fixed an error in compare.cpp. 2019-12-11 14:19:40 -08:00
Stowell, Mark L 1b2d1839ba Merge remote-tracking branch 'origin/master' into lib-extras-dev 2019-12-11 11:20:07 -08:00
Stowell, Mark L cb620bf4e9 Merge remote-tracking branch 'origin/master' into get-values-dev 2019-12-11 11:18:42 -08:00
Stowell, Mark L 99bcdaa9dc Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-12-11 11:17:06 -08:00
Stowell, Mark L 7b1cba25d0 Adding unit test for complex operator class 2019-12-10 22:38:26 -08:00
Veselin Dobrev fb8ae593bf Merge branch 'master' into yohann/okina-gpu-libceed 2019-12-10 21:39:00 -08:00
Veselin Dobrev 2562bcda20 Update the minimal needed git-hash for libCEED.
Small cleanup of #include directives.
2019-12-10 20:21:14 -08:00
Vladimir Tomov 387307fcbb Merge branch 'master' into feature/okina-jacobi 2019-12-10 18:24:36 -08:00
Will Pazner c625e19c22 Copy member variables to const locals for GPU kernels 2019-12-10 18:15:13 -08:00
Yohann Dudouit b30187fc87 Set ceedDataPtr to NULL in constructors. 2019-12-10 16:34:49 -08:00
Will Pazner 5b1789392c Make L2ElementRestriction work on GPU
Added Read() and Write() calls and wrapped loops in MFEM_FORALL
2019-12-10 15:33:06 -08:00
Tomov 242557fe13 Minor. 2019-12-10 10:47:43 -08:00
Pratik Nayak e1fb06d479 Fix the printing and convergence check to screen. 2019-12-10 11:27:09 +01:00
blaz f3375302de Merge branch 'vtk_vtu' of https://github.com/mfem/mfem into vtk_vtu 2019-12-09 22:34:05 -08:00
blaz 4aa3158ca2 modified: ../fem/datacollection.cpp
modified:   mesh.cpp
2019-12-09 22:31:37 -08:00
Julian Andrej 96c1794487 always use ginkgo solver for the tests 2019-12-09 12:32:25 -08:00
Julian Andrej 78cae9b6af add example to build and test system in cmake 2019-12-09 12:29:55 -08:00
Tzanio Kolev e60a123b88 Update CHANGELOG 2019-12-08 10:05:53 -08:00
Tzanio ce29e30e51 Minimal changes in CHANGELOG 2019-12-08 10:02:59 -08:00
Tzanio c76a693890 Merge branch 'vtk_vtu' of github.com:mfem/mfem into vtk_vtu 2019-12-08 10:00:11 -08:00
Tzanio 6cc4235431 Fixed accidental tabs in CHANGELOG 2019-12-08 09:59:40 -08:00
Tzanio cddcaad49d Merge branch 'master' into vtk_vtu
Conflicts:
	CHANGELOG
2019-12-08 09:59:20 -08:00
Tzanio 9780c94246 Merge branch 'master' into vtk_vtu
Conflicts:
	CHANGELOG
2019-12-08 09:57:51 -08:00
blaz 4ca6e0a1a0 modified: CHANGELOG 2019-12-07 23:49:38 -08:00
blaz 2892a1a159 modified: CHANGELOG 2019-12-07 23:37:39 -08:00
Pratik Nayak 10cf4b367f Move ginkgo example to ginkgo/ex1.cpp. 2019-12-07 13:53:29 +01:00
blaz c924dfd426 Merge branch 'vtk_vtu' of https://github.com/mfem/mfem into vtk_vtu 2019-12-06 22:55:54 -08:00
blaz 0be7c715c6 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp

The modified code add the possibility to save gridfunctions in xml-paraview format. Example can be found in ex5p.cpp.
2019-12-06 22:53:09 -08:00
Veselin Dobrev db97372598 Improve the messages printed by the method Mesh::GeneratePartitioning()
so that only rank 0 prints the message when running in parallel.

Fix inaccurate warnings (in DEBUG mode) about inverted elements which
were generated when refining a periodic parallel mesh in 2D.

Some formatting improvements.
2019-12-06 19:01:59 -08:00
Tzanio dd17389c1f make style 2019-12-06 14:49:45 -08:00
Tzanio 200da763a3 make style 2019-12-06 14:35:46 -08:00
Tzanio c5b373fc9e make style 2019-12-06 14:27:48 -08:00
Tzanio 5ad2b30fc2 make style 2019-12-06 14:25:27 -08:00
Tzanio c5057e1b11 make style 2019-12-06 14:21:07 -08:00
Dylan Copeland 7cd0c7f6cb Reverting to the master version of ex1p, by removing a product operator test that demonstrates the new feature for the PR. 2019-12-06 11:48:31 -08:00
blaz 2dbd31901b modified: datacollection.cpp 2019-12-05 16:12:21 -08:00
Stowell, Mark L a34a775299 Removing incorrect files from CMake files (they were already listed in another subdirectory) 2019-12-05 13:50:08 -08:00
Stowell, Mark L 657d116b33 Adding files related to complex classes to CMake files 2019-12-05 13:42:55 -08:00
Stowell, Mark L fa1d962850 Removing dead code 2019-12-05 00:37:49 -08:00
Stowell, Mark L e27befa91f Using more descriptive method name for AssembleComplexSparseMatrix 2019-12-05 00:37:25 -08:00
Stowell, Mark L 42e2e02d77 Adding set/get convention methods to complex linear form classes 2019-12-05 00:36:43 -08:00
Stowell, Mark L eb56f8d80e Expanding or correcting comments in headers related to complex classes 2019-12-05 00:36:09 -08:00
Stowell, Mark L 2cb0a0e5f1 Using more descriptive window titles 2019-12-05 00:33:52 -08:00
Stowell, Mark L 2aecabc7e7 Removing trailing underscores in member data names to conform to MFEM standard 2019-12-04 16:48:43 -08:00
Stowell, Mark L a0e4ea2a34 Adding ex22 and ex22p to CMake build 2019-12-04 16:32:24 -08:00
Julian Andrej 231dd67600 Merge branch 'vecmass-vecdiff-dev' of github.com:mfem/mfem into vecmass-vecdiff-dev 2019-12-04 11:52:01 -08:00
Julian Andrej 80b796d0f8 Added missing files to cmake build system 2019-12-04 11:51:10 -08:00
Pratik Nayak d14287dd47 Review update.
+ Fix an include issue.
+ Fix some issues with the makefile build system.
2019-12-04 08:49:15 +01:00
Stowell, Mark L ad16a51fe0 Fixing errors that arose in testing 2019-12-03 20:24:14 -08:00
Vladimir Tomov 453876a32c Minor. 2019-12-03 19:09:14 -08:00
Stowell, Mark L 1c78456d63 Adding get-values miniapp 2019-12-03 17:07:46 -08:00
Yohann Dudouit 0052685d39 Remove memory leak for ex6 -pa -d ceed-cuda. 2019-12-03 16:50:39 -08:00
Julian Andrej 40c3c76742 removed placeholders 2019-12-03 11:07:55 -08:00
Julian Andrej a8e9ed5576 removed unnecessary comments 2019-12-03 11:00:26 -08:00
Julian Andrej cc1fa3afdd More descriptive name for indicating of non owned integrators in BilinearForm 2019-12-03 10:49:02 -08:00
Yohann Dudouit e05e7748fd Fix circular dependency. 2019-12-02 18:00:12 -08:00
blaz 5d2ef58b15 Merge branch 'vtk_vtu' of https://github.com/mfem/mfem into vtk_vtu 2019-12-02 16:02:02 -08:00
blaz 6adc0ee115 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp
2019-12-02 16:01:14 -08:00
Andrew T. Barker 89838888d8 bugfix: correct diagonal coefficient (thanks to @artv3) 2019-12-02 12:10:02 -08:00
Andrew T. Barker c395923f4a matrix-free jacobi: add unit test for fichera mesh. 2019-12-02 12:05:01 -08:00
Yohann Dudouit fed1783fd5 Use CeedOperatorApplyAdd instead of CeedOperatorApply. 2019-12-02 10:33:56 -08:00
Yohann Dudouit 8699467c61 destroy CeedData in integrators containing one. 2019-12-02 10:25:30 -08:00
Tzanio 85f4427f12 make style 2019-12-01 13:13:31 -08:00
blaz c45160db6e modified: ex5.cpp
modified:   ex5p.cpp
2019-11-27 10:24:39 -08:00
blaz 566d3bcc2b modified: examples/ex5p.cpp
modified:   fem/datacollection.cpp
2019-11-26 15:34:49 -08:00
Stowell, Mark L f3fcfa67c1 Switching to standard essential BC treatment for solver performance gain (thanks to Socratis for catching this!) 2019-11-26 14:57:03 -08:00
blaz 2d48216caf modified: fem/datacollection.hpp 2019-11-26 09:30:56 -08:00
blaz 886ae8c20b modified: datacollection.hpp 2019-11-26 09:30:09 -08:00
blaz c48388ef5e modified: ex5p.cpp 2019-11-26 09:20:42 -08:00
Jakub Červený 1b3aefb467 make style, update CHANGELOG 2019-11-26 14:45:04 +01:00
Jakub Červený eba915947c Cleanup. 2019-11-26 14:27:33 +01:00
Stowell, Mark L ba4d20463d make style 2019-11-25 15:02:26 -08:00
Stowell, Mark L 08447b9ab0 Adding face integrator support 2019-11-25 15:01:27 -08:00
Stowell, Mark L 22cc2171eb Removing more debug code 2019-11-25 14:48:33 -08:00
Stowell, Mark L 05c31920e5 Removing debug code 2019-11-25 14:48:10 -08:00
Stowell, Mark L 3befd52a89 Adding boundary integrator support 2019-11-25 14:47:52 -08:00
Stowell, Mark L 5810563c5e Adding new macros for HYPRE memory allocation functions 2019-11-25 14:36:42 -08:00
Ketan Mittal 9b5609310d resolving conflict in CHANGELOG 2019-11-25 10:50:35 -08:00
k10 d17bbb583b Merge branch 'master' into findpts-gslib-dev 2019-11-25 10:41:48 -08:00
Jakub Červený 8cd967b7cc ParGridFunction::ProjectBdrCoefficientTangent: no need for separate NC branch. 2019-11-22 10:14:58 +01:00
Jakub Červený a1a2a44bc0 Clean-up and minor changes. 2019-11-22 10:14:29 +01:00
Jakub Červený e7fab9a056 Merge branch 'master' into nc-tet-dev 2019-11-22 09:19:25 +01:00
Jakub Červený 42f3573275 3D version of Hilbert sort added. 2019-11-21 20:23:55 +01:00
Jakub Červený 5a08cfb5d7 Alternate Hilbert sort algorithm that's easier to extend to 3D. 2019-11-21 16:53:52 +01:00
blaz c98367435b modified: examples/ex5p.cpp
modified:   fem/datacollection.cpp
	modified:   fem/datacollection.hpp
2019-11-20 16:47:02 -08:00
Will Pazner 8dce4cf447 Fix naming and organization of tests 2019-11-20 14:07:46 -08:00
Will Pazner f7475f9b08 Add consistency test for PA vector mass and diffusion 2019-11-20 14:03:02 -08:00
Will Pazner a4e33dd8ad Bugfix for BilinearForm::Mult with PA 2019-11-20 14:02:39 -08:00
Julian Andrej b8dd7b8a63 added vector mass and vector diffusion pa integrators 2019-11-20 11:52:53 -08:00
Jakub Červený 875add7077 Merge branch 'master' into hilbert-sort-dev 2019-11-20 20:40:43 +01:00
Jakub Červený b3726c43c5 Mesh::GetHilbertElementOrdering: spatial sort of elements. 2019-11-20 19:29:08 +01:00
Jakub Červený 2b92b7a051 Renamed GetGeckoElementReordering -> GetGeckoElementOrdering. 2019-11-20 19:28:18 +01:00
blaz f5980449ac modified: mesh/mesh.hpp 2019-11-20 09:59:17 -08:00
blaz 3b07bc4f52 modified: examples/ex5.cpp
modified:   examples/ex5p.cpp
	modified:   fem/datacollection.cpp
	modified:   fem/datacollection.hpp
	modified:   mesh/mesh.cpp
2019-11-20 09:59:02 -08:00
blaz 92d330b5ca modified: examples/ex5p.cpp 2019-11-19 22:28:41 -08:00
blaz a2390986c1 modified: examples/ex5.cpp
modified:   examples/ex5p.cpp
	modified:   fem/datacollection.cpp
	modified:   fem/datacollection.hpp
2019-11-19 22:19:16 -08:00
blaz 2fdd1702b3 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp
2019-11-19 21:05:50 -08:00
Will Pazner ae4f5b95fa Add test case verifying PA convection
Verify that the partial assembly and fully assembly for nonlinear convection give the same results in 2D and 3D
2019-11-19 17:12:23 -08:00
Will Pazner 9f61cd440a Get rid of explicit template kernel instantiations
Instead use the generic fallback kernel with bounds known only at runtime
2019-11-19 17:11:50 -08:00
Will Pazner 50ea6dc081 Add 3D shared memory kernel for nonlinear convection 2019-11-19 17:10:59 -08:00
Will Pazner e4562b0509 Bugfix: PA for NonlinearForm now works in serial 2019-11-19 17:10:39 -08:00
Julian Andrej 8eff8e133f correct function signature 2019-11-19 17:05:41 -08:00
Julian Andrej 6ac0949886 added shared memory kernels 2019-11-19 16:59:54 -08:00
Julian Andrej 96e5f71350 add more unit tests 2019-11-19 16:11:54 -08:00
Julian Andrej 98b68c378d add unit tests for mixed PA integrators 2019-11-19 14:14:31 -08:00
Veselin Dobrev a42d209056 Fix libCEED's version requirement. 2019-11-18 20:28:46 -08:00
Julian Andrej c84af9586d first step picking hunks for rectangular operators 2019-11-18 16:49:44 -08:00
Veselin Dobrev b87ec735d2 Apply 'make style' and other small tweaks. 2019-11-18 16:34:52 -08:00
Julian Andrej 12fd291158 make style 2019-11-18 14:52:15 -08:00
Julian Andrej f14c15ad7a Add PA for nonlinear integrators and an example vector convection integrator 2019-11-18 14:47:20 -08:00
Tzanio 2420773157 make style 2019-11-16 17:03:36 -08:00
blaz ad1e4681e2 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp
2019-11-15 20:28:53 -08:00
blaz e4c82ea746 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp
2019-11-15 15:14:08 -08:00
blaz fee506b509 modified: fem/datacollection.cpp
modified:   fem/datacollection.hpp
	modified:   mesh/mesh.cpp
	modified:   mesh/mesh.hpp
2019-11-15 09:21:28 -08:00
Ketan Mittal b0adbd64fb minor 2019-11-12 14:35:39 -08:00
Ketan Mittal e312442467 updating comments 2019-11-12 11:41:33 -08:00
Pratik Nayak c8da9199c5 Update CHANGELOG and INSTALL. 2019-11-12 10:18:37 +01:00
Ketan Mittal 2c845f5ce5 minor 2019-11-11 11:30:42 -08:00
Ketan Mittal 2d74c9e0f1 update CHANGELOG and compare.cpp style to match miniapps 2019-11-11 10:18:21 -08:00
Ketan Mittal 8351147f94 update CHANGELOG and remove README since installation details are now in INSTALL 2019-11-11 10:04:35 -08:00
Tzanio c9d022c259 make style 2019-11-11 09:01:57 -08:00
Yohann Dudouit 7488a91216 Merge branch 'yohann/okina-gpu-libceed' of github.com:mfem/mfem into yohann/okina-gpu-libceed 2019-11-08 09:54:44 -08:00
Veselin Dobrev e18916ec92 Make sure CMake build works with libceed. In the GNU make build
system, install libceed q-function headers.

Add a runtime check for q-function header location: install/source
directories.
2019-11-07 16:48:55 -08:00
camierjs 5823d789d9 Use non Occa diagonal 2019-11-07 16:12:17 -08:00
camierjs 8998d0bc03 SmemPAMassAssembleDiagonal 2D, 3D and op, x & y cleanup 2019-11-07 15:52:11 -08:00
camierjs 6521ff13f3 Less shared memory for SmemPADiffusionDiagonal3D 2019-11-07 12:55:56 -08:00
Yohann Dudouit 8decda8986 Add documentation to libCEED functions in MFEM. 2019-11-07 11:29:31 -08:00
Pratik Nayak ad051d6ca6 Add makefile support and ILU preconditioner to ex1. 2019-11-07 12:09:10 +01:00
Yohann Dudouit e03e973342 Add libCEED to CMake. 2019-11-06 15:38:17 -08:00
Yohann Dudouit 6b8414d820 First set of modifications based on technical meeting.
- guards CEED parts in bilininteg.hpp
- Capitalise some CEED related functions
- Replace Device::IsEnabled with Device::Allows(CUDA)
- Remove dead code in libceed/mass.cpp
2019-11-06 14:24:04 -08:00
Ketan Mittal 3b0c33a885 minor 2019-11-06 09:49:04 -08:00
Ketan Mittal a6ce73c975 updating changelog and install 2019-11-06 09:26:33 -08:00
Vladimir Tomov 74241fc9ac Added files and functions for a sample run the compare.cpp.
Around 61 KB.
2019-11-05 11:01:50 -08:00
Ketan Mittal aa1a32f8bb minor change to compare 2019-11-04 16:20:20 -08:00
Ketan Mittal 7dbfbb2b33 merging with master 2019-11-04 16:14:14 -08:00
Tzanio b7cac4145a make style 2019-11-03 13:32:21 -08:00
Yohann Dudouit 69810197f6 Use enum class CeedCoeff instead of enum. 2019-10-30 12:29:35 -07:00
Pratik Nayak c22786ce61 Add ginkgo CGSolver example to ex1. 2019-10-29 10:44:49 +01:00
Yohann Dudouit 2a9cd29202 Remove unused files. 2019-10-28 12:23:47 -07:00
camierjs 57076e0118 Merge branch 'master' into feature/okina-jacobi 2019-10-25 10:43:24 -07:00
camierjs 0c4bf32b50 Merge branch 'master' into corona 2019-10-25 10:39:22 -07:00
Yohann Dudouit b1784fc793 Replaces void* by CeedData* in Mass and Diffusion integrators. 2019-10-24 16:52:51 -07:00
Tzanio 8bd488a331 Merge branch 'master' into hiop-dev
Conflicts:
	config/config.hpp.in
	makefile
2019-10-24 12:42:21 -07:00
Pratik Nayak cb07d9873d Add the Ginkgo iterative solvers interface and their implementations.
+ Take the executor as a string and create the gko::Executor within the Base
class.
+ Try to keep the solver interface as similar to the original mfem solver as
possible.
2019-10-22 16:46:45 +02:00
Pratik Nayak 2bf1517453 Add Ginkgo library CMake detection and linking. 2019-10-21 17:28:39 +02:00
Stefano Zampini c53fb9962e PetscParMatrix: support for triple matrix matrix products 2019-10-20 17:08:46 +03:00
Stefano Zampini 3363aa82c9 PetscParMatrix: support constructing a block-diagonal operator starting from a SparseMatrix
Fix bug in sorting column indices
2019-10-20 17:07:54 +03:00
Yohann Dudouit 4b9971d377 Merge branch 'yohann/okina-gpu-libceed' of github.com:mfem/mfem into yohann/okina-gpu-libceed 2019-10-18 16:47:17 -07:00
Yohann Dudouit 1a6f2226b3 Print the name of the backend used by libCEED and check that it is the requested one. 2019-10-18 16:46:37 -07:00
Yohann 675b507a7c Merge branch 'master' into yohann/okina-gpu-libceed 2019-10-17 15:46:30 -07:00
Yohann Dudouit 3a3bbeae4d Default cpu backend to /cpu/self since it picks automatically the best cpu backend available. 2019-10-17 15:26:37 -07:00
Yohann Dudouit dfc7f9f66d Rename a function to make it self documented. 2019-10-07 17:27:59 -07:00
Dylan Copeland 9378398cae Added checks in ProductOperator, TripleProductOperator, and RAPOperator, to prevent Solvers from using uninitialized vectors in iterative mode. Added a test of this feature in ex1p.cpp. 2019-09-24 15:13:13 -07:00
camierjs d245dc8485 Merge branch 'master' into yohann/okina-gpu-libceed 2019-09-23 10:41:52 -07:00
camierjs be15513f1f Merge branch 'master' into feature/okina-jacobi 2019-09-23 10:40:56 -07:00
camierjs 789ea867fe Merge branch 'master' into corona 2019-09-23 09:38:37 -07:00
Veselin Dobrev 5dbf3c8b9f Change the default CPU backend for libCEED. 2019-09-13 15:18:23 -07:00
Yohann Dudouit 407705c8de Define CEED_QFUNCTION. 2019-09-11 18:43:42 -07:00
Yohann Dudouit 0627783135 Add typedef when MFEM_USE_CEED=NO. 2019-09-11 18:22:49 -07:00
Yohann Dudouit c9fc6dfada Move libCEED files in a libceed subdirectory. 2019-09-11 18:00:31 -07:00
Yohann Dudouit c226405378 Remove ceed.hpp file. 2019-09-11 18:00:31 -07:00
Yohann Dudouit c6faec60f8 Cleaning due to Single Source QFunction. 2019-09-11 18:00:31 -07:00
Yohann Dudouit 3cb600cf8b Use correct sizes for In/Outputs of qFunctions. 2019-09-11 18:00:31 -07:00
Yohann Dudouit b610b36173 Update MFEM to support last version of libCEED. 2019-09-11 18:00:31 -07:00
Tzanio 930daccddb minor edits 2019-09-11 18:00:28 -07:00
Tzanio d84b46bc45 make style 2019-09-11 17:58:53 -07:00
Yohann Dudouit 51060626fe Fix a bug in the definition of the restriction for the diffusion op. 2019-09-11 17:57:52 -07:00
Yohann Dudouit ff69c272cc Update documentation of 'Configure' function. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 6bc7fa8216 Add #ifdef MFEM_USE_CEED guards. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 26f1114105 Separate diffusion and mass specific code from libceed.cpp. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 45bbc247a4 Make /gpu/cuda/gen the default ceed-cuda backend. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 0e812c6a01 split qFunctions for diff_apply in three functions. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 177fd7f80d Update .qf file 2019-09-11 17:57:52 -07:00
Yohann Dudouit b64de4db84 Use the Context to apply correctly the qFunction for cuda-gen backend. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 0e6ec8be7e Add qf file for cuda-gen backend. 2019-09-11 17:57:52 -07:00
Yohann Dudouit a09ce63307 libCEED takes into account the integration rule order of MFEM. 2019-09-11 17:57:52 -07:00
Yohann Dudouit 40f4200bb5 Solve double free bug. 2019-09-11 17:57:52 -07:00
camierjs 14c9a845a2 ceed_option fix 2019-09-11 17:57:52 -07:00
camierjs 09a93c96fb Rebase with ex1 working 2019-09-11 17:57:51 -07:00
Yohann Dudouit 9a3d7a34ad Rebase compiles. 2019-09-11 17:57:51 -07:00
Yohann Dudouit 05fcff7325 Move bilininteg_ext into bilininteg_diffusion/mass 2019-09-11 17:57:51 -07:00
Yohann Dudouit 87869636cc gpu debugging 2019-09-11 17:57:51 -07:00
Yohann Dudouit 9aa6823a0b CHANGELOG update 2019-09-11 17:57:51 -07:00
Tzanio 90c4f4fd78 make style 2019-09-11 17:57:51 -07:00
Yohann Dudouit bde675abff Add #ifdef MFEM_USE_CEED guards 2019-09-11 17:57:51 -07:00
Yohann Dudouit 065817c5e1 Remove GetCeed() function 2019-09-11 17:57:49 -07:00
Yohann Dudouit 689c259c80 Add MFEM_USE_CEED option 2019-09-11 17:56:08 -07:00
Yohann Dudouit 05cf4c40b0 Add option to libCEED backend ':' is the option separator.
- example: ./ex1 -pa -d ceed-cpu:/cpu/self/avx/blocked
  - bug: ./ex1 -pa -d ceed-cuda:/gpu/cuda/reg computes 'nan'
  - however, ./ex1 -pa -d ceed-cpu:/gpu/cuda/reg works
  - bug: ./ex1 -pa -d ceed-cpu:/gpu/cuda/ref crashes in libCEED
  - however, ./ex1 -pa -d ceed-cuda:/gpu/cuda/ref works...
2019-09-11 17:54:53 -07:00
Yohann Dudouit 9ddb97e100 Fix GridFunctionCoefficient for GPU backends. 2019-09-11 17:54:53 -07:00
Yohann Dudouit b69a90eb16 Initial commit for okina gpu libceed.
Ceed and Cuda backend working together.

Use MFEM_SOURCE_DIR to find okl and cu files

Separate most of ceed integration in libceed.?pp
2019-09-11 17:54:36 -07:00
camierjs 8ddd828872 Add hipVectorMin and hipVectorDot 2019-09-10 15:08:46 -07:00
camierjs aa256a9a3c CONFIG_FILE_DEF escape quotes 2019-09-10 14:31:49 -07:00
Tzanio e39b2936c9 make style 2019-08-29 16:32:08 -07:00
Mark L. Stowell 406eda67e5 Adding newline because this sometimes matters.
Thanks, @jandrej , for adding this line.
2019-08-29 15:03:10 -07:00
Julian Andrej cdedb53914 fix cmake build with shared libs 2019-08-29 14:59:16 -07:00
Andrew T. Barker 41feec751f Remove debug print in bilininteg_diffusion.cpp (thanks to @drzisga) 2019-08-26 13:55:00 -07:00
Vladimir Tomov 5f5b6958f2 Updated /doc. 2019-08-22 12:39:27 -07:00
Stowell, Mark L 25696d300f Merge remote-tracking branch 'origin/master' into lib-extras-dev
# Conflicts:
#	makefile
2019-08-22 10:51:38 -07:00
Stowell, Mark L 25fc017cee Renumbering complex examples as ex22 2019-08-21 17:00:20 -07:00
Andrew T. Barker 6273c3b3cc Finish renaming OperatorJacobiSmoother unit test. 2019-08-21 16:48:30 -07:00
Andrew T. Barker 72017a850c Rename unit test to match class. 2019-08-21 16:47:18 -07:00
Andrew T. Barker fa9cea9486 Rename VectorSmoother -> OperatorJacobiSmoother 2019-08-21 16:45:48 -07:00
Tzanio 5b684c3f79 make style 2019-08-21 16:44:27 -07:00
Andrew T. Barker 875da56cd2 bilininteg_diffusion: consistent naming between 2D and 3D jacobi 2019-08-21 16:38:03 -07:00
Andrew T. Barker 3a2fd4b165 bilininteg_diffusion: slightly better variable names in Jacobi diagonal tensor contractions 2019-08-21 16:36:05 -07:00
Stowell, Mark L d8af7a54a1 Renumbering ex21 2019-08-21 16:25:39 -07:00
Tomov 60f37e6f19 Updated gitignore. 2019-08-21 16:24:37 -07:00
Stowell, Mark L f2220194b6 Merge remote-tracking branch 'origin/master' into complex-mfem-dev
# Conflicts:
#	examples/CMakeLists.txt
#	examples/makefile
2019-08-21 16:19:54 -07:00
Stowell, Mark L ceec7e125b Removing dead code 2019-08-21 16:00:27 -07:00
Tomov 1c51f0e5a5 Updated INSTALL. 2019-08-21 14:51:17 -07:00
Tomov 7e0d9d6603 MFEM_USE_HIOP check in the hiop example. 2019-08-21 14:35:36 -07:00
Tomov 1a5364b83b Updated CHANGELOG. 2019-08-21 14:31:14 -07:00
Vladimir Tomov d0d0dc0356 Improved the description of the hiop examples. 2019-08-20 16:59:51 -07:00
Vladimir Tomov e16916691f Merge branch 'hiop-dev' of ssh://github.com/mfem/mfem into hiop-dev
Conflicts:
	CMakeLists.txt
2019-08-20 16:35:20 -07:00
Vladimir Tomov 54797b3574 Changed some Get..() methods to return constant pointers, styling. 2019-08-20 16:31:32 -07:00
Vladimir Tomov 87a45c85d3 Minor. 2019-08-20 15:57:52 -07:00
Vladimir Tomov 52c90d03cb Merge branch 'master' into hiop-dev
Conflicts:
	CMakeLists.txt
	config/cmake/config.hpp.in
	config/config.hpp.in
	config/defaults.cmake
	config/defaults.mk
	makefile
2019-08-20 15:17:19 -07:00
Vladimir Tomov d700ed6b5b Small example to compare 2 functions on different meshes. 2019-08-13 12:37:22 -07:00
Jakub Červený c1ddee3588 Fixed ParNCMesh::GetFaceGeometry for slave faces beyond the ghost layer. 2019-07-29 15:22:23 +02:00
Jakub Červený 6194f90de4 Added verification of valid geom in GetBareDofs. 2019-07-29 14:58:53 +02:00
Jakub Cerveny 165b3e5a75 Fixed a bug in ParallelDerefinementMatrix (crash due to bad diag matrix). 2019-07-25 15:30:16 +02:00
Jakub Červený b05f5796fa NCMesh::GetDerefinementTransforms: identity always produced for existing Geoms. 2019-07-24 18:17:07 +02:00
Jakub Červený 89585aaef4 DenseTensor: added ASSERTS on i,j,k to check access. 2019-07-24 18:15:47 +02:00
Jakub Červený bb663c8418 Fixed FESpace::GetDegenerateFaceDofs for RT spaces. 2019-07-24 14:19:13 +02:00
Jakub Červený 73e8018ddf Example 15p works for star-mixed.mesh. 2019-07-24 11:50:36 +02:00
Jakub Červený 20a4f06a95 Debugging ParallelDerefinementMatrix. 2019-07-23 18:57:26 +02:00
Jakub Červený 6b40d23f6d WIP ParallelDerefinementMatrix for mixed meshes. 2019-07-23 15:31:04 +02:00
Jakub Červený c9e3c4b064 It seems we no longer need the two branches in ParGridFunction::ProjectBdrCoefficient
for conforming and NC meshes. Also fixes the assert going off in ex15p.
2019-07-05 16:53:40 +02:00
Jakub Červený 3223ba9a94 Fixed a problem that was triggering the check in ParFESpace::GetEssentialTrueDofs
for 2D meshes. Essential DOFs are now calculated correctly.
2019-07-05 16:51:33 +02:00
Jakub Červený c17f3bf8e1 ParFESpace::RebalanceMatrix now supports mixed meshes. 2019-07-01 18:04:52 +02:00
camierjs 68f360f38c 3D Jacobi PA precond. for Diffusion on GPU 2019-06-28 15:04:02 -07:00
camierjs 53cd24ebfb 2D Jacobi diffusion kernel on GPU 2019-06-27 16:40:25 -07:00
Andrew T. Barker c7e445d1dc unittest: add test for VectorSmoother 2019-06-26 13:27:59 -07:00
Andrew T. Barker c0839e6e0c ex1p: implement pa jacobi preconditioning
Performance is disappointing at the moment.
2019-06-26 13:27:57 -07:00
Andrew T. Barker 8b146d4c06 ex1: implement Jacobi preconditioning for partial assembly 2019-06-26 13:27:53 -07:00
Andrew T. Barker d72c0d524f solvers: draft a (potentially partially assembled) VectorSmoother 2019-06-26 13:27:49 -07:00
Andrew T. Barker d31cd56e8c DiffusionIntegrator: implement 3D tensorized diagonal assembly
This is an ugly implementation, I'm sure the memory accesses are
terrible, but the asymptotic computational complexity should be
correct.
2019-06-26 13:27:37 -07:00
Andrew T. Barker 21ccf19b0c BilinearIntegrator: implement 2D diffusion diagonal assembly kernel 2019-06-26 13:27:24 -07:00
Andrew T. Barker 4dd00e1759 DiffusionIntegrator: skeleton of methods for PA diagonal assembly 2019-06-26 13:27:20 -07:00
Andrew T. Barker ad0bfa2e11 unittest: test 3D mass diagonal PA assembly 2019-06-26 13:27:17 -07:00
Andrew T. Barker 4035974dd6 MassIntegrator: implement PA in 3D 2019-06-26 13:27:13 -07:00
Andrew T. Barker d0a12de060 ParBilinearForm: add mpi-parallel AssembleDiagonal() method 2019-06-26 13:27:08 -07:00
Andrew T. Barker 87e01ff443 PA mass diagonal works for multiple elements, does not work in parallel. 2019-06-26 13:27:03 -07:00
Andrew T. Barker 01fce1bdb5 Add very basic unit test for 2D mass diagonal with partial assembly. 2019-06-26 13:26:56 -07:00
Andrew T. Barker 5f15e063bc bilinearform: diagonal assembly appears to work for one element in 2d for mass integrator 2019-06-26 13:26:52 -07:00
Andrew T. Barker a262d9e910 bilininteg: draft a kernel to assemble the mass diagonal
(this is untested)
2019-06-26 13:26:42 -07:00
Andrew T. Barker 21065c2334 bilinearform: add some boilerplate for device-based diagonal assembly 2019-06-26 13:26:25 -07:00
Jakub Červený 3a3c6b37e4 Cleaned up, updated CHANGELOG. 2019-06-26 13:50:28 +02:00
Jakub Červený 40ca7932a9 NC limit for tets. ex15 & ex15p seem to work. 2019-06-21 17:53:54 +02:00
Jakub Červený 9b21135c54 Coarse-fine transforms for tets. 2019-06-21 16:58:44 +02:00
Jakub Červený b57130cc2d Fixes in parallel code. 2019-06-21 16:03:25 +02:00
Jakub Červený 16b44bb495 Removed unnecessary members NCMesh::gi_*. 2019-06-21 11:35:59 +02:00
Jakub Červený 072c18bd2d Fixed GetDegenerateFaceDofs for triangular faces. 2019-06-21 11:20:30 +02:00
Jakub Červený e9ff4f1533 Edge-face constraints for tets almost working. 2019-06-20 17:22:09 +02:00
Jakub Červený af869b3559 Fixed tet neighbor calculation. Parallel code now works. 2019-06-14 12:24:38 +02:00
Jakub Červený 9874e7e6ed ElementSet::DecodeTree: printing refinement path to help debugging. 2019-06-14 09:41:42 +02:00
Jakub Červený a1411c9647 Fixed face->index and face_geom of ghost faces. 2019-06-13 12:57:42 +02:00
Jakub Červený 666b2f8745 Still debugging a problem in ParNCMesh::OnMeshUpdated. 2019-06-12 17:49:39 +02:00
Jakub Červený c9a424d998 Debugging parallel code. 2019-06-07 17:11:27 +02:00
Ido Akkerman 6041b10037 Restoring deleted routines 2019-06-06 16:47:31 +02:00
Jakub Červený 7fa1bd51ad Tet derefinement works. 2019-06-06 16:14:32 +02:00
Jakub Červený 3313f49050 Removing diagonal length calculations. Type 0 seems to suffice. 2019-06-06 11:59:42 +02:00
Ido Akkerman 5c9a6c6628 Merge branch 'master' into adams-bashforth-dev 2019-06-05 09:31:21 +02:00
Jakub Červený 52ec1d9f68 Shortest diagonal algorithm modeled after Mesh::UniformRefinement3D. 2019-05-31 18:29:43 +02:00
Jakub Červený 9f5de7fa79 WIP shortest diagonal tet refinement approach. 2019-05-30 22:41:40 +02:00
Jakub Červený 9b101e71e5 Trying a more stable tet octasection algorithm. 2019-05-30 15:47:11 +02:00
Jakub Červený bf3e2590dc Changes in Mesh due to NC tet refinement. 2019-05-30 15:46:51 +02:00
Jakub Červený 185be2fd19 Starting non-conforming tet refinement branch. 2019-05-29 17:20:27 +02:00
Tzanio 12bf8c88ea Merge branch 'master' into hiop-dev
Conflicts:
	CMakeLists.txt
	config/cmake/config.hpp.in
	config/config.hpp.in
	config/defaults.cmake
	config/defaults.mk
	makefile
2019-05-25 22:32:43 -07:00
Tzanio 7de1dcd195 Merge branch 'master' into findpts-gslib-dev
Conflicts:
	makefile
2019-05-25 22:26:56 -07:00
Stowell, Mark L 701c22c52d Merge remote-tracking branch 'origin/master' into lib-extras-dev
# Conflicts:
#	makefile
2019-05-01 10:01:57 -07:00
Stowell, Mark L 63f22c7d8d Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-05-01 09:58:43 -07:00
Stowell, Mark L 89046f0bc1 Adding proposed lib file to .gitignore 2019-04-29 23:54:17 -07:00
Vladimir Tomov daadc7cb64 Removed old files, moved the RTaylor 2D mesh in /data. 2019-04-26 10:55:11 -07:00
Vladimir Tomov c664b7a951 Missed a conflict. 2019-04-25 16:13:50 -07:00
Vladimir Tomov 965b4d8601 Merge branch 'master' into findpts-gslib-dev
Conflicts:
	config/config.hpp.in
	config/defaults.mk
	makefile
2019-04-25 16:10:52 -07:00
Vladimir Tomov a17f27ec8c Warnings. 2019-04-25 15:56:04 -07:00
Vladimir Tomov 877294ff95 Updated the serial example. 2019-04-25 14:40:20 -07:00
Vladimir Tomov 50a6cb1a58 Minor - changed names (to mfem conventions), added glvis vis. 2019-04-25 13:01:46 -07:00
Vladimir Tomov 821b88f238 Minor. 2019-04-24 18:57:23 -07:00
Vladimir Tomov 2f383ec7d1 Edited the output. 2019-04-24 18:51:34 -07:00
Vladimir Tomov e5a603ef54 Fixed the #include gslib. 2019-04-24 18:01:19 -07:00
Vladimir Tomov 8dce2e45a9 Minor edits. 2019-04-24 17:57:51 -07:00
Vladimir Tomov f6734ae218 Generated points over the whole domain, for all tasks.
Some changes in the output.
2019-04-24 17:06:07 -07:00
Vladimir Tomov d9b87d9581 Changes in the error computations. 2019-04-24 12:45:59 -07:00
Vladimir Tomov bfb3bb9418 Slightly modified the _eval(), added comments, fixed styling. 2019-04-23 19:06:24 -07:00
Stowell, Mark L 9ba929c93b Building libmfem-extras whenever miniapps are built 2019-04-17 11:18:10 -07:00
Stowell, Mark L 02055af32e Adding miniapps/common objects to separate library file 2019-04-17 11:13:24 -07:00
Vladimir Tomov c95a528c8c Comment. 2019-04-16 18:39:49 -07:00
Vladimir Tomov 077c812fba Comments, parameter names, styling for gslib_findpts(). 2019-04-16 18:37:40 -07:00
Vladimir Tomov 1db45ef256 Updates in gslib_findpts_setup. 2019-04-15 16:46:17 -07:00
Stowell, Mark L 9590a700fb Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-04-12 14:47:53 -07:00
Vladimir Tomov acefc1561a Minor - alignments and variable names. 2019-04-09 15:20:00 -07:00
Stowell, Mark L 05f8edfa7e Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-04-09 13:59:49 -07:00
Vladimir Tomov c338a01031 Minor. 2019-04-03 16:05:57 -07:00
Stowell, Mark L b0d33417ff Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-04-01 11:01:05 -07:00
Vladimir Tomov cdea5255a8 Restored deleted stuff from defaults.mk. 2019-03-29 19:35:57 -07:00
Vladimir Tomov c0c0a8b757 Merge branch 'master' into findpts-gslib-dev 2019-03-29 19:29:58 -07:00
Vladimir Tomov 8fa21788a9 Updated examples/gslib/README. 2019-03-29 19:20:24 -07:00
Vladimir Tomov 71e0a5c316 Deleted the old gslib tar file.
Moved the gslib folder outside mfem, as the other external libs.
Renamed the example folder to examples/gslib.
Moved the gslib code to fem/gslib.hpp and cpp.
Updated the build system for the above changes.
2019-03-29 19:12:44 -07:00
Vladimir Tomov bfc3b1a218 File findpts_mfem.h not needed. 2019-03-29 15:49:41 -07:00
Vladimir Tomov acb08fedf1 Comments. 2019-03-27 17:36:23 -07:00
Vladimir Tomov e5df36424f Fixed a build error in debug mode. 2019-03-27 17:23:52 -07:00
Ketan Mittal c4563de483 fixed to work with latest gslib 2019-03-27 09:27:44 -05:00
Vladimir Tomov 37abc5dbf7 Travis errors. 2019-03-26 14:07:30 -07:00
Vladimir Tomov bdf6aff492 Addressed PR comments. 2019-03-26 12:28:57 -07:00
Vladimir Tomov f94220c4b8 Better nonlinear inequality testing in the examples. 2019-03-26 12:07:36 -07:00
Vladimir Tomov fcec881123 Update error.cpp 2019-03-22 10:41:23 -07:00
Vladimir Tomov ee98bb915d Updated the sample runs. 2019-03-21 18:57:13 -07:00
Vladimir Tomov 998b58e65d Hid the public pointers in OptimizationProblem. 2019-03-21 18:16:26 -07:00
Stowell, Mark L de3e858f23 Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2019-03-20 15:26:15 -07:00
Stowell, Mark L 1ce87423d9 Removing extra blank line 2019-03-20 15:23:30 -07:00
Ketan Mittal 0499ae3ea9 modifying example and adding a .h file to use the latest gslib version 2019-03-10 13:38:29 -05:00
Vladimir Tomov 1d3356ff75 Fixed alignments, unused variables. 2019-03-04 18:21:10 -08:00
Vladimir Tomov a6ab978d8e Const declarations. 2019-01-08 15:16:44 -08:00
Vladimir Tomov 8877d4f3e6 Minor. 2018-12-28 15:50:16 -08:00
Vladimir Tomov a64f853136 Added hiop's rel_tol success status. 2018-12-27 18:07:30 -08:00
Vladimir Tomov 56bb088863 Added nonlinear inequalities for the hiop examples. 2018-12-20 19:31:49 -08:00
Stowell, Mark L e43ae87148 merge in latest master 2018-12-17 10:58:14 -08:00
Cosmin G. Petra 2ed99750ac added MFEM_VERIFY on the bounds for the inequality constraints 2018-12-13 07:31:22 -08:00
Vladimir Tomov 5a92f8bd37 Test with nonlinear inequalities in ex9. 2018-12-06 17:39:45 -08:00
Stowell, Mark L 311157fc82 Merge remote-tracking branch 'origin/master' into complex-mfem-dev
# Conflicts:
#	examples/CMakeLists.txt
#	examples/makefile
#	fem/linearform.hpp
#	fem/plinearform.hpp
2018-11-26 14:10:31 -08:00
Ketan Mittal 0e8c3cdd97 vectorized part where GLL mesh is setup 2018-11-25 18:20:46 -06:00
Ketan Mittal 507523526c Merge branch 'master' of https://github.com/MFEM/mfem into findpts-gslib-dev 2018-11-25 16:49:05 -06:00
Vladimir Tomov 1889bc574a Set max_iter and extract final_iter. 2018-11-14 19:03:33 -08:00
Vladimir Tomov c6e3dc871b Minor edits, added TODO items. 2018-11-14 18:50:57 -08:00
Vladimir Tomov 2544806e9f Merge branch 'master' into hiop-dev 2018-11-14 18:21:52 -08:00
Vladimir Tomov e8ab3cc5b3 Moved the reduction of the objective function. 2018-11-14 18:19:50 -08:00
Vladimir Tomov ead4fea5f1 Implemented the parallel example. 2018-11-14 17:27:30 -08:00
Stowell, Mark L 0f1e1dc2a1 Adding comments to clarify the need for these otherwise inefficient methods 2018-11-11 16:11:13 -08:00
Stowell, Mark L 17eb38b800 Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-11-08 18:01:10 -08:00
Stowell, Mark L a2d5bc0198 make style 2018-10-26 20:56:08 -07:00
Stowell, Mark L 04bcbb4456 Adding serial example 'ex21' and improving comments in 'ex21p' 2018-10-26 20:53:06 -07:00
Stowell, Mark L 5b02795032 Updating sample runs in ex21p and modifying the "clean" make target 2018-10-26 19:51:36 -07:00
Stowell, Mark L cbe5703c80 Renaming "ex21p_proposed" to "ex21p". 2018-10-26 19:47:08 -07:00
Stowell, Mark L 3c936a3c5d Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-10-26 19:43:28 -07:00
Tzanio 5aaeefc900 make style 2018-10-26 10:53:10 -07:00
Veselin Dobrev 0133e81953 In defaults.cmake, HiOp depends BLAS and LAPACK. 2018-10-23 20:54:17 -07:00
Veselin Dobrev 27e5c13ec4 A few tweaks in the build systems. 2018-10-23 20:28:39 -07:00
Veselin Dobrev f2505e9cc9 In defaults.mk, add LAPACK_LIB to the HIOP_LIB.
Remove config/user.mk.
2018-10-23 20:03:39 -07:00
Stowell, Mark L 207b0b1c71 Removing unneeded "using" declaration 2018-10-17 10:49:56 -07:00
Veselin Dobrev 78b8ac2e86 Update the Doxygen comment for the LinearForm ctor with externally
allocated data.
2018-10-16 18:03:58 -07:00
Cosmin G. Petra ccf2d3986a marked constraints for re-evaluation also in the objective and grad evaluations
most of the time, 'new_x' comes as 'true' in 'eval_f' (but not always)
2018-10-15 12:53:04 -07:00
Cosmin G. Petra 0bfb4ee8fa removed bounds relaxation from the driver
this is now done by hiop internally
2018-10-15 12:50:53 -07:00
Vladimir Tomov c7f41cd462 Updated the serial example. 2018-10-12 17:51:23 -07:00
Vladimir Tomov c4e5b0fb57 Merge branch 'master' into hiop-dev
Conflicts:
	config/config.hpp.in
2018-10-11 10:45:28 -07:00
Stowell, Mark L 401495f70f Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-10-10 21:08:13 -07:00
Vladimir Tomov d63477a4bc Merge branch 'master' into hiop-dev 2018-09-28 19:41:46 -07:00
Vladimir Tomov 95898f1ae6 Cleanup, minor edits. 2018-09-27 15:01:13 -07:00
Vladimir Tomov 7fa59c288e Moved get_vecdistr_info in the cpp, updated comments,
fixed communication error.
2018-09-27 13:48:30 -07:00
Vladimir Tomov adf040b0a1 Cleanup of things left from the old interface. 2018-09-26 13:22:22 -07:00
Vladimir Tomov 1a1f0c4b6a Updated a comment. 2018-09-26 13:14:01 -07:00
Vladimir Tomov 7b1d388729 Updated eval_Jac_cons. 2018-09-26 13:12:04 -07:00
Vladimir Tomov dd73f81b3c Updated get_starting_point. 2018-09-26 11:58:17 -07:00
Stowell, Mark L 4dadf8a5e9 Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-09-25 13:18:40 -07:00
Vladimir Tomov c26b737cf0 Defined eval_cons through the new interface. 2018-09-24 15:16:30 -07:00
Vladimir Tomov d047e98606 Simplified the OptimizationProblem interface for the objective.
Redefined some of the adapter methods to use the new interface.
2018-09-24 13:05:15 -07:00
Vladimir Tomov b8380b1979 Added links between OptimizationProblem and the various
solvers and adapters.
Renamed HiopProblemSpec to HiopOptimizationProblem.
2018-09-19 15:22:01 -07:00
Vladimir Tomov 811887c907 Added OptimizationProblem to OptimizationSolver, updated the
SLBQP interface.
2018-09-18 14:18:54 -07:00
Vladimir Tomov 4f63327b39 Simplified OptimizationProblem. 2018-09-18 11:53:39 -07:00
Vladimir Tomov 00f365e0a7 Introduced class OptimizationProblem that defines MFEM's interface
for specifying optimization problems.
2018-09-17 12:40:42 -07:00
Vladimir Tomov ead6425a9a Removed unnecessary #define. 2018-09-12 14:15:56 -07:00
Vladimir Tomov 695aed0789 Merge pull request #603 from benyee/hiop-dev
Make HiopNlpOptimizer more general
2018-09-12 14:15:06 -07:00
Ben Chung Yee 269b2c9f62 Use verbosity level of Optimizer base class, tweak comment 2018-09-11 09:04:41 -07:00
Ben Chung Yee c4ea03fc63 Use OptimizationSolver's attribute to set tolerance
OptimizationSolver::SetAbsTol can be used to set this tolerance.
2018-09-11 09:04:41 -07:00
Ben Chung Yee 95d1e5bf94 Provide a means of setting the initial guess
To set the initial guess, set use_initial_x_value to be true and provide
it via the output x vector in Mult()
2018-09-11 09:04:40 -07:00
Ben Chung Yee 283732ce94 Fix bug in linear constraint.
Constraint is w*x = a, but the code had w*x-a=a
2018-09-11 09:04:40 -07:00
Ben Chung Yee e56c5f67aa Code formatting/spacing changes
Enforce 3 indents everywhere, limit to 80 chars per line
2018-09-11 09:04:40 -07:00
Ben Chung Yee f4857335f3 Implement generalized Hiop SLBQP solver
We can now solve general quadratic objective functions.
Examples are modified to test this generalized solver.
2018-09-11 09:04:34 -07:00
Ben Chung Yee 45fb2e04ab Create shell for more general HiopNlpOptimizer class
This new class will be able to handle a general quadratic objective
function.  The old class is renamed HiopNlpOptimizer_Simple and
inherits from the more general new HiopNlpOptimizer class.
2018-09-11 09:04:12 -07:00
Stowell, Mark L dd63145272 Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-09-08 14:51:57 -07:00
Stowell, Mark L 9daae69378 Removing examples superseded by ex21p 2018-09-08 09:39:56 -07:00
Cosmin G. Petra 0b324b8130 serial driver works with hiop irrespective of whether mfem and hiop use MPI or not 2018-09-03 10:54:55 -07:00
Vladimir Tomov 6889792b67 Initial changes in the gnumake system. 2018-08-30 17:35:05 -07:00
Stowell, Mark L a56a71fc8d Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-08-29 09:28:55 -07:00
Vladimir Tomov 5340e06c4d Added some missing #ifdef MFEM_USE_MPI. 2018-08-23 19:01:03 -07:00
Vladimir Tomov 41eaca0af4 Prepared the serial example. 2018-08-23 18:27:27 -07:00
Stowell, Mark L f543df3cfa Editing header 2018-08-23 11:44:08 -07:00
Stowell, Mark L 6731ca8ba5 Allowing user to specify refinement levels 2018-08-23 11:36:45 -07:00
Stowell, Mark L 0ba15377bb Editing sample runs 2018-08-23 11:36:26 -07:00
Stowell, Mark L 5a0ffc4603 Adding a check for appropriate dimension and problem type combinations 2018-08-23 11:36:06 -07:00
Stowell, Mark L 881a00b82c Adding a complex-valued example tentatively numbered as ex21 2018-08-23 10:31:05 -07:00
Stowell, Mark L baae130a29 bugfix 2018-08-23 10:29:42 -07:00
Stowell, Mark L bbd2c56062 Adding boundary projection methods for complex grid functions 2018-08-23 10:29:27 -07:00
Stowell, Mark L 5df566608e Merge remote-tracking branch 'origin/master' into complex-mfem-dev 2018-08-22 19:16:28 -07:00
Stowell, Mark L 03a9d6968e Removing references to hertz minapp 2018-08-22 19:15:25 -07:00
Stowell, Mark L 31261cfb67 Removing new miniapp from this branch 2018-08-22 19:10:04 -07:00
Cosmin G. Petra 59a13d02c5 fixed a bug in the way the vector distribution was retrieved and passed to hiop 2018-08-22 14:03:20 -07:00
Vladimir Tomov 60ebfa350b Fixed bugs in the example. 2018-08-15 18:43:12 -07:00
Cosmin G. Petra 96adf18733 added the CMakeLists.txt for examples/hiop
ex9p seems to work fine with hiop
2018-08-15 16:50:37 -07:00
Cosmin G. Petra 24d73daf72 updated ex9p driver to work with hiop
including CMake part of the build system needed to build the driver
2018-08-15 16:45:14 -07:00
Cosmin G. Petra 1538784329 reorganized optimizer solver class and updated Hiop implementation class(es)
- added an abstract class OptimizationSolver (inheriting from IterativeSolver) common to all optimizers
- Hiop optimizers returns the solution
- spec class for Hiop updated
    - to work with local vectors
    - to correctly copy from input mfem vectors
    - compute vector distribution without relying on dynamic casts to hypre vectors
- plus other smallish changes
2018-08-15 16:39:56 -07:00
Cosmin G. Petra ae5652fdaf hiop interface class completed 2018-08-06 15:51:39 -07:00
Vladimir Tomov a8cdd28a29 Added computation of errors (exact sltn, mass, symmetry) in the driver. 2018-08-03 16:32:38 -07:00
Cosmin G. Petra 9cb1c7f5e0 complete proposed interface with hiop 2018-08-03 11:08:54 -07:00
Vladimir Tomov 32ecc69e9f Working version of the driver. 2018-07-31 18:04:16 -07:00
Vladimir Tomov 2034462372 Edits in the driver. 2018-07-30 18:35:15 -07:00
Cosmin G. Petra d27215c07b Merge branch 'hiop-dev' of github.com:mfem/mfem into hiop-dev
Conflicts:
	linalg/hiop.hpp
2018-07-30 16:02:55 -07:00
Cosmin G. Petra d3eaa8d5ac added skeleton of the class that specifies the problem to hiop 2018-07-30 15:57:49 -07:00
Cosmin G. Petra 0d31870445 updated the CMake build system for HIOP
works with HIOP_DIR specification in cmake command
provides correct include and lib paths within CMakeLists.txt
hiop interface implementation class compiles within MFEM
2018-07-30 15:48:23 -07:00
Ketan Mittal 5b3590f231 updated documentation 2018-07-26 08:25:22 -07:00
Ketan Mittal 35effbf336 minor fixes 2018-07-26 08:11:31 -07:00
Ketan Mittal 072448862a minor fixes 2018-07-25 16:30:28 -07:00
Ketan Mittal 7058549441 minor change to gslib so that it does not need to be built before mfem 2018-07-25 15:52:19 -07:00
Ketan Mittal 508a528680 minor changes to exp,exs,findpts.pdf,defaults.mk 2018-07-25 12:32:18 -07:00
Ketan Mittal fec66e45f4 minor change 2018-07-25 11:24:24 -07:00
Ketan Mittal 7e429ca104 adding makefile 2018-07-25 10:49:34 -07:00
Vladimir Tomov 536ecf89bd Created examples/hiop and started the ex9 based driver. 2018-07-24 18:13:54 -07:00
Ketan Mittal 291e0c660f updated pdf 2018-07-24 09:53:55 -07:00
Ketan Mittal 8fdb55afca modified wrapper and added documentation 2018-07-24 09:52:04 -07:00
Cosmin G. Petra d84be9789c added initial hiop solver files for mfem and updated the build system 2018-07-23 15:50:11 -07:00
Ketan Mittal 4cef947875 added gslib tar 2018-07-23 15:14:24 -07:00
Ketan Mittal b4b24ffee1 some modifications 2018-07-23 15:07:17 -07:00
Ketan Mittal 07ce3f9ba8 wrapper updated 2018-07-20 11:30:41 -07:00
Ketan Mittal 8d50f48191 minor modifications 2018-07-19 15:24:10 -07:00
Ketan Mittal 33d91942c1 minor fix to wrapper 2018-07-19 13:41:59 -07:00
Ketan Mittal f0db806fba example modified for new wrapper 2018-07-19 13:37:16 -07:00
Ketan Mittal 75faa76881 wrappers modified to be compatible with mfem Vector class 2018-07-19 13:35:58 -07:00
Ketan Mittal f787f15e79 updated for random points 2018-07-18 14:33:05 -07:00
Ketan Mittal 1d0a31c902 adding wrapper for findpts 2018-07-13 15:26:37 -07:00
Ido Akkerman a7c9df6538 Adding Adams Moulton 2018-06-12 10:01:51 +02:00
idoakkerman 8ecaad22ae Adding Adams Bashforth time intergrator 2018-06-11 17:25:52 +02:00
Stowell, Mark L 1380583449 make style 2018-03-14 14:01:58 -07:00
Stowell, Mark L a359a9b946 Changes derived from lessons learned with the HCurl damped oscillator example 2018-03-13 16:45:35 -07:00
Stowell, Mark L 61e12383f3 Making parallel visualization more simple 2018-03-13 16:44:29 -07:00
Stowell, Mark L 94f05b9c46 Adding HCurl damped oscillator example 2018-03-13 16:44:03 -07:00
Stowell, Mark L 2e680a6bfd Implementing usable Update methods for the complex FEM classes for use with AMR 2018-03-13 16:43:10 -07:00
Stowell, Mark L 964ed4530f Adding damped oscillator examples for testing 2018-03-08 10:18:14 -08:00
Stowell, Mark L 40fe63ee0d Removing tentative support for static condensation 2018-03-08 10:16:52 -08:00
Stowell, Mark L 94d9d15c1e Adding serial versions of the complex FEM classes. 2018-03-06 08:47:26 -08:00
Stowell, Mark L d9793ee7bb Passing enumeration argument by value rather than const reference. 2018-03-06 08:46:56 -08:00
Stowell, Mark L 4241426903 Moving complex FEM classes to fem/complex_fem.[ch]pp files 2018-03-05 15:10:38 -08:00
Stowell, Mark L 9bb7a6d254 Moving ComplexHypreParMatrix to complex_operator.[ch]pp files 2018-03-05 14:36:46 -08:00
Stowell, Mark L c57e4c2fe2 Adding GetType method to ComplexOperator classes 2018-03-05 14:35:57 -08:00
Stowell, Mark L 1dde0575d4 Improving comments in ComplexOperator 2018-03-05 14:35:15 -08:00
Stowell, Mark L ee334a73cb Adding accessor methods for grabbing real or imaginary part of complex operators 2018-03-05 14:34:06 -08:00
Stowell, Mark L daa1f8f0c1 Adding complex operator types to Operator::Type enumeration 2018-03-05 14:32:35 -08:00
Stowell, Mark L 57892877e1 Adding a ScaledOperator class for easy scalar multiplication of existing operators. 2018-03-05 13:39:38 -08:00
Stowell, Mark L b38f84db45 Merge remote-tracking branch 'origin/master' into hertz-dev
# Conflicts:
#	linalg/operator.cpp
#	linalg/operator.hpp
#	linalg/sparsemat.cpp
#	linalg/sparsemat.hpp
2018-03-05 13:37:45 -08:00
Stowell, Mark L 3cc20ed019 Merge remote-tracking branch 'origin/master' into hertz-dev 2018-02-28 19:39:27 -08:00
Stowell, Mark L aaa7ac7328 Adding support for different conventions in ParComplexLinearForm 2018-02-28 19:37:58 -08:00
Stowell, Mark L 41fef7e18f Adding error checking and support for BLOCK_SYMMETRIC case to FormLinearSystem 2018-02-28 19:37:17 -08:00
Stowell, Mark L 34049fa1f4 Merge remote-tracking branch 'origin/master' into hertz-dev
# Conflicts:
#	linalg/operator.hpp
2018-02-12 12:01:12 -08:00
Stowell, Mark L a6c5fee64d merge with master 2018-01-22 18:57:37 -08:00
Stowell, Mark L 7932a79ffc Merge remote-tracking branch 'origin/master' into hertz-dev
# Conflicts:
#	fem/bilinearform.hpp
#	miniapps/electromagnetics/makefile
2018-01-19 15:17:22 -08:00
Stowell, Mark L 7f4e6d38aa Initializing source vector 2018-01-02 10:59:58 -08:00
Stowell, Mark L 2ec5f36781 Improving ParSesquilinearForm::FormLinearSystem 2018-01-02 10:59:22 -08:00
Stowell, Mark L bd1a09566c Adjusting the handling of ABCs 2018-01-02 10:58:34 -08:00
Stowell, Mark L 127d20c07d Adjusting the handling of Dirichlet BCs 2018-01-02 10:57:21 -08:00
Stowell, Mark L 565e14462e make style 2018-01-02 10:55:43 -08:00
Stowell, Mark L 478ccc192f Adding boundary integral restriction based on boundary attribute 2018-01-02 10:52:51 -08:00
Stowell, Mark L 06ebf50302 Adding optional solvers 2018-01-02 10:51:19 -08:00
Stowell, Mark L 7728e2b62d Off-by-one error in ABC material coefficient 2018-01-02 10:47:36 -08:00
Stowell, Mark L 035f07d22e Adding some notes to the comment block 2018-01-02 10:46:13 -08:00
Stowell, Mark L 907b32211a Adding support for user defined surface admittance 2017-12-27 22:09:12 -08:00
Stowell, Mark L 2713f01aa9 First draft of ParSesquilinearForm::FormLinearSystem method 2017-12-27 22:06:02 -08:00
Stowell, Mark L 97a5758e85 Adding Operator::Type enumeration entries for complex operator types 2017-12-27 22:05:15 -08:00
Stowell, Mark L 00c2bcb102 Adding convention to the ParSesquilinearForm 2017-12-27 22:04:21 -08:00
Stowell, Mark L fab2df28f4 Adding boundary attribute to AddBoundaryIntegrator 2017-12-19 20:08:28 -08:00
Stowell, Mark L d1c2b1fa58 Adding first draft of boundary condition code 2017-12-19 20:05:46 -08:00
Stowell, Mark L b2e0ad2ff2 Adding solver test code 2017-12-19 20:00:41 -08:00
Stowell, Mark L 003ad1feb0 Adding sample runs 2017-12-19 19:59:13 -08:00
Stowell, Mark L f32ddb2994 Yet another bugfix... 2017-12-16 18:44:20 -08:00
Stowell, Mark L 8a34538fbc make style 2017-12-16 15:18:05 -08:00
Stowell, Mark L 868bab0d3c Add comments and remove debugging code 2017-12-16 15:16:50 -08:00
Stowell, Mark L 8ef4b02a44 Correcting the interleaving of off-diagonal columns 2017-12-16 15:03:11 -08:00
Stowell, Mark L 6ee99b4544 Fixing memory leaks 2017-12-16 01:42:43 -08:00
Stowell, Mark L 18add4dc5f Fixing (partly) offd columns 2017-12-16 01:42:24 -08:00
Stowell, Mark L 52273290b0 Adjusting data ownership 2017-12-15 14:28:22 -08:00
Stowell, Mark L e1d19bc312 Adjusting matrix data ownership 2017-12-15 14:19:25 -08:00
Stowell, Mark L d97aae1017 make style 2017-12-15 13:38:37 -08:00
Stowell, Mark L 74685235c0 Testing ComplexHypreParMatrix 2017-12-15 12:01:12 -08:00
Stowell, Mark L 8a661d1224 Implementing ComplexHypreParMatrix::GetSystemMatrix 2017-12-15 12:00:45 -08:00
Stowell, Mark L 93d6bf23ba Setting default frequency 2017-12-15 11:58:46 -08:00
Stowell, Mark L 6ad8c010a1 Changing physics constants to 'const' 2017-12-15 11:58:24 -08:00
Stowell, Mark L 4454cc8483 style change 2017-12-15 11:57:45 -08:00
Stowell, Mark L 9d9bd1b8ae Switching to ComplexHypreParMatrix return type 2017-12-15 11:57:19 -08:00
Stowell, Mark L 60b872ada7 Adding methods to access real and imaginary parts of complex operators 2017-12-13 14:58:18 -08:00
Stowell, Mark L c6edf8c571 Bugfix in ParSesquilinearForm 2017-12-13 14:39:30 -08:00
Stowell, Mark L d641040aad Bugfix to support rectangular matrices 2017-12-12 19:10:51 -08:00
Stowell, Mark L 69211e8864 Bugfix in ParComplexLinearForm 2017-12-12 18:32:02 -08:00
Stowell, Mark L 3b2e7715fc Adding ParComplexGridFunction::ParallelProject method 2017-12-12 18:31:35 -08:00
Stowell, Mark L a141e9ecae Using new method names for access real/imag parts of grid functions 2017-12-11 09:39:07 -08:00
Stowell, Mark L 99daa214f5 Add access to real and imaginary parts of objects following std::complex as an example 2017-12-11 09:38:23 -08:00
Stowell, Mark L 3eab0bf4fa bugfix 2017-12-11 08:41:47 -08:00
Stowell, Mark L 5f031e1e63 Changing name of enumeration value 2017-12-11 08:40:23 -08:00
Stowell, Mark L 8e54676401 Change convention 2017-12-11 08:37:09 -08:00
Stowell, Mark L cce81be347 make style 2017-12-11 08:34:34 -08:00
Stowell, Mark L 833dcaf496 Merge remote-tracking branch 'origin/cmplx-op-dev' into hertz-dev 2017-12-11 08:33:25 -08:00
Stowell, Mark L ec1273849a Switching to the new complex FEM objects 2017-12-10 22:07:09 -08:00
Stowell, Mark L d559e65281 Adding first draft of ParComplexGridFunction class 2017-12-10 22:06:38 -08:00
Stowell, Mark L 84c3f6c91c Adding LinearForm constructor which takes a data array 2017-12-10 22:06:11 -08:00
Stowell, Mark L 43e9fb6559 Adding first draft of ParComplexLinearForm 2017-12-10 22:05:39 -08:00
Stowell, Mark L 89142b5283 Setting up integrators and sources 2017-12-10 17:11:47 -08:00
Stowell, Mark L 4085838f3f Cleaning up compiler warning 2017-12-10 13:55:15 -08:00
Stowell, Mark L bf8a0bca62 Adding first draft of ParSesquilinearForm class 2017-12-10 13:54:44 -08:00
Stowell, Mark L 065e54fdbb Preparing for the ParSesquilinearForm 2017-12-09 22:05:30 -08:00
Stowell, Mark L 2f856345db Copy-n-paste from Tesla 2017-12-03 14:15:46 -08:00
Stowell, Mark L 26c38ed953 Adding new miniapp to makefile 2017-12-02 22:59:44 -08:00
Stowell, Mark L bef80ef700 Adding initial miniapp files 2017-12-02 22:59:28 -08:00
184 changed files with 29567 additions and 1001 deletions
+15
View File
@@ -9,6 +9,7 @@
# Object and library files
*.o
/libmfem.*
/miniapps/common/libmfem-common.*
# CMake generated files
CMakeCache.txt
@@ -53,6 +54,7 @@ examples/displaced.mesh
examples/mesh.*
examples/ex5.mesh
examples/Example5*
examples/PVExample*
examples/Example9*
examples/Example15*
examples/Example16*
@@ -60,6 +62,8 @@ examples/sphere_refined.*
examples/sol.*
examples/sol_u.*
examples/sol_p.*
examples/sol_r.*
examples/sol_i.*
examples/ex9.mesh
examples/ex9-mesh.*
examples/ex9-init.*
@@ -113,6 +117,7 @@ examples/petsc/sol.*
examples/petsc/sol_p.*
examples/petsc/sol_u.*
examples/petsc/Example5*
examples/petsc/ex9.mesh
examples/petsc/ex9-mesh.*
examples/petsc/ex9-init.*
examples/petsc/ex9-final.*
@@ -124,6 +129,11 @@ examples/petsc/elastic_energy.*
examples/pumi/ex1
examples/pumi/ex[126]p
examples/hiop/ex9.mesh
examples/hiop/ex9-mesh.*
examples/hiop/ex9-init.*
examples/hiop/ex9-final.*
examples/pumi/refined.mesh
examples/pumi/sol.gf
examples/pumi/mesh.*
@@ -170,6 +180,7 @@ miniapps/tools/display-basis
miniapps/tools/load-dc
miniapps/tools/convert-dc
miniapps/tools/lor-transfer
miniapps/tools/get-values
miniapps/nurbs/ex1
miniapps/nurbs/ex1p
@@ -180,6 +191,10 @@ miniapps/nurbs/sol.*
miniapps/nurbs/mode_*
miniapps/nurbs/Example1*
miniapps/gslib/field-diff
miniapps/gslib/findpts
miniapps/gslib/pfindpts
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
+113 -16
View File
@@ -1,28 +1,114 @@
sudo: false
language: cpp
matrix:
sudo: false
stages:
- checks
- tests
- optional
jobs:
include:
#
# Linux
#
- os: linux
# ========================
# Checks
# ========================
# - code-style
# - documentation
# - gitignore
- stage: checks
os: linux
name: "code-style"
addons:
apt:
packages:
- astyle=2.05.1-0ubuntu1
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
- ./runtest code-style
- stage: checks
os: linux
name: "documentation"
addons:
apt:
packages:
- doxygen
- graphviz
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
- ./runtest documentation
- stage: checks
os: linux
name: "gitignore"
addons:
apt:
packages:
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
- make all -j3
- make test-noclean
- cd tests/scripts
- ./runtest gitignore
# ========================
# Optional Checks/Tests
# ========================
# - branch-history
- stage: optional
name: "branch-history"
# need full git history for the binary/big files check
git:
depth: false
script:
- cd ${TRAVIS_BUILD_DIR}
# update master
- git fetch origin master:master
# checkout a branch (otherwise Travis works in detached head)
- git checkout -b travis_tests
- cd tests/scripts
- ./runtest branch-history
# ========================
# Linux tests
# ========================
# - serial + debug
# - serial
# - parallel + debug
# - parallel
- stage: tests
os: linux
compiler: gcc
name: "Linux: Serial + Debug"
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: linux
compiler: gcc
name: "Linux: Serial"
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: linux
compiler: gcc
name: "Linux: Parallel + Debug"
addons:
apt:
# sources:
@@ -49,9 +135,10 @@ matrix:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: linux
compiler: gcc
name: "Linux: Parallel"
addons:
apt:
# sources:
@@ -78,28 +165,37 @@ matrix:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
# Mac OS X
#
# ========================
# Mac OS X tests
# ========================
# - serial + debug
# - serial
# - parallel + debug
# - parallel
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial + Debug"
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial"
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel + Debug"
env: DEBUG=YES
MPI=YES
CODECOV=NO
@@ -115,10 +211,11 @@ matrix:
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
#
- os: osx
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel"
env: DEBUG=NO
MPI=YES
CODECOV=YES
+126 -13
View File
@@ -13,48 +13,161 @@ Version 4.0.1 (development)
Improved GPU support
--------------------
- Added support for matrix-free diagonal smoothers on GPUs.
- Added initial support for AMD GPUs based on HIP: a C++ runtime API and kernel
language that can run on both AMD and NVIDIA hardware. With this change, the
list of backends is: "occa-cuda", "raja-cuda", "cuda", "hip", "occa-omp",
"raja-omp", "omp", "occa-cpu", "raja-cpu", and "cpu".
language that can run on both AMD and NVIDIA hardware. With this change and
the libCEED addition below, the current list of available backends is:
"ceed-cuda", "occa-cuda", "raja-cuda", "cuda", "hip", "occa-omp", "raja-omp",
"omp", "ceed-cpu", "occa-cpu", "raja-cpu", and "cpu".
- Improved RAJA backend and multi-GPU MPI communications.
Discretization improvements
---------------------------
- Added support for non-conforming prism AMR, including coarsening and parallel
load balancing. Anisotropic prism refinement is only available in the serial
version at the moment.
libCEED support
---------------
- Added support for libCEED, the portable library for high-order operator
evaluation developed by the Center for Efficient Exascale Discretizations in
the Exascale Computing Project, https://github.com/CEED/libCEED.
- This initial integration includes Mass and Diffusion integrators. libCEED GPU
backends can be used without specific MFEM configuration, however it is highly
recommended to use the "cuda" build option to minimize memory transfers.
- Both CPU and GPU modes are available as MFEM device backends (ceed-cpu and
ceed-cuda), using some of the best performing CPU and GPU backends from
libCEED, see the sample runs in examples 1 and 6.
Meshing improvements
--------------------
- Added support for non-conforming AMR on prisms and tetrahedra, including
coarsening and parallel load balancing. Anisotropic prism refinement is only
available in the serial version at the moment.
- The TMOP mesh optimization algorithms were extended to support r-adaptivity.
Target matrices can now be constructed either via a given analytical function
(e.g. spatial dependence of size, aspect ratio, etc., for each element) or via
a (Par)GridFunction specified on the original mesh.
- The TMOP mesh optimization algorithms have been improved to support AMR meshes.
- New method Mesh::GetHilbertElementOrdering for sorting mesh elements along the
Hilbert curve. The ordering can be used to improve caching and parallel
partitioning in non-conforming AMR.
- Added support for creating refined versions of periodic meshes, making use of
the new L2ElementRestriction class. This class also allows for computing
geometric factors on periodic meshes using partial assembly.
- The TMOP mesh optimization algorithms have been improved to support AMR meshes.
- Improved element numbering after uniform mesh refinement.
Discretization improvements
---------------------------
- Added support for GSLIB-FindPoints, a general high-order interpolation utility
that can robustly evaluate a GridFunction in an arbitrary collection of points
in physical space. See INSTALL for details on building MFEM with GSLIB, and
miniapps/gslib for examples of how to use this feature.
- Added support for complex-valued finite element operators and fields using a
2x2 block structured linear system to mimic complex arithmetic. New classes
include: ComplexGridFunction, SesquilinearForm, ComplexLinearForm, and their
parallel counterparts.
- Two integrators related to Stokes problems, (Q grad u, v) and (Q div v, u),
where u and the components of v are in H1, were added/modified to support full
and partial assembly modes. See the new GradientIntegrator and the updated
VectorDivergenceIntegrator classes in fem/bilininteg.hpp, as well as the PA
kernels in fem/bilininteg_gradient.cpp and fem/bilininteg_divergence.cpp.
- Diagonals of partially assembled operators can now be computed efficiently.
See the new methods AssembleDiagonal in BilinearForm, AssembleDiagonalPA in
BilinearFormIntegrator and the implementations in fem/bilininteg_*.cpp.
- Added initial support for NonlinearForms to support the partial assembly mode.
- Added a nonlinear vector valued convection integrator (Q u \cdot grad u, v)
where u_i and v_i are in H1. This form occurs e.g. in the Navier-Stokes
equations. The integrator supports the partial assembly mode for its
action. In full assembly mode we also provide the GetGradient method that
computes the linearized version of the integrator.
- Added a new method, MixedBilinearForm::FormRectangularLinearSystem, that can
be used to impose boundary conditions on the non-square off-diagonal blocks of
a block operator (similar to FormLinearSystem in the square case).
- Extended the support for partial assembly to vector mass and vector diffusion
bilinear integrators.
Linear and nonlinear solvers
----------------------------
- Added a general interface for specifying and solving nonlinear constrained
optimization problems through the new classes OptimizationProblem and
OptimizationSolver, see linalg/solver.hpp
- Added support for HiOp, a lightweight HPC solver for nonlinear optimization
problems see class HiOpNLPOptimizer and the example codes in examples/hiop.
- Added support for Ginkgo, a high-performance linear algebra library for GPU
and manycore nodes, with a focus on sparse solution of linear systems. For
more details see linalg/ginkgo.hpp and the example code in examples/gingko.
- Added Adams-Bashforth and Adams-Moulton time integrators.
New and updated examples and miniapps
-------------------------------------
- Added two new miniapps: Find Points (serial + parallel) and Field Diff in
miniapps/gslib that show how GSLIB-FindPoints can be used to interpolate a
(Par)GridFunction in an arbitrary number of physical space points in 2D and
3D. The GridFunction must be in H1 and in the same space as the mesh that is
used to find the points.
- Added a new example, Example 22/22p, to demonstrate the use of the new
complex-valued finite element operators. The example defines and solves
a family of time-harmonic PDEs related to damped harmonic oscillators.
- Updated Example 1/1p to use diagonal preconditioning in partial assembly mode.
- The mesh-optimizer and pmesh-optimizer miniapps have been updated to
demonstrate the new r-adaptivity capabilities of TMOP.
- New options to reorder and partition the mesh in the mesh-explorer miniapp.
- The (p)mesh-optimizer miniapp has been updated to demonstrate mesh
optimization for an AMR mesh.
- Added a modification of Example 1 in examples/ginkgo that demonstrates the use
of the Gingko interface to solve a linear system.
- Added a modification of ex9 in examples/hiop that demonstrates the nonlinear
constrained optimization interface and the use of the SLBQP and HiOp solvers.
Improved testing
----------------
- Added a new directory, tests/scripts, with several shell scripts that perform
simple checks on the code including: code styling, documentation formatting,
proper use of .gitignore, and preventing the accidental commit of large files.
- It is recommended that developers run the above tests scripts (via the runtest
script) before pushing to GitHub. See the README file in tests/scripts.
- The Travis CI settings have been updated to include an initial Checks stage
which currently runs the code-style, documentation and gitignore test scripts,
as well as a final stage for optional checks/tests which currently runs the
branch-history script.
Miscellaneous
-------------
- Upgraded the SUNDIALS interface to utilize SUNDIALS version 5.0. This
necessitated a complete rework of the interface and requires changes at
the application level. Example usage of this new interface can be found
in the examples/sundials directory.
- Upgraded the SUNDIALS interface to utilize version 5.0. This necessitated a
complete rework of the interface and requires changes at the application
level. Example usage of the new interface can be found in examples/sundials.
- Added support for output in the ParaView XML format. See Examples 5/5p, 9/9p
and the new ParaViewDataCollection class.
- Collected object files from the miniapps/common directory into a new library,
libmfem-common for the convenience of application developers. The new library
is now used in several miniapps in the electromagnetic and tools directories.
- Added unit tests for time integrators.
Version 4.0, released on May 24, 2019
+28 -2
View File
@@ -164,6 +164,15 @@ if (MFEM_USE_METIS)
find_package(METIS REQUIRED)
endif()
if (MFEM_USE_GINKGO)
find_package(Ginkgo REQUIRED)
if (Ginkgo_FOUND)
get_target_property(Ginkgo_INCLUDE_DIRS
Ginkgo::ginkgo INTERFACE_INCLUDE_DIRECTORIES)
set(Ginkgo_LIBRARIES Ginkgo::ginkgo)
endif()
endif()
# GZSTREAM -> zlib
if (MFEM_USE_GZSTREAM)
find_package(ZLIB REQUIRED)
@@ -248,6 +257,10 @@ if (MFEM_USE_MPFR)
find_package(MPFR REQUIRED)
endif()
if (MFEM_USE_CEED)
find_package(libCEED REQUIRED)
endif()
if (MFEM_USE_CONDUIT)
find_package(Conduit REQUIRED conduit relay blueprint )
endif()
@@ -273,6 +286,12 @@ if (MFEM_USE_PUMI)
endif()
endif()
# HiOp optimizer
if (MFEM_USE_HIOP)
find_package(HIOP REQUIRED)
# find_package updates HIOP_FOUND, HIOP_INCLUDE_DIRS, HIOP_LIBRARIES
endif()
# CUDA
if (MFEM_USE_CUDA)
set(CMAKE_CUDA_STANDARD 11)
@@ -322,8 +341,8 @@ endif()
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
# be before SuiteSparse.
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR PUMI
POSIXCLOCKS MFEMBacktrace ZLIB OCCA RAJA)
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO Ginkgo GNUTLS NETCDF MPFR
PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
@@ -535,6 +554,13 @@ if (MFEM_USE_OCCA)
FILES_MATCHING PATTERN "*.okl")
endif()
# Install the libCEED files
if (MFEM_USE_CEED)
install(DIRECTORY ${MFEM_SOURCE_DIRS}
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
FILES_MATCHING PATTERN "fem/libceed/*.h")
endif()
# Install ${HEADERS}
# ---
# foreach (HDR ${HEADERS})
+5
View File
@@ -83,6 +83,8 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
│ └── web
│ └── examples
├── examples
│ ├── ginkgo
│ ├── hiop
│ ├── petsc
│ ├── pumi
│ └── sundials
@@ -93,6 +95,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
├── miniapps
│ ├── common
│ ├── electromagnetics
│ ├── gslib
│ ├── meshing
│ ├── nurbs
│ ├── performance
@@ -350,6 +353,7 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
@@ -364,6 +368,7 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
+57 -7
View File
@@ -349,6 +349,12 @@ MFEM_USE_STRUMPACK = YES/NO
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_GINKGO = YES/NO
Enable MFEM functionality based on the Ginkgo library, which provides
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
https://github.com/ginkgo-project/ginkgo. When enabled, the user can use
Ginkgo's solvers and preconditioners as shown in examples/ginkgo/.
MFEM_USE_GNUTLS = YES/NO
Enable secure socket support in class socketstream, using the auxiliary
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
@@ -415,6 +421,10 @@ MFEM_USE_PUMI = YES/NO
models and effectively supports automated adaptive analysis. PUMI enables
support for parallel unstructured mesh modifications in MFEM.
MFEM_USE_HIOP = YES/NO
Enable the usage of HiOp (https://github.com/LLNL/hiop) in MFEM. HiOp is an
HPC solver for nonlinear optimization problems.
MFEM_USE_CUDA = YES/NO
Enables support for CUDA devices in MFEM. CUDA is a parallel computing
platform and programming model for general computing on graphical processing
@@ -442,6 +452,17 @@ MFEM_USE_OCCA = YES/NO
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_GSLIB = YES/NO
Enables MFEM functionality based on the GSLIB library, and specifically its
FindPoints component, which provides a robust algorithms to evaluate finite
element functions in a collection of points in physical space. When enabled,
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
MFEM_USE_CEED = YES/NO
Enables support for the libCEED library in MFEM. libCEED is a portable
library for performant high-order operator evaluation developed by the Center
for Efficient Exascale Discretizations in the Exascale Computing Project.
MFEM_BUILD_TAG = (any value)
An optional tag to characterize the build. Exported to config/config.mk.
Can be used to identify the MFEM build from other makefiles.
@@ -518,6 +539,12 @@ The specific libraries and their options are:
URL: http://portal.nersc.gov/project/sparse/strumpack
Options: STRUMPACK_OPT, STRUMPACK_LIB.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
C++ compiler that supports the C++-11 standard. For additional requirements
and dependencies of specific modules see the Ginkgo webpage below.
URL: https://ginkgo-project.github.io
Options: GINKGO_OPT (Not used), GINKGO_LIB.
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
one can get the library through the Homebrew package manager (http://brew.sh).
@@ -549,29 +576,48 @@ The specific libraries and their options are:
https://support.hdfgroup.org/HDF5 (HDF5)
Options: SIDRE_OPT, SIDRE_LIB.
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
- Conduit (optional), used when MFEM_USE_CONDUIT = YES. Conduit Mesh Blueprint
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
URL: https://github.com/LLNL/conduit (Conduit)
https://support.hdfgroup.org/HDF5 (HDF5)
Options: CONDUIT_OPT, CONDUIT_LIB.
- PUMI, used when MFEM_USE_PUMI = YES.
- PUMI (optional), used when MFEM_USE_PUMI = YES.
URL: https://scorec.rpi.edu/pumi
Options: PUMI_OPT, PUMI_LIB.
- CUDA, used when MFEM_USE_CUDA = YES.
- HiOp (optional), used when MFEM_USE_HIOP = YES.
URL: https://github.com/LLNL/hiop
Options: HIOP_OPT, HIOP_LIB.
- GSLIB (optional), used when MFEM_USE_GSLIB = YES. The gslib library must be
built prior to the MFEM build, as follows: download gslib-1.0.5, untar it at
the same level as MFEM and create a symbolic link: "ln -s gslib-1.0.5 gslib".
Build gslib in parallel or in serial based on the desired MFEM build: "make
clean; make CC=mpicc" or "make clean; make CC=gcc MPI=0". Build MFEM with
MFEM_USE_GSLIB=YES.
URL: https://github.com/gslib/gslib/archive/v1.0.5.tar.gz
Options: GSLIB_OPT, GSLIB_LIB.
- CUDA (optional), used when MFEM_USE_CUDA = YES.
URL: https://developer.nvidia.com/cuda-toolkit
Options: CUDA_CXX, CUDA_ARCH, CUDA_OPT, CUDA_LIB.
- HIP, used when MFEM_USE_HIP = YES.
- HIP (optional), used when MFEM_USE_HIP = YES.
URL: https://rocm.github.io/ROCmInstall.html
Options: HIP_CXX, HIP_ARCH, HIP_OPT, HIP_LIB.
- OCCA, used when MFEM_USE_OCCA = YES.
- OCCA (optional), used when MFEM_USE_OCCA = YES.
URL: https://libocca.org
Options: OCCA_DIR, OCCA_OPT, OCCA_LIB.
- RAJA, used when MFEM_USE_RAJA = YES.
- libCEED (optional), used when MFEM_USE_CEED = YES. Requires libCEED's master
branch, specifically, git-hash c00ee0d or later.
URL: https://github.com/CEED/libCEED
https://ceed.exascaleproject.org/libceed
Options: CEED_DIR, CEED_OPT, CEED_LIB.
- RAJA (optional), used when MFEM_USE_RAJA = YES.
Beginning with MFEM v4.1, only RAJA v0.10.0+ is supported.
URL: https://github.com/LLNL/RAJA
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
@@ -591,7 +637,6 @@ The specific libraries and their options are:
URL: https://zlib.net
Options: ZLIB_OPT, ZLIB_LIB.
Building with CMake
===================
The MFEM build system consists of two steps: configuration and compilation.
@@ -707,13 +752,16 @@ MFEM_USE_MESQUITE
MFEM_USE_SUITESPARSE
MFEM_USE_SUPERLU
MFEM_USE_STRUMPACK
MFEM_USE_GINKGO
MFEM_USE_GNUTLS
MFEM_USE_NETCDF
MFEM_USE_MPFR
MFEM_USE_GZSTREAM
MFEM_USE_PUMI
MFEM_USE_HIOP
MFEM_USE_CUDA
MFEM_USE_OCCA
MFEM_USE_CEED
MFEM_USE_RAJA
MFEM_USE_SIDRE
@@ -756,12 +804,14 @@ The CMake build system adds auto-detection for the following packages/libraries:
- SuiteSparse
- SuperLUDist, STRUMPACK
- ParMETIS
- Ginkgo
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
- NETCDF
- MPFR
- LIBUNWIND
- POSIXCLOCKS
- PUMI
- HIOP
- OCCA
- RAJA
- AXOM - Used when MFEM_USE_SIDRE is enabled
+4
View File
@@ -232,6 +232,10 @@ IF (DEFINED TPL_ENABLE_GECKO)
SET(MFEM_USE_GECKO ${TPL_ENABLE_GECKO} CACHE BOOL "Enable GECKO usage" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_GINKGO)
SET(MFEM_USE_GINKGO ${TPL_ENABLE_GINKGO} CACHE BOOL "Enable GINKGO usage" FORCE)
ENDIF()
IF (DEFINED TPL_ENABLE_GNUTLS)
SET(MFEM_USE_GNUTLS ${TPL_ENABLE_GNUTLS} CACHE BOOL "Enable GNUTLS usage" FORCE)
ENDIF()
+2
View File
@@ -34,6 +34,7 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
@@ -44,6 +45,7 @@ set(MFEM_USE_PUMI @MFEM_USE_PUMI@)
set(MFEM_USE_CUDA @MFEM_USE_CUDA@)
set(MFEM_USE_OCCA @MFEM_USE_OCCA@)
set(MFEM_USE_RAJA @MFEM_USE_RAJA@)
set(MFEM_USE_CEED @MFEM_USE_CEED@)
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
+9
View File
@@ -92,6 +92,9 @@
// Enable functionality based on the Gecko library
#cmakedefine MFEM_USE_GECKO
// Enable functionality based on the Ginkgo library
#cmakedefine MFEM_USE_GINKGO
// Enable MFEM functionality based on the GnuTLS library
#cmakedefine MFEM_USE_GNUTLS
@@ -110,6 +113,9 @@
// Enable MFEM functionality based on the PUMI library
#cmakedefine MFEM_USE_PUMI
// Enable MFEM functionality based on the HiOp library
#cmakedefine MFEM_USE_HIOP
// Build the GPU/CUDA-enabled version of the MFEM library.
// Requires a CUDA compiler (nvcc).
#cmakedefine MFEM_USE_CUDA
@@ -120,6 +126,9 @@
// Enable MFEM functionality based on the OCCA library
#cmakedefine MFEM_USE_OCCA
// Enable MFEM functionality based on the libCEED library
#cmakedefine MFEM_USE_CEED
// Which library functions to use in class StopWatch for measuring time.
// For a list of the available options, see INSTALL.
// If not defined, an option is selected automatically.
+36
View File
@@ -0,0 +1,36 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Sets the following variables:
# - HIOP_FOUND
# - HIOP_INCLUDE_DIRS
# - HIOP_LIBRARIES
include(MfemCmakeUtilities)
mfem_find_package(HIOP HIOP HIOP_DIR
"include" "hiopInterface.hpp"
"lib" "hiop"
"Paths to headers required by HIOP."
"Libraries required by HIOP.")
# this test fails with parallel MFEM since mpi.h is not available (cxx compiler is used for some reason)
# CHECK_BUILD HIOP_VERSION_OK TRUE
#"
##include <hiopInterface.hpp>
#using namespace hiop;
#int main(int argc, char *argv[])
#{
# MPI_Init(&argc, &argv);
# MPI_Comm comm = MPI_COMM_WORLD;
#
# return 0;
#}
#")
+19
View File
@@ -0,0 +1,19 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Defines the following variables:
# - CEED_FOUND
# - CEED_LIBRARIES
# - CEED_INCLUDE_DIRS
include(MfemCmakeUtilities)
mfem_find_package(libCEED CEED CEED_DIR "include" ceed.h "lib" ceed
"Paths to headers required by libCEED." "Libraries required by libCEED.")
@@ -814,7 +814,7 @@ function(mfem_export_mk_files)
foreach(lib ${TPL_LIBRARIES})
get_filename_component(suffix ${lib} EXT)
# handle interfaces (e.g., SCOREC::apf)
if ("${lib}" MATCHES "SCOREC::.*")
if ("${lib}" MATCHES "SCOREC::.*" OR "${lib}" MATCHES "Ginkgo::.*")
elseif (NOT "${lib}" MATCHES "SCOREC::.*" AND "${lib}" MATCHES ".*::.*")
message(FATAL_ERROR "***** interface lib found ... exiting *****")
# handle static and shared libs
+12
View File
@@ -100,6 +100,9 @@
// Enable functionality based on the Gecko library
// #define MFEM_USE_GECKO
// Enable MFEM features based on the Ginkgo library
// #define MFEM_USE_GINKGO
// Enable secure socket streams based on the GNUTLS library
// #define MFEM_USE_GNUTLS
@@ -121,6 +124,12 @@
// Enable MFEM functionality based on the PUMI library
// #define MFEM_USE_PUMI
// Enable MFEM functionality based on the HIOP library.
// #define MFEM_USE_HIOP
// Enable MFEM functionality based on the GSLIB library
// #define MFEM_USE_GSLIB
// Build the NVIDIA GPU/CUDA-enabled version of the MFEM library.
// Requires a CUDA compiler (nvcc).
// #define MFEM_USE_CUDA
@@ -135,6 +144,9 @@
// Enable functionality based on the OCCA library.
// #define MFEM_USE_OCCA
// Enable functionality based on the libCEED library.
// #define MFEM_USE_CEED
// Version of HYPRE used for building MFEM.
// #define MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
+4
View File
@@ -34,6 +34,7 @@ MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_GECKO = @MFEM_USE_GECKO@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
MFEM_USE_NETCDF = @MFEM_USE_NETCDF@
MFEM_USE_PETSC = @MFEM_USE_PETSC@
@@ -41,10 +42,13 @@ MFEM_USE_MPFR = @MFEM_USE_MPFR@
MFEM_USE_SIDRE = @MFEM_USE_SIDRE@
MFEM_USE_CONDUIT = @MFEM_USE_CONDUIT@
MFEM_USE_PUMI = @MFEM_USE_PUMI@
MFEM_USE_HIOP = @MFEM_USE_HIOP@
MFEM_USE_GSLIB = @MFEM_USE_GSLIB@
MFEM_USE_CUDA = @MFEM_USE_CUDA@
MFEM_USE_HIP = @MFEM_USE_HIP@
MFEM_USE_RAJA = @MFEM_USE_RAJA@
MFEM_USE_OCCA = @MFEM_USE_OCCA@
MFEM_USE_CEED = @MFEM_USE_CEED@
# Compiler, compile options, and link options
MFEM_CXX = @MFEM_CXX@
+20
View File
@@ -35,6 +35,7 @@ option(MFEM_USE_SUITESPARSE "Enable SuiteSparse usage" OFF)
option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_GECKO "Enable GECKO usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_GNUTLS "Enable GNUTLS usage" OFF)
option(MFEM_USE_NETCDF "Enable NETCDF usage" OFF)
option(MFEM_USE_PETSC "Enable PETSc support." OFF)
@@ -42,9 +43,13 @@ option(MFEM_USE_MPFR "Enable MPFR usage." OFF)
option(MFEM_USE_SIDRE "Enable Axom/Sidre usage" OFF)
option(MFEM_USE_CONDUIT "Enable Conduit usage" OFF)
option(MFEM_USE_PUMI "Enable PUMI" OFF)
option(MFEM_USE_HIOP "Enable HiOp" OFF)
option(MFEM_USE_CUDA "Enable CUDA" OFF)
option(MFEM_USE_OCCA "Enable OCCA" OFF)
option(MFEM_USE_RAJA "Enable RAJA" OFF)
option(MFEM_USE_CEED "Enable CEED" OFF)
option(MFEM_USE_ADEPT "Enable AD using ADEPT" OFF)
option(MFEM_USE_CODIPACK "Enable AD using CoDiPack" OFF)
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
@@ -136,6 +141,8 @@ set(ScaLAPACK_TARGET_NAMES scalapack)
set(GECKO_DIR "${MFEM_DIR}/../gecko" CACHE PATH "Path to the Gecko library.")
set(Ginkgo_DIR "${MFEM_DIR}/../ginkgo" CACHE PATH "Path to the Ginkgo library.")
set(GNUTLS_DIR "" CACHE PATH "Path to the GnuTLS library.")
set(NETCDF_DIR "" CACHE PATH "Path to the NetCDF library.")
@@ -160,14 +167,27 @@ set(Axom_REQUIRED_PACKAGES "Conduit/relay/blueprint" CACHE STRING
set(PUMI_DIR "${MFEM_DIR}/../pumi-2.1.0" CACHE STRING
"Directory where PUMI is installed")
set(HIOP_DIR "${MFEM_DIR}/../hiop/install" CACHE STRING
"Directory where HiOp is installed")
set(HIOP_REQUIRED_PACKAGES "BLAS" "LAPACK" CACHE STRING
"Packages that HiOp depends on.")
set(OCCA_DIR "${MFEM_DIR}/../occa" CACHE PATH "Path to OCCA")
set(RAJA_DIR "${MFEM_DIR}/../raja" CACHE PATH "Path to RAJA")
set(CEED_DIR "${MFEM_DIR}/../libCEED" CACHE PATH "Path to libCEED")
set(BLAS_INCLUDE_DIRS "" CACHE STRING "Path to BLAS headers.")
set(BLAS_LIBRARIES "" CACHE STRING "The BLAS library.")
set(LAPACK_INCLUDE_DIRS "" CACHE STRING "Path to LAPACK headers.")
set(LAPACK_LIBRARIES "" CACHE STRING "The LAPACK library.")
set(ADEPT_INCLUDE_DIRS "${MFEM_DIR}/../adept-1.1op/include" CACHE STRING "Path to ADEPT headers.")
set(ADEPT_LIBRARIES "-L${MFEM_DIR}/../adept-1.1op/lib -ladept" CACHE STRING "The ADEPT library.")
set(CODIPACK_INCLUDE_DIRS "${MFEM_DIR}/../CoDiPack/include" CACHE STRING "Path to CoDiPack headers.")
# Some useful variables:
set(CMAKE_SKIP_PREPROCESSED_SOURCE_RULES ON) # Skip *.i rules
set(CMAKE_SKIP_ASSEMBLY_SOURCE_RULES ON) # Skip *.s rules
+24
View File
@@ -122,6 +122,7 @@ MFEM_USE_SUITESPARSE = NO
MFEM_USE_SUPERLU = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_GECKO = NO
MFEM_USE_GINKGO = NO
MFEM_USE_GNUTLS = NO
MFEM_USE_NETCDF = NO
MFEM_USE_PETSC = NO
@@ -129,10 +130,13 @@ MFEM_USE_MPFR = NO
MFEM_USE_SIDRE = NO
MFEM_USE_CONDUIT = NO
MFEM_USE_PUMI = NO
MFEM_USE_HIOP = NO
MFEM_USE_GSLIB = NO
MFEM_USE_CUDA = NO
MFEM_USE_HIP = NO
MFEM_USE_RAJA = NO
MFEM_USE_OCCA = NO
MFEM_USE_CEED = NO
# Compile and link options for zlib.
ZLIB_DIR =
@@ -244,6 +248,11 @@ GECKO_DIR = @MFEM_DIR@/../gecko
GECKO_OPT = -I$(GECKO_DIR)/inc
GECKO_LIB = -L$(GECKO_DIR)/lib -lgecko
# Ginkgo library configuration (currently not needed)
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
GINKGO_OPT = -isystem $(GINKGO_DIR)/include
GINKGO_LIB = $(XLINKER)-rpath,$(GINKGO_DIR)/lib -L$(GINKGO_DIR)/lib -lginkgo -lginkgo_omp -lginkgo_cuda -lginkgo_reference
# GnuTLS library configuration
GNUTLS_OPT =
GNUTLS_LIB = -lgnutls
@@ -308,6 +317,16 @@ PUMI_OPT = -I$(PUMI_DIR)/include
PUMI_LIB = -L$(PUMI_DIR)/lib -lpumi -lcrv -lma -lmds -lapf -lpcu -lgmi -lparma\
-llion -lmth -lapf_zoltan -lspr
# HIOP
HIOP_DIR = @MFEM_DIR@/../hiop/install
HIOP_OPT = -I$(HIOP_DIR)/include
HIOP_LIB = -L$(HIOP_DIR)/lib -lhiop $(LAPACK_LIB)
# GSLIB library
GSLIB_DIR = @MFEM_DIR@/../gslib/build
GSLIB_OPT = -I$(GSLIB_DIR)/include
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
# CUDA library configuration (currently not needed)
CUDA_OPT =
CUDA_LIB =
@@ -321,6 +340,11 @@ OCCA_DIR = @MFEM_DIR@/../occa
OCCA_OPT = -I$(OCCA_DIR)/include
OCCA_LIB = $(XLINKER)-rpath,$(OCCA_DIR)/lib -L$(OCCA_DIR)/lib -locca
# libCEED library configuration
CEED_DIR ?= @MFEM_DIR@/../libCEED
CEED_OPT = -I$(CEED_DIR)/include
CEED_LIB = $(XLINKER)-rpath,$(CEED_DIR)/lib -L$(CEED_DIR)/lib -lceed
# RAJA library configuration
RAJA_DIR = @MFEM_DIR@/../raja
RAJA_OPT = -I$(RAJA_DIR)/include
+35 -1
View File
@@ -152,6 +152,15 @@ function extract_sample_runs()
fi
if [ "$device_runs" == "yes" ]; then
runs=`printf "%s" "$runs" | grep ".* -d .*"`
if [ "$have_occa" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d occa-.*"`
fi
if [ "$have_raja" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d raja-.*"`
fi
if [ "$have_ceed" == "no" ]; then
runs=`printf "%s" "$runs" | grep -v ".* -d ceed-.*"`
fi
else
runs=`printf "%s" "$runs" | grep -v ".* -d .*"`
fi
@@ -276,7 +285,8 @@ case "$1" in
;;
-dev)
device_runs="yes"
mfem_config+=" MFEM_USE_CUDA=YES MFEM_USE_OCCA=YES MFEM_USE_RAJA=YES MFEM_USE_OPENMP=YES"
mfem_config+=" MFEM_USE_CUDA=YES MFEM_USE_OPENMP=YES"
# OCCA, RAJA, libCEED are enabled below, if available
;;
-v)
valgrind="yes"
@@ -457,6 +467,30 @@ fi
TIMEFORMAT="${base_timeformat}"
# Setup optional libraries when not using externally built MFEM:
if [ "${built}" == "no" ]; then
have_occa="no"
have_raja="no"
have_ceed="no"
if [ "${device_runs}" == "yes" ]; then
if [ -n "${CUDA_ARCH}" ]; then
mfem_config+=" CUDA_ARCH=${CUDA_ARCH}"
fi
if [ -d "${mfem_dir}/../occa" ]; then
mfem_config+=" MFEM_USE_OCCA=YES"
have_occa="yes"
fi
if [ -d "${mfem_dir}/../raja" ]; then
mfem_config+=" MFEM_USE_RAJA=YES"
have_raja="yes"
fi
if [ -d "${mfem_dir}/../libCEED" ]; then
mfem_config+=" MFEM_USE_CEED=YES"
have_ceed="yes"
fi
fi
fi
function set_echo_log()
{
local dirname=`dirname "$1"`
+1
View File
@@ -82,6 +82,7 @@ test-par-NO: $(SEQ_$(MFEM_TESTS):=-test-seq)
test-ser: test-par-NO
test-par: test-par-YES
test: all test-par-$(MFEM_USE_MPI) clean-exec
test-noclean: all test-par-$(MFEM_USE_MPI)
test-clean: ; @rm -f *.stderr
test-print: mfem-test=printf " $(3) [$(2) ./$(1) -no-vis $(if $(4),$(4) )]\n"
test-print: mfem-test-file=printf " $(3) [$(2) ./$(1) -no-vis ]\n"
+1409
View File
File diff suppressed because it is too large Load Diff
+2
View File
@@ -768,11 +768,13 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/examples \
@MFEM_SOURCE_DIR@/examples/petsc \
@MFEM_SOURCE_DIR@/examples/pumi \
@MFEM_SOURCE_DIR@/examples/hiop \
@MFEM_SOURCE_DIR@/examples/sundials \
@MFEM_SOURCE_DIR@/miniapps/common \
@MFEM_SOURCE_DIR@/miniapps/meshing \
@MFEM_SOURCE_DIR@/miniapps/tools \
@MFEM_SOURCE_DIR@/miniapps/nurbs \
@MFEM_SOURCE_DIR@/miniapps/gslib \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/performance
+12 -1
View File
@@ -81,6 +81,8 @@ namespace mfem {
* - <a class="el" href="ex20p_8cpp_source.html">Example 20p</a>: parallel symplectic ODE integration
* - <a class="el" href="ex21_8cpp_source.html">Example 21</a>: adaptive mesh refinement for linear elasticity
* - <a class="el" href="ex21p_8cpp_source.html">Example 21p</a>: parallel adaptive mesh refinement for linear elasticity
* - <a class="el" href="ex22_8cpp_source.html">Example 22</a>: complex-valued linear systems for damped harmonic oscillators
* - <a class="el" href="ex22p_8cpp_source.html">Example 22p</a>: parallel complex-valued linear systems for damped harmonic oscillators
*
* <H4>SUNDIALS Examples</H4>
* - Variants of Examples
@@ -115,6 +117,12 @@ namespace mfem {
* <a class="el" href="pumi_2ex6p_8cpp_source.html">6p</a>
* demonstrating the use of MFEM's \link pumi.hpp PUMI classes\endlink
*
* <H4>HiOp Examples</H4>
* - Variants of Examples
* <a class="el" href="hiop_2ex9_8cpp_source.html">9</a> and
* <a class="el" href="hiop_2ex9p_8cpp_source.html">9p</a>,
* demonstrating the use of MFEM's \link hiop.hpp HiOp classes\endlink
*
* <H3>Miniapps</H3>
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
@@ -128,9 +136,12 @@ namespace mfem {
* - <a class="el" href="mesh-explorer_8cpp_source.html">Mesh Explorer</a>: visualize and manipulate meshes
* - <a class="el" href="mesh-optimizer_8cpp_source.html">Mesh Optimizer</a>: optimize high-order meshes, <a class="el" href="mesh-optimizer_8cpp_source.html">serial</a> and <a class="el" href="pmesh-optimizer_8cpp_source.html">parallel</a> versions
* - <a class="el" href="display-basis_8cpp_source.html">Display Basis</a>: visualize finite element basis functions
* - <a class="el" href="get-values_8cpp_source.html">Get Values</a>: extract field values via DataCollection classes
* - <a class="el" href="load-dc_8cpp_source.html">Load DC</a>: visualize fields saved via DataCollection classes
* - <a class="el" href="convert-dc_8cpp_source.html">Convert DC</a>: convert between diffirent DataCollection formats
* - <a class="el" href="convert-dc_8cpp_source.html">Convert DC</a>: convert between different DataCollection formats
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
*
+15
View File
@@ -28,6 +28,7 @@ list(APPEND ALL_EXE_SRCS
ex19.cpp
ex20.cpp
ex21.cpp
ex22.cpp
)
if (MFEM_USE_MPI)
@@ -53,6 +54,7 @@ if (MFEM_USE_MPI)
ex19p.cpp
ex20p.cpp
ex21p.cpp
ex22p.cpp
)
endif()
@@ -100,6 +102,11 @@ if (MFEM_USE_SUNDIALS)
add_subdirectory(sundials)
endif()
# Include the examples/sundials directory if SUNDIALS is enabled.
if (MFEM_USE_GINKGO)
add_subdirectory(ginkgo)
endif()
# Include the examples/petsc directory if PETSc is enabled.
if (MFEM_USE_PETSC)
add_subdirectory(petsc)
@@ -109,3 +116,11 @@ endif()
if (MFEM_USE_PUMI)
add_subdirectory(pumi)
endif()
if (MFEM_USE_HIOP)
add_subdirectory(hiop)
endif()
if (MFEM_USE_ADEPT)
add_subdirectory(ad)
endif()
+5 -2
View File
@@ -31,6 +31,8 @@
// ex1 -pa -d occa-cuda
// ex1 -pa -d raja-omp
// ex1 -pa -d occa-omp
// ex1 -pa -d ceed-cpu
// ex1 -pa -d ceed-cuda
// ex1 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define a
@@ -194,9 +196,10 @@ int main(int argc, char *argv[])
umf_solver.Mult(B, X);
#endif
}
else // No preconditioning for now in partial assembly mode.
else // Jacobi preconditioning in partial assembly mode
{
CG(*A, B, X, 1, 2000, 1e-12, 0.0);
OperatorJacobiSmoother M(*a, ess_tdof_list);
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
}
// 12. Recover the solution as a finite element grid function.
+2 -2
View File
@@ -8,7 +8,7 @@
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 462 -n 10 -o 2 -elast
// mpirun -np 4 ex12p -m ../data/beam-hex.mesh -s 3878
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 81
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3876 -o 2 -sys
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3877 -o 2 -sys
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh -s 4544 -n 6 -o 3 -elast
// mpirun -np 4 ex12p -m ../data/beam-quad-nurbs.mesh
// mpirun -np 4 ex12p -m ../data/beam-hex-nurbs.mesh
@@ -57,7 +57,7 @@ int main(int argc, char *argv[])
const char *mesh_file = "../data/beam-tri.mesh";
int order = 1;
int nev = 5;
int seed = 75;
int seed = 66;
bool visualization = 1;
bool amg_elast = 0;
+1 -1
View File
@@ -3,7 +3,7 @@
// Compile with: make ex13p
//
// Sample runs: mpirun -np 4 ex13p -m ../data/star.mesh
// mpirun -np 4 ex13p -m ../data/square-disc.mesh -o 2
// mpirun -np 4 ex13p -m ../data/square-disc.mesh -o 2 -n 4
// mpirun -np 4 ex13p -m ../data/beam-tet.mesh
// mpirun -np 4 ex13p -m ../data/beam-hex.mesh
// mpirun -np 4 ex13p -m ../data/escher.mesh
+1 -4
View File
@@ -16,9 +16,6 @@
// ex15 -m ../data/ball-nurbs.mesh -tf 0.3
// ex15 -m ../data/mobius-strip.mesh
// ex15 -m ../data/amr-quad.mesh
//
// Conforming meshes (no derefinement):
//
// ex15 -m ../data/square-disc.mesh
// ex15 -m ../data/escher.mesh -r 2 -tf 0.3
//
@@ -130,7 +127,7 @@ int main(int argc, char *argv[])
if (ref_levels > 0) { ref_levels--; }
mesh.SetCurvature(2);
}
mesh.EnsureNCMesh();
mesh.EnsureNCMesh(true);
for (int l = 0; l < ref_levels; l++)
{
mesh.UniformRefinement();
+1 -4
View File
@@ -16,9 +16,6 @@
// mpirun -np 4 ex15p -m ../data/ball-nurbs.mesh -tf 0.5
// mpirun -np 4 ex15p -m ../data/mobius-strip.mesh
// mpirun -np 4 ex15p -m ../data/amr-quad.mesh
//
// Conforming meshes (no load balancing and derefinement):
//
// mpirun -np 4 ex15p -m ../data/square-disc.mesh
// mpirun -np 4 ex15p -m ../data/escher.mesh -r 2 -tf 0.3
//
@@ -146,7 +143,7 @@ int main(int argc, char *argv[])
if (ref_levels > 0) { ref_levels--; }
mesh->SetCurvature(2);
}
mesh->EnsureNCMesh();
mesh->EnsureNCMesh(true);
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
+11 -2
View File
@@ -29,6 +29,8 @@
// mpirun -np 4 ex1p -pa -d cuda
// mpirun -np 4 ex1p -pa -d occa-cuda
// mpirun -np 4 ex1p -pa -d raja-omp
// mpirun -np 4 ex1p -pa -d ceed-cpu
// mpirun -np 4 ex1p -pa -d ceed-cuda
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
@@ -208,9 +210,16 @@ int main(int argc, char *argv[])
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use no preconditioner, for now.
// * With partial assembly, use Jacobi smoothing, for now.
Solver *prec = NULL;
if (!pa) { prec = new HypreBoomerAMG; }
if (pa)
{
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
}
else
{
prec = new HypreBoomerAMG;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
+561
View File
@@ -0,0 +1,561 @@
// MFEM Example 22
//
// Compile with: make ex22
//
// Sample runs: ex22 -m ../data/inline-segment.mesh -o 3
// ex22 -m ../data/inline-tri.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1
// ex22 -m ../data/inline-quad.mesh -o 3 -p 2
// ex22 -m ../data/inline-tet.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2 -p 1
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
//
// 1) A scalar H1 field
// -Div(a Grad u) - omega^2 b u + i omega c u = 0
//
// 2) A vector H(Curl) field
// Curl(a Curl u) - omega^2 b u + i omega c u = 0
//
// 3) A vector H(Div) field
// -Grad(a Div u) - omega^2 b u + i omega c u = 0
//
// In each case the field is driven by a forced oscillation, with
// angular frequency omega, imposed at the boundary or a portion
// of the boundary.
//
// In electromagnetics, the coefficients are typically named the
// permeability, mu = 1/a, permittivity, epsilon = b, and
// conductivity, sigma = c. The user can specify these constants
// using either set of names.
//
// The example also demonstrates how to display a time-varying
// solution as a sequence of fields sent to a single GLVis socket.
//
// We recommend viewing examples 1, 3 and 4 before viewing this
// example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
static double mu_ = 1.0;
static double epsilon_ = 1.0;
static double sigma_ = 20.0;
static double omega_ = 10.0;
double u0_real_exact(const Vector &);
double u0_imag_exact(const Vector &);
void u1_real_exact(const Vector &, Vector &);
void u1_imag_exact(const Vector &, Vector &);
void u2_real_exact(const Vector &, Vector &);
void u2_imag_exact(const Vector &, Vector &);
bool check_for_inline_mesh(const char * mesh_file);
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
const char *mesh_file = "../data/inline-quad.mesh";
int ref_levels = 0;
int order = 1;
int prob = 0;
double freq = -1.0;
double a_coef = 0.0;
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&ref_levels, "-r", "--refine",
"Number of times to refine the mesh uniformly.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H_1, 1: H(Curl), or 2: H(Div) "
"damped harmonic oscillator.");
args.AddOption(&a_coef, "-a", "--stiffness-coef",
"Stiffness coefficient (spring constant or 1/mu).");
args.AddOption(&epsilon_, "-b", "--mass-coef",
"Mass coefficient (or epsilon).");
args.AddOption(&sigma_, "-c", "--damping-coef",
"Damping coefficient (or sigma).");
args.AddOption(&mu_, "-mu", "--permeability",
"Permeability of free space (or 1/(spring constant)).");
args.AddOption(&epsilon_, "-eps", "--permittivity",
"Permittivity of free space (or mass constant).");
args.AddOption(&sigma_, "-sigma", "--conductivity",
"Conductivity (or damping constant).");
args.AddOption(&freq, "-f", "--frequency",
"Frequency (in Hz).");
args.AddOption(&herm_conv, "-herm", "--hermitian", "-no-herm",
"--no-hermitian", "Use convention for Hermitian operators.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
MFEM_VERIFY(prob >= 0 && prob <=2,
"Unrecognized problem type: " << prob);
if ( a_coef != 0.0 )
{
mu_ = 1.0 / a_coef;
}
if ( freq > 0.0 )
{
omega_ = 2.0 * M_PI * freq;
}
exact_sol = check_for_inline_mesh(mesh_file);
if (exact_sol)
{
cout << "Identified a mesh with known exact solution" << endl;
}
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 2. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes
// with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Refine the mesh to increase resolution. In this example we do
// 'ref_levels' of uniform refinement where the user specifies
// the number of levels with the '-r' option.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 4. Define a finite element space on the mesh. Here we use continuous
// Lagrange, Nedelec, or Raviart-Thomas finite elements of the specified
// order.
if (dim == 1 && prob != 0 )
{
cout << "Switching to problem type 0, H1 basis functions, "
<< "for 1 dimensional mesh." << endl;
prob = 0;
}
FiniteElementCollection *fec = NULL;
switch (prob)
{
case 0: fec = new H1_FECollection(order, dim); break;
case 1: fec = new ND_FECollection(order, dim); break;
case 2: fec = new RT_FECollection(order - 1, dim); break;
default: break; // This should be unreachable
}
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
cout << "Number of finite element unknowns: " << fespace->GetTrueVSize()
<< endl;
// 5. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined based on the type
// of mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (mesh->bdr_attributes.Size())
{
ess_bdr.SetSize(mesh->bdr_attributes.Max());
ess_bdr = 1;
if (exact_sol)
{
switch (prob)
{
case 0: ess_bdr = 0; ess_bdr[0] = 1; break;
default: ess_bdr = 1; ess_bdr[2] = 0; break;
}
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 6. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
ComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 7. Define the solution vector u as a complex finite element grid function
// corresponding to fespace. Initialize u with initial guess of 1+0i or
// the exact solution if it is known.
ComplexGridFunction u(fespace);
ComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ComplexGridFunction(fespace); }
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
Vector zeroVec(dim); zeroVec = 0.0;
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u0_r, u0_i); }
break;
case 1:
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u1_r, u1_i); }
break;
case 2:
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u2_r, u2_i); }
break;
default: break; // This should be unreachable
}
if (visualization && exact_sol)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u_exact->real()
<< "window_title 'Exact: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u_exact->imag()
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 8. Set up the sesquilinear form a(.,.) on the finite element space
// corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + i omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) - omega^2 b + i omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + i omega c
//
ConstantCoefficient stiffnessCoef(1.0/mu_);
ConstantCoefficient massCoef(-omega_ * omega_ * epsilon_);
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 8a. Set up the bilinear form for the preconditioner corresponding to the
// appropriate operator
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) + omega^2 b + omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + omega c
//
BilinearForm *pcOp = new BilinearForm(fespace);
switch (prob)
{
case 0:
pcOp->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(lossCoef));
break;
case 1:
pcOp->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(negMassCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
case 2:
pcOp->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 9. Assemble the form and the corresponding linear system, applying any
// necessary transformations such as: assembly, eliminating boundary
// conditions, conforming constraints for non-conforming AMR, etc.
a->Assemble();
pcOp->Assemble();
OperatorHandle A;
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
{
ComplexSparseMatrix * Asp =
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Asp->real().Width() << endl << endl;
}
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the appropriate sparse smoother.
{
Array<int> blockOffsets;
blockOffsets.SetSize(3);
blockOffsets[0] = 0;
blockOffsets[1] = PCOp.Ptr()->Height();
blockOffsets[2] = PCOp.Ptr()->Height();
blockOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockOffsets);
Operator * pc_r = NULL;
Operator * pc_i = NULL;
switch (prob)
{
case 0: // fallthrough to case 2
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
break;
default: break; // This should be unreachable
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
1.0:-1.0);
BDP.SetDiagonalBlock(0, pc_r);
BDP.SetDiagonalBlock(1, pc_i);
BDP.owns_blocks = 1;
GMRESSolver gmres;
gmres.SetPreconditioner(BDP);
gmres.SetOperator(*A.Ptr());
gmres.SetRelTol(1e-12);
gmres.SetMaxIter(1000);
gmres.SetPrintLevel(1);
gmres.Mult(B, U);
}
// 11. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a->RecoverFEMSolution(U, b, u);
if (exact_sol)
{
double err_r = -1.0;
double err_i = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
break;
default: break; // This should be unreachable
}
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << endl;
}
// 12. Save the refined mesh and the solution. This output can be viewed
// later using GLVis: "glvis -m mesh -g sol".
{
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
ofstream sol_r_ofs("sol_r.gf");
ofstream sol_i_ofs("sol_i.gf");
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
}
// 13. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u.real()
<< "window_title 'Solution: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u.imag()
<< "window_title 'Solution: Imaginary Part'" << flush;
}
if (visualization && exact_sol)
{
*u_exact -= u;
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *mesh << u_exact->real()
<< "window_title 'Error: Real Part'" << flush;
sol_sock_i << "solution\n" << *mesh << u_exact->imag()
<< "window_title 'Error: Imaginary Part'" << flush;
}
if (visualization)
{
GridFunction u_t(fespace);
u_t = u.real();
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << u_t
<< "window_title 'Harmonic Solution (t = 0.0 T)'"
<< "pause\n" << flush;
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
int num_frames = 32;
int i = 0;
while (sol_sock)
{
double t = (double)(i % num_frames) / num_frames;
ostringstream oss;
oss << "Harmonic Solution (t = " << t << " T)";
add(cos( 2.0 * M_PI * t), u.real(),
sin(-2.0 * M_PI * t), u.imag(), u_t);
sol_sock << "solution\n" << *mesh << u_t
<< "window_title '" << oss.str() << "'" << flush;
i++;
}
}
// 14. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
delete fespace;
delete fec;
delete mesh;
return 0;
}
bool check_for_inline_mesh(const char * mesh_file)
{
string file(mesh_file);
size_t p0 = file.find_last_of("/");
string s0 = file.substr((p0==string::npos)?0:(p0+1),7);
return s0 == "inline-";
}
complex<double> u0_exact(const Vector &x)
{
int dim = x.Size();
complex<double> i(0.0, 1.0);
complex<double> alpha = (epsilon_ * omega_ - i * sigma_);
complex<double> kappa = std::sqrt(mu_ * omega_* alpha);
return std::exp(-i * kappa * x[dim - 1]);
}
double u0_real_exact(const Vector &x)
{
return u0_exact(x).real();
}
double u0_imag_exact(const Vector &x)
{
return u0_exact(x).imag();
}
void u1_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_real_exact(x);
}
void u1_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_real_exact(x);
}
void u2_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_imag_exact(x);
}
+622
View File
@@ -0,0 +1,622 @@
// MFEM Example 22 - Parallel Version
//
// Compile with: make ex22p
//
// Sample runs: mpirun -np 4 ex22p -m ../data/inline-segment.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-tri.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 1
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 2
// mpirun -np 4 ex22p -m ../data/inline-tet.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 1
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
//
// 1) A scalar H1 field
// -Div(a Grad u) - omega^2 b u + i omega c u = 0
//
// 2) A vector H(Curl) field
// Curl(a Curl u) - omega^2 b u + i omega c u = 0
//
// 3) A vector H(Div) field
// -Grad(a Div u) - omega^2 b u + i omega c u = 0
//
// In each case the field is driven by a forced oscillation, with
// angular frequency omega, imposed at the boundary or a portion
// of the boundary.
//
// In electromagnetics the coefficients are typically named the
// permeability, mu = 1/a, permittivity, epsilon = b, and
// conductivity, sigma = c. The user can specify these constants
// using either set of names.
//
// The example also demonstrates how to display a time-varying
// solution as a sequence of fields sent to a single GLVis socket.
//
// We recommend viewing examples 1, 3 and 4 before viewing this
// example.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
static double mu_ = 1.0;
static double epsilon_ = 1.0;
static double sigma_ = 20.0;
static double omega_ = 10.0;
double u0_real_exact(const Vector &);
double u0_imag_exact(const Vector &);
void u1_real_exact(const Vector &, Vector &);
void u1_imag_exact(const Vector &, Vector &);
void u2_real_exact(const Vector &, Vector &);
void u2_imag_exact(const Vector &, Vector &);
bool check_for_inline_mesh(const char * mesh_file);
int main(int argc, char *argv[])
{
// 1. Initialize MPI.
int num_procs, myid;
MPI_Init(&argc, &argv);
MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
MPI_Comm_rank(MPI_COMM_WORLD, &myid);
// 2. Parse command-line options.
const char *mesh_file = "../data/inline-quad.mesh";
int ser_ref_levels = 1;
int par_ref_levels = 1;
int order = 1;
int prob = 0;
double freq = -1.0;
double a_coef = 0.0;
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
"Number of times to refine the mesh uniformly in serial.");
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
"Number of times to refine the mesh uniformly in parallel.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree).");
args.AddOption(&prob, "-p", "--problem-type",
"Choose between 0: H_1, 1: H(Curl), or 2: H(Div) "
"damped harmonic oscillator.");
args.AddOption(&a_coef, "-a", "--stiffness-coef",
"Stiffness coefficient (spring constant or 1/mu).");
args.AddOption(&epsilon_, "-b", "--mass-coef",
"Mass coefficient (or epsilon).");
args.AddOption(&sigma_, "-c", "--damping-coef",
"Damping coefficient (or sigma).");
args.AddOption(&mu_, "-mu", "--permeability",
"Permeability of free space (or 1/(spring constant)).");
args.AddOption(&epsilon_, "-eps", "--permittivity",
"Permittivity of free space (or mass constant).");
args.AddOption(&sigma_, "-sigma", "--conductivity",
"Conductivity (or damping constant).");
args.AddOption(&freq, "-f", "--frequency",
"Frequency (in Hz).");
args.AddOption(&herm_conv, "-herm", "--hermitian", "-no-herm",
"--no-hermitian", "Use convention for Hermitian operators.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
if (myid == 0)
{
args.PrintUsage(cout);
}
MPI_Finalize();
return 1;
}
if (myid == 0)
{
args.PrintOptions(cout);
}
MFEM_VERIFY(prob >= 0 && prob <=2,
"Unrecognized problem type: " << prob);
if ( a_coef != 0.0 )
{
mu_ = 1.0 / a_coef;
}
if ( freq > 0.0 )
{
omega_ = 2.0 * M_PI * freq;
}
exact_sol = check_for_inline_mesh(mesh_file);
if (myid == 0 && exact_sol)
{
cout << "Identified a mesh with known exact solution" << endl;
}
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 3. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the serial mesh on all processors to increase the resolution.
for (int l = 0; l < ser_ref_levels; l++)
{
mesh->UniformRefinement();
}
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
for (int l = 0; l < par_ref_levels; l++)
{
pmesh->UniformRefinement();
}
// 6. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange, Nedelec, or Raviart-Thomas finite elements of
// the specified order.
if (dim == 1 && prob != 0 )
{
if (myid == 0)
{
cout << "Switching to problem type 0, H1 basis functions, "
<< "for 1 dimensional mesh." << endl;
}
prob = 0;
}
FiniteElementCollection *fec = NULL;
switch (prob)
{
case 0: fec = new H1_FECollection(order, dim); break;
case 1: fec = new ND_FECollection(order, dim); break;
case 2: fec = new RT_FECollection(order - 1, dim); break;
default: break; // This should be unreachable
}
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
HYPRE_Int size = fespace->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl;
}
// 7. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the type of mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (pmesh->bdr_attributes.Size())
{
ess_bdr.SetSize(pmesh->bdr_attributes.Max());
ess_bdr = 1;
if (exact_sol)
{
switch (prob)
{
case 0: ess_bdr = 0; ess_bdr[0] = 1; break;
default: ess_bdr = 1; ess_bdr[2] = 0; break;
}
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 8. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system.
ParComplexLinearForm b(fespace, conv);
b.Vector::operator=(0.0);
// 9. Define the solution vector u as a parallel complex finite element grid
// function corresponding to fespace. Initialize u with initial guess of
// 1+0i or the exact solution if it is known.
ParComplexGridFunction u(fespace);
ParComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ParComplexGridFunction(fespace); }
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
Vector zeroVec(dim); zeroVec = 0.0;
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
u = 0.0;
switch (prob)
{
case 0:
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u0_r, u0_i); }
break;
case 1:
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u1_r, u1_i); }
break;
case 2:
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
if (exact_sol) { u_exact->ProjectCoefficient(u2_r, u2_i); }
break;
default: break; // This should be unreachable
}
if (visualization && exact_sol)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Exact: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Exact: Imaginary Part'" << flush;
}
// 10. Set up the parallel sesquilinear form a(.,.) on the finite element
// space corresponding to the damped harmonic oscillator operator of the
// appropriate type:
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + i omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) - omega^2 b + i omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + i omega c
//
ConstantCoefficient stiffnessCoef(1.0/mu_);
ConstantCoefficient massCoef(-omega_ * omega_ * epsilon_);
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 10a. Set up the parallel bilinear form for the preconditioner
// corresponding to the appropriate operator
//
// 0) A scalar H1 field
// -Div(a Grad) - omega^2 b + omega c
//
// 1) A vector H(Curl) field
// Curl(a Curl) + omega^2 b + omega c
//
// 2) A vector H(Div) field
// -Grad(a Div) - omega^2 b + omega c
//
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
switch (prob)
{
case 0:
pcOp->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new MassIntegrator(lossCoef));
break;
case 1:
pcOp->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(negMassCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
case 2:
pcOp->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef));
pcOp->AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
break;
default: break; // This should be unreachable
}
// 11. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, etc.
a->Assemble();
pcOp->Assemble();
OperatorHandle A;
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
if (myid == 0)
{
ComplexHypreParMatrix * Ahyp =
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
}
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
// diagonal preconditioner based on the appropriate multigrid
// preconditioner from hypre.
{
Array<HYPRE_Int> blockTrueOffsets;
blockTrueOffsets.SetSize(3);
blockTrueOffsets[0] = 0;
blockTrueOffsets[1] = PCOp.Ptr()->Height();
blockTrueOffsets[2] = PCOp.Ptr()->Height();
blockTrueOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
Operator * pc_r = NULL;
Operator * pc_i = NULL;
switch (prob)
{
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
break;
case 2:
if (dim == 2 )
{
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
1.0:-1.0);
BDP.SetDiagonalBlock(0, pc_r);
BDP.SetDiagonalBlock(1, pc_i);
BDP.owns_blocks = 1;
FGMRESSolver fgmres(MPI_COMM_WORLD);
fgmres.SetPreconditioner(BDP);
fgmres.SetOperator(*A.Ptr());
fgmres.SetRelTol(1e-12);
fgmres.SetMaxIter(1000);
fgmres.SetPrintLevel(1);
fgmres.Mult(B, U);
}
// 13. Recover the parallel grid function corresponding to U. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(U, b, u);
if (exact_sol)
{
double err_r = -1.0;
double err_i = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
break;
default: break; // This should be unreachable
}
if ( myid == 0 )
{
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << endl;
}
}
// 14. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_r_name, sol_i_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh->Print(mesh_ofs);
ofstream sol_r_ofs(sol_r_name.str().c_str());
ofstream sol_i_ofs(sol_i_name.str().c_str());
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
}
// 15. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u.real()
<< "window_title 'Solution: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u.imag()
<< "window_title 'Solution: Imaginary Part'" << flush;
}
if (visualization && exact_sol)
{
*u_exact -= u;
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock_r(vishost, visport);
socketstream sol_sock_i(vishost, visport);
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
sol_sock_r.precision(8);
sol_sock_i.precision(8);
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
<< "window_title 'Error: Real Part'" << flush;
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
<< "window_title 'Error: Imaginary Part'" << flush;
}
if (visualization)
{
ParGridFunction u_t(fespace);
u_t = u.real();
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << *pmesh << u_t
<< "window_title 'Harmonic Solution (t = 0.0 T)'"
<< "pause\n" << flush;
if (myid == 0)
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
int num_frames = 32;
int i = 0;
while (sol_sock)
{
double t = (double)(i % num_frames) / num_frames;
ostringstream oss;
oss << "Harmonic Solution (t = " << t << " T)";
add(cos( 2.0 * M_PI * t), u.real(),
sin(-2.0 * M_PI * t), u.imag(), u_t);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock << "solution\n" << *pmesh << u_t
<< "window_title '" << oss.str() << "'" << flush;
i++;
}
}
// 16. Free the used memory.
delete a;
delete u_exact;
delete pcOp;
delete fespace;
delete fec;
delete pmesh;
MPI_Finalize();
return 0;
}
bool check_for_inline_mesh(const char * mesh_file)
{
string file(mesh_file);
size_t p0 = file.find_last_of("/");
string s0 = file.substr((p0==string::npos)?0:(p0+1),7);
return s0 == "inline-";
}
complex<double> u0_exact(const Vector &x)
{
int dim = x.Size();
complex<double> i(0.0, 1.0);
complex<double> alpha = (epsilon_ * omega_ - i * sigma_);
complex<double> kappa = std::sqrt(mu_ * omega_* alpha);
return std::exp(-i * kappa * x[dim - 1]);
}
double u0_real_exact(const Vector &x)
{
return u0_exact(x).real();
}
double u0_imag_exact(const Vector &x)
{
return u0_exact(x).imag();
}
void u1_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_real_exact(x);
}
void u1_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_real_exact(x);
}
void u2_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_imag_exact(x);
}
+12 -3
View File
@@ -21,7 +21,7 @@
//
// The example demonstrates the use of the BlockMatrix class, as
// well as the collective saving of several grid functions in a
// VisIt (visit.llnl.gov) visualization format.
// VisIt (visit.llnl.gov) and ParaView (paraview.org) formats.
//
// We recommend viewing examples 1-4 before viewing this example.
@@ -264,7 +264,16 @@ int main(int argc, char *argv[])
visit_dc.RegisterField("pressure", &p);
visit_dc.Save();
// 14. Send the solution by socket to a GLVis server.
// 14. Save data in the ParaView format
ParaViewDataCollection paraview_dc("PVExample5S", mesh);
paraview_dc.SetLevelsOfDetail(2);
paraview_dc.SetCycle(1);
paraview_dc.SetTime(0.0); // set the time
paraview_dc.RegisterField("velocity",&u);
paraview_dc.RegisterField("pressure",&p);
paraview_dc.Save();
// 15. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -277,7 +286,7 @@ int main(int argc, char *argv[])
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
}
// 15. Free the used memory.
// 16. Free the used memory.
delete fform;
delete gform;
delete invM;
+12 -4
View File
@@ -21,7 +21,7 @@
//
// The example demonstrates the use of the BlockMatrix class, as
// well as the collective saving of several grid functions in a
// VisIt (visit.llnl.gov) visualization format.
// VisIt (visit.llnl.gov) and ParaView (paraview.org) formats.
//
// We recommend viewing examples 1-4 before viewing this example.
@@ -239,7 +239,6 @@ int main(int argc, char *argv[])
// 12. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(500);
double rtol(1.e-6);
double atol(1.e-10);
@@ -326,7 +325,16 @@ int main(int argc, char *argv[])
DataCollection::PARALLEL_FORMAT);
visit_dc.Save();
// 16. Send the solution by socket to a GLVis server.
// 16. Save data in the ParaView format
ParaViewDataCollection paraview_dc("PVExample5P", pmesh);
paraview_dc.SetLevelsOfDetail(1);
paraview_dc.SetCycle(1);
paraview_dc.SetTime(0.0);
paraview_dc.RegisterField("velocity",u);
paraview_dc.RegisterField("pressure",p);
paraview_dc.Save();
// 17. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -346,7 +354,7 @@ int main(int argc, char *argv[])
<< endl;
}
// 17. Free the used memory.
// 18. Free the used memory.
delete fform;
delete gform;
delete u;
+2
View File
@@ -19,6 +19,8 @@
// ex6 -pa -d cuda
// ex6 -pa -d occa-cuda
// ex6 -pa -d raja-omp
// ex6 -pa -d ceed-cpu
// ex6 -pa -d ceed-cuda
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
+2
View File
@@ -19,6 +19,8 @@
// mpirun -np 4 ex6p -pa -d cuda
// mpirun -np 4 ex6p -pa -d occa-cuda
// mpirun -np 4 ex6p -pa -d raja-omp
// mpirun -np 4 ex6p -pa -d ceed-cpu
// mpirun -np 4 ex6p -pa -d ceed-cuda
//
// Description: This is a version of Example 1 with a simple adaptive mesh
// refinement loop. The problem being solved is again the Laplace
+24 -1
View File
@@ -26,7 +26,8 @@
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
// with VisIt (visit.llnl.gov) and ParaView (paraview.org) is also
// illustrated.
#include "mfem.hpp"
#include <fstream>
@@ -89,6 +90,7 @@ int main(int argc, char *argv[])
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool paraview = false;
bool binary = false;
int vis_steps = 5;
@@ -117,6 +119,9 @@ int main(int argc, char *argv[])
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
@@ -237,6 +242,16 @@ int main(int argc, char *argv[])
dc->Save();
}
ParaViewDataCollection *pd = NULL;
if (paraview)
{
pd = new ParaViewDataCollection("PVExample9S", &mesh);
pd->RegisterField("solution", &u);
pd->SetLevelsOfDetail(2);
pd->SetCycle(0);
pd->SetTime(0.0);
}
socketstream sout;
if (visualization)
{
@@ -294,6 +309,13 @@ int main(int argc, char *argv[])
dc->SetTime(t);
dc->Save();
}
if (paraview)
{
pd->SetCycle(ti);
pd->SetTime(t);
pd->Save();
}
}
}
@@ -307,6 +329,7 @@ int main(int argc, char *argv[])
// 10. Free the used memory.
delete ode_solver;
delete pd;
delete dc;
return 0;
+25 -1
View File
@@ -26,7 +26,8 @@
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
// with VisIt (visit.llnl.gov) and ParaView (paraview.org) is also
// illustrated.
#include "mfem.hpp"
#include <fstream>
@@ -95,6 +96,7 @@ int main(int argc, char *argv[])
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool paraview = false;
bool binary = false;
int vis_steps = 5;
@@ -125,6 +127,9 @@ int main(int argc, char *argv[])
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&paraview, "-paraview", "--paraview-datafiles", "-no-paraview",
"--no-paraview-datafiles",
"Save data files for ParaView (paraview.org) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
@@ -281,6 +286,17 @@ int main(int argc, char *argv[])
dc->Save();
}
ParaViewDataCollection *pd = NULL;
if (paraview)
{
pd = new ParaViewDataCollection("PVExample9P", pmesh);
pd->RegisterField("solution", u);
pd->SetLevelsOfDetail(2);
pd->SetCycle(0);
pd->SetTime(0.0);
pd->Save();
}
socketstream sout;
if (visualization)
{
@@ -352,6 +368,13 @@ int main(int argc, char *argv[])
dc->SetTime(t);
dc->Save();
}
if (paraview)
{
pd->SetCycle(ti);
pd->SetTime(t);
pd->Save();
}
}
}
@@ -378,6 +401,7 @@ int main(int argc, char *argv[])
delete fes;
delete pmesh;
delete ode_solver;
delete pd;
delete dc;
MPI_Finalize();
+59
View File
@@ -0,0 +1,59 @@
# Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
set(GINKGO_EXAMPLES_SRCS)
list(APPEND GINKGO_EXAMPLES_SRCS
ex1.cpp
)
# Include the source directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Add "test_ginkgo" target, see below.
add_custom_target(test_ginkgo
${CMAKE_CTEST_COMMAND} -R ginkgo USES_TERMINAL)
# Add one executable per cpp file, adding "ginkgo_" as prefix. Sets
# "test_ginkgo" as a target that depends on the given examples.
set(PFX ginkgo_)
add_mfem_examples(GINKGO_EXAMPLES_SRCS ${PFX} "" test_ginkgo)
# Testing.
# The GINKGO tests can be run separately using the target "test_ginkgo"
# which builds the examples and runs:
# ctest -R ginkgo
# Command line options for the tests.
set(EX1_COMMON_OPTS ex1 -m ../data/star.mesh --use_gko_solver)
set(EX1_TEST_OPTS ${EX9_COMMON_OPTS})
# Add the tests: one test per source file.
foreach(SRC_FILE ${GINKGO_EXAMPLES_SRCS})
get_filename_component(SRC_FILENAME ${SRC_FILE} NAME)
string(REPLACE ".cpp" "" TEST_NAME ${SRC_FILENAME})
string(TOUPPER ${TEST_NAME} UP_TEST_NAME)
set(TEST_NAME ${PFX}${TEST_NAME})
set(THIS_TEST_OPTIONS "-no-vis")
list(APPEND THIS_TEST_OPTIONS ${${UP_TEST_NAME}_TEST_OPTS})
# message(STATUS "Test ${TEST_NAME} options: ${THIS_TEST_OPTIONS}")
if (NOT (${TEST_NAME} MATCHES ".*p$"))
add_test(NAME ${TEST_NAME}_ser
COMMAND ${TEST_NAME} ${THIS_TEST_OPTIONS})
else()
add_test(NAME ${TEST_NAME}_np=4
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:${TEST_NAME}> ${THIS_TEST_OPTIONS}
${MPIEXEC_POSTFLAGS})
endif()
endforeach()
+259
View File
@@ -0,0 +1,259 @@
// MFEM Example 1
// GINKGO Modification
//
// Compile with: make ex1
//
// Sample runs: ex1 -m ../data/square-disc.mesh
// ex1 -m ../data/star.mesh
// ex1 -m ../data/star-mixed.mesh
// ex1 -m ../data/escher.mesh
// ex1 -m ../data/fichera.mesh
// ex1 -m ../data/fichera-mixed.mesh
// ex1 -m ../data/toroid-wedge.mesh
// ex1 -m ../data/square-disc-p2.vtk -o 2
// ex1 -m ../data/square-disc-p3.mesh -o 3
// ex1 -m ../data/square-disc-nurbs.mesh -o -1
// ex1 -m ../data/star-mixed-p2.mesh -o 2
// ex1 -m ../data/disc-nurbs.mesh -o -1
// ex1 -m ../data/pipe-nurbs.mesh -o -1
// ex1 -m ../data/fichera-mixed-p2.mesh -o 2
// ex1 -m ../data/star-surf.mesh
// ex1 -m ../data/square-disc-surf.mesh
// ex1 -m ../data/inline-segment.mesh
// ex1 -m ../data/amr-quad.mesh
// ex1 -m ../data/amr-hex.mesh
// ex1 -m ../data/fichera-amr.mesh
// ex1 -m ../data/mobius-strip.mesh
// ex1 -m ../data/mobius-strip.mesh -o -1 -sc
//
// Device sample runs:
// ex1 -pa -d cuda
// ex1 -pa -d raja-cuda
// ex1 -pa -d occa-cuda
// ex1 -pa -d raja-omp
// ex1 -pa -d occa-omp
// ex1 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Laplace problem
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
// Specifically, we discretize using a FE space of the specified
// order, or if order < 1 using an isoparametric/isogeometric
// space (i.e. quadratic for quadratic curvilinear mesh, NURBS for
// NURBS mesh, etc.)
//
// The example highlights the use of mesh refinement, finite
// element grid functions, as well as linear and bilinear forms
// corresponding to the left-hand side and right-hand side of the
// discrete linear system. We also cover the explicit elimination
// of essential boundary conditions, static condensation, and the
// optional connection to the GLVis tool for visualization.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
#ifndef MFEM_USE_GINKGO
#error This example requires that MFEM is built with MFEM_USE_GINKGO=YES
#endif
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
const char *mesh_file = "../../data/star.mesh";
int order = 1;
bool static_cond = false;
bool pa = false;
const char *device_config = "cpu";
bool visualization = true;
bool use_ginkgo_solver= true;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree) or -1 for"
" isoparametric space.");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&use_ginkgo_solver, "-gko", "--use_gko_solver", "-no-gko",
"--no-gko-solver",
"Solve using ginkgo.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
// largest number that gives a final mesh with no more than 50,000
// elements.
{
int ref_levels =
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
}
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
}
else if (mesh->GetNodes())
{
fec = mesh->GetNodes()->OwnFEC();
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
}
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
cout << "Number of finite element unknowns: "
<< fespace->GetTrueVSize() << endl;
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined by marking all
// the boundary attributes from the mesh as essential (Dirichlet) and
// converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (mesh->bdr_attributes.Size())
{
Array<int> ess_bdr(mesh->bdr_attributes.Max());
ess_bdr = 1;
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
// the basis functions in the finite element fespace.
LinearForm *b = new LinearForm(fespace);
ConstantCoefficient one(1.0);
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
// 8. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
GridFunction x(fespace);
x = 0.0;
// 9. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
BilinearForm *a = new BilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
// 10. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: eliminating boundary
// conditions, applying conforming constraints for non-conforming AMR,
// static condensation, etc.
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
OperatorPtr A;
Vector B, X;
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
cout << "Size of linear system: " << A->Height() << endl;
// 11. Solve the linear system A X = B.
if (!pa)
{
if (use_ginkgo_solver)
{
#ifdef MFEM_USE_GINKGO
// Solve the linear system with CG + ILU from Ginkgo.
std::string executor = "reference";
auto exec = gko::ReferenceExecutor::create();
auto ilu_precond =
gko::preconditioner::Ilu<gko::solver::LowerTrs<>,
gko::solver::UpperTrs<>, false>::build()
.on(exec);
GinkgoWrappers::CGSolver ginkgo_solver(executor, 1, 2000, 1e-12, 0.0,
ilu_precond.release() );
ginkgo_solver.solve(&((SparseMatrix&)(*A)), X, B);
#endif
}
else
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
}
}
else // No preconditioning for now in partial assembly mode.
{
CG(*A, B, X, 1, 2000, 1e-12, 0.0);
}
// 12. Recover the solution as a finite element grid function.
a->RecoverFEMSolution(X, *b, x);
// 13. Save the refined mesh and the solution. This output can be viewed later
// using GLVis: "glvis -m refined.mesh -g sol.gf".
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh->Print(mesh_ofs);
ofstream sol_ofs("sol.gf");
sol_ofs.precision(8);
x.Save(sol_ofs);
// 14. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << *mesh << x << flush;
}
// 15. Free the used memory.
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete mesh;
return 0;
}
+81
View File
@@ -0,0 +1,81 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/ginkgo/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
# Currently there are only serial Ginkgo examples
SEQ_EXAMPLES = ex1
PAR_EXAMPLES =
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
else
EXAMPLES = $(PAR_EXAMPLES) $(SEQ_EXAMPLES)
endif
.SUFFIXES:
.SUFFIXES: .o .cpp .mk
.PHONY: all clean clean-build clean-exec
# Remove built-in rule
%: %.cpp
# Replace the default implicit rule for *.cpp files
%: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK)
$(MFEM_CXX) $(MFEM_FLAGS) $< -o $@ $(MFEM_LIBS)
all: $(EXAMPLES)
ifeq ($(MFEM_USE_GINKGO),NO)
$(EXAMPLES):
$(error MFEM is not configured with GINKO)
endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
# Testing: Parallel vs. serial runs
RUN_MPI_NP = $(MFEM_MPIEXEC) $(MFEM_MPIEXEC_NP)
RUN_MPI = $(RUN_MPI_NP) $(MFEM_MPI_NP)
SERIAL_NAME := Serial GINKGO example
PARALLEL_NAME := Parallel GINKGO example
%-test-par: %
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_NAME))
%-test-seq: %
@$(call mfem-test,$<,, $(SERIAL_NAME))
# Testing: Example-specific execution options:
ex1-test-par: ex1
@$(call mfem-test,$<, $(RUN_MPI_NP) 1, $(PARALLEL_NAME))
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
# Generate an error message if the MFEM library is not built and exit
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh sol.gf mesh.* sol.*
+63
View File
@@ -0,0 +1,63 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
set(HIOP_EXAMPLES_SRCS)
list(APPEND HIOP_EXAMPLES_SRCS ex9.cpp)
if (MFEM_USE_MPI)
list(APPEND HIOP_EXAMPLES_SRCS ex9p.cpp)
endif()
# Include the source directory where mfem.hpp and mfem-performance.hpp are.
include_directories(BEFORE ${PROJECT_BINARY_DIR})
# Add "test_hiop" target, see below.
add_custom_target(test_hiop
${CMAKE_CTEST_COMMAND} -R hiop USES_TERMINAL)
# Add one executable per cpp file, adding "hiop_" as prefix. Sets
# "test_hiop" as a target that depends on the given examples.
set(PFX hiop_)
add_mfem_examples(HIOP_EXAMPLES_SRCS ${PFX} "" test_hiop)
# Testing.
# The HIOP tests can be run separately using the target "test_hiop"
# which builds the examples and runs:
# ctest -R hiop
# Command line options for the tests.
# Example 9:
set(EX9_COMMON_OPTS -m ../../data/periodic-segment.mesh -p 0 -dt 0.005)
set(EX9_TEST_OPTS ${EX9_COMMON_OPTS} -r 2 )
set(EX9P_TEST_OPTS ${EX9_COMMON_OPTS})
# Add the tests: one test per source file.
foreach(SRC_FILE ${HIOP_EXAMPLES_SRCS})
get_filename_component(SRC_FILENAME ${SRC_FILE} NAME)
string(REPLACE ".cpp" "" TEST_NAME ${SRC_FILENAME})
string(TOUPPER ${TEST_NAME} UP_TEST_NAME)
set(TEST_NAME ${PFX}${TEST_NAME})
set(THIS_TEST_OPTIONS "-no-vis")
list(APPEND THIS_TEST_OPTIONS ${${UP_TEST_NAME}_TEST_OPTS})
# message(STATUS "Test ${TEST_NAME} options: ${THIS_TEST_OPTIONS}")
if (NOT (${TEST_NAME} MATCHES ".*p$"))
add_test(NAME ${TEST_NAME}_ser
COMMAND ${TEST_NAME} ${THIS_TEST_OPTIONS})
else()
add_test(NAME ${TEST_NAME}_np=4
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} 4
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:${TEST_NAME}> ${THIS_TEST_OPTIONS}
${MPIEXEC_POSTFLAGS})
endif()
endforeach()
+684
View File
@@ -0,0 +1,684 @@
// MFEM Example 9 with Nonlinear Constrained Optimization
//
// Compile with: make ex9
//
// Sample runs:
//
// ex9 -m ../../data/periodic-segment.mesh -r 3 -p 0 -o 2 -dt 0.002 -opt 1
// ex9 -m ../../data/periodic-segment.mesh -r 3 -p 0 -o 2 -dt 0.002 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 0 -r 2 -dt 0.01 -tf 10 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 0 -r 2 -dt 0.01 -tf 10 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 2
//
// ex9 -m ../../data/amr-quad.mesh -p 1 -r 1 -dt 0.002 -tf 9 -opt 1
// ex9 -m ../../data/amr-quad.mesh -p 1 -r 1 -dt 0.002 -tf 9 -opt 2
//
// ex9 -m ../../data/disc-nurbs.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 1
// ex9 -m ../../data/disc-nurbs.mesh -p 1 -r 2 -dt 0.005 -tf 9 -opt 2
//
// ex9 -m ../../data/disc-nurbs.mesh -p 2 -r 2 -dt 0.01 -tf 9 -opt 1
// ex9 -m ../../data/disc-nurbs.mesh -p 2 -r 2 -dt 0.01 -tf 9 -opt 2
//
// ex9 -m ../../data/periodic-square.mesh -p 3 -r 3 -dt 0.0025 -tf 9 -opt 1
// ex9 -m ../../data/periodic-square.mesh -p 3 -r 3 -dt 0.0025 -tf 9 -opt 2
//
// ex9 -m ../../data/periodic-cube.mesh -p 0 -r 2 -o 2 -dt 0.02 -tf 8 -opt 1
// ex9 -m ../../data/periodic-cube.mesh -p 0 -r 2 -o 2 -dt 0.02 -tf 8 -opt 2
// Description: This example modifies the standard MFEM ex9 by adding nonlinear
// constrained optimization capabilities through the SLBQP and
// HIOP solvers. It demonstrates how a user can define a custom
// class OptimizationProblem that includes linear/nonlinear
// equality/inequality constraints. This optimization is applied
// as post-processing to the solution of the transport equation.
//
// Description of ex9:
// This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
// Choice for the problem setup. The fluid velocity, initial condition and
// inflow boundary condition are chosen based on this parameter.
int problem;
// Nonlinear optimizer.
int optimizer_type;
// Velocity coefficient
bool invert_velocity = false;
void velocity_function(const Vector &x, Vector &v);
// Initial condition
double u0_function(const Vector &x);
// Inflow boundary condition
double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
/// Computes C(x) = sum w_i x_i, where w is a given Vector.
class LinearScaleOperator : public Operator
{
private:
const Vector &w;
mutable DenseMatrix grad;
public:
LinearScaleOperator(const Vector &weight)
: Operator(1, weight.Size()), w(weight), grad(1, width)
{
for (int i = 0; i < width; i++) { grad(0, i) = w(i); }
}
virtual void Mult(const Vector &x, Vector &y) const
{
y(0) = w * x;
}
virtual Operator &GetGradient(const Vector &x) const
{
return grad;
}
};
/// Nonlinear monotone bounded operator to test nonlinear ineq constraints.
/// Computes D(x) = tanh(sum(x_i)).
class TanhSumOperator : public Operator
{
private:
mutable DenseMatrix grad;
public:
TanhSumOperator(int size) : Operator(1, size), grad(1, width) { }
virtual void Mult(const Vector &x, Vector &y) const
{
y(0) = std::tanh(x.Sum());
}
virtual Operator &GetGradient(const Vector &x) const
{
const double ts = std::tanh(x.Sum());
const double dtanh = 1.0 - ts * ts;
for (int i = 0; i < width; i++) { grad(0, i) = dtanh; }
return grad;
}
};
/** Monotone and conservative a-posteriori correction for transport solutions:
* Find x that minimizes 0.5 || x - x_HO ||^2, subject to
* sum w_i x_i = mass,
* tanh(sum(x_i_min)) <= tanh(sum(x_i)) <= tanh(sum(x_i_max)),
* x_i_min <= x_i <= x_i_max,
*/
class OptimizedTransportProblem : public OptimizationProblem
{
private:
const Vector &x_HO;
Vector massvec, d_lo, d_hi;
const LinearScaleOperator LSoper;
const TanhSumOperator TSoper;
public:
OptimizedTransportProblem(const Vector &xho, const Vector &w, double mass,
const Vector &xmin, const Vector &xmax)
: OptimizationProblem(xho.Size(), NULL, NULL),
x_HO(xho), massvec(1), d_lo(1), d_hi(1),
LSoper(w), TSoper(w.Size())
{
C = &LSoper;
massvec(0) = mass;
SetEqualityConstraint(massvec);
D = &TSoper;
d_lo(0) = std::tanh(xmin.Sum());
d_hi(0) = std::tanh(xmax.Sum());
MFEM_ASSERT(d_lo(0) < d_hi(0),
"The bounds produce an infeasible optimization problem");
SetInequalityConstraint(d_lo, d_hi);
SetSolutionBounds(xmin, xmax);
}
virtual double CalcObjective(const Vector &x) const
{
double res = 0.0;
for (int i = 0; i < input_size; i++)
{
const double d = x(i) - x_HO(i);
res += d * d;
}
return 0.5 * res;
}
virtual void CalcObjectiveGrad(const Vector &x, Vector &grad) const
{
for (int i = 0; i < input_size; i++) { grad(i) = x(i) - x_HO(i); }
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
{
private:
SparseMatrix &M, &K;
const Vector &b;
DSmoother M_prec;
CGSolver M_solver;
mutable Vector z;
double dt;
BilinearForm &bf;
Vector &M_rowsums;
public:
FE_Evolution(SparseMatrix &_M, SparseMatrix &_K, const Vector &_b,
BilinearForm &_bf, Vector &M_rs);
void SetTimeStep(double _dt) { dt = _dt; }
void SetK(SparseMatrix &_K) { K = _K; }
virtual void Mult(const Vector &x, Vector &y) const;
virtual ~FE_Evolution() { }
};
int main(int argc, char *argv[])
{
// 1. Parse command-line options.
problem = 0;
optimizer_type = 1;
const char *mesh_file = "../../data/periodic-hexagon.mesh";
int ref_levels = 2;
int order = 3;
int ode_solver_type = 3;
double t_final = 1.0;
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool binary = false;
int vis_steps = 5;
int precision = 8;
cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ref_levels, "-r", "--refine",
"Number of times to refine the mesh uniformly.");
args.AddOption(&order, "-o", "--order",
"Order (degree) of the finite elements.");
args.AddOption(&optimizer_type, "-opt", "--optimizer",
"Nonlinear optimizer: 1 - SLBQP,\n\t"
" 2 - HIOP.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP, 4 - RK4, 6 - RK6.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
"Time step.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
// 2. Read the mesh from the given mesh file. We can handle geometrically
// periodic meshes in this code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 3. Define the ODE solver used for time integration. Several explicit
// Runge-Kutta methods are available.
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
default:
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
delete mesh;
return 3;
}
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement, where 'ref_levels' is a
// command-line parameter. If the mesh is of NURBS type, we convert it to
// a (piecewise-polynomial) high-order mesh.
for (int lev = 0; lev < ref_levels; lev++)
{
mesh->UniformRefinement();
}
if (mesh->NURBSext)
{
mesh->SetCurvature(max(order, 1));
}
mesh->GetBoundingBox(bb_min, bb_max, max(order, 1));
// 5. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
DG_FECollection fec(order, dim, BasisType::Positive);
FiniteElementSpace fes(mesh, &fec);
cout << "Number of unknowns: " << fes.GetVSize() << endl;
// 6. Set up and assemble the bilinear and linear forms corresponding to the
// DG discretization. The DGTraceIntegrator involves integrals over mesh
// interior faces.
VectorFunctionCoefficient velocity(dim, velocity_function);
FunctionCoefficient inflow(inflow_function);
FunctionCoefficient u0(u0_function);
BilinearForm m(&fes);
m.AddDomainIntegrator(new MassIntegrator);
BilinearForm k(&fes);
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k.AddInteriorFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
k.AddBdrFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
LinearForm b(&fes);
b.AddBdrFaceIntegrator(
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
m.Assemble();
m.Finalize();
int skip_zeros = 0;
k.Assemble(skip_zeros);
k.Finalize(skip_zeros);
b.Assemble();
// 7. Define the initial conditions, save the corresponding grid function to
// a file and (optionally) save data in the VisIt format and initialize
// GLVis visualization.
GridFunction u(&fes);
u.ProjectCoefficient(u0);
{
ofstream omesh("ex9.mesh");
omesh.precision(precision);
mesh->Print(omesh);
ofstream osol("ex9-init.gf");
osol.precision(precision);
u.Save(osol);
}
// Create data collection for solution output: either VisItDataCollection for
// ascii data files, or SidreDataCollection for binary data files.
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example9", mesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example9", mesh);
dc->SetPrecision(precision);
}
dc->RegisterField("solution", &u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
visualization = false;
cout << "GLVis visualization disabled.\n";
}
else
{
sout.precision(precision);
sout << "solution\n" << *mesh << u;
sout << "pause\n";
sout << flush;
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
Vector M_rowsums(m.Size());
m.SpMat().GetRowSums(M_rowsums);
// 8. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(m.SpMat(), k.SpMat(), b, k, M_rowsums);
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
// Compute initial volume.
const double vol0 = M_rowsums * u;
bool done = false;
for (int ti = 0; !done; )
{
double dt_real = min(dt, t_final - t);
adv.SetTimeStep(dt_real);
ode_solver->Step(u, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
cout << "time step: " << ti << ", time: " << t << endl;
if (visualization)
{
sout << "solution\n" << *mesh << u << flush;
}
if (visit)
{
dc->SetCycle(ti);
dc->SetTime(t);
dc->Save();
}
}
}
// Print the error vs exact solution.
const double max_error = u.ComputeMaxError(u0),
l1_error = u.ComputeL1Error(u0),
l2_error = u.ComputeL2Error(u0);
std::cout << "Linf error = " << max_error << endl
<< "L1 error = " << l1_error << endl
<< "L2 error = " << l2_error << endl;
// Print error in volume.
const double vol = M_rowsums * u;
std::cout << "Vol error = " << vol - vol0 << endl;
// 9. Save the final solution. This output can be viewed later using GLVis:
// "glvis -m ex9.mesh -g ex9-final.gf".
{
ofstream osol("ex9-final.gf");
osol.precision(precision);
u.Save(osol);
}
// 10. Free the used memory.
delete ode_solver;
delete dc;
delete mesh;
return 0;
}
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(SparseMatrix &_M, SparseMatrix &_K,
const Vector &_b, BilinearForm &_bf, Vector &M_rs)
: TimeDependentOperator(_M.Size()),
M(_M), K(_K), b(_b), M_prec(), M_solver(), z(_M.Size()),
bf(_bf), M_rowsums(M_rs)
{
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(M);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void FE_Evolution::Mult(const Vector &x, Vector &y) const
{
// Compute bounds y_min, y_max for y from x on the ldofs.
const int dofs = x.Size();
Vector y_min(dofs), y_max(dofs);
const int *In = bf.SpMat().GetI(), *Jn = bf.SpMat().GetJ();
for (int i = 0, k = 0; i < dofs; i++)
{
double x_i_min = +std::numeric_limits<double>::infinity();
double x_i_max = -std::numeric_limits<double>::infinity();
for (int end = In[i+1]; k < end; k++)
{
const int j = Jn[k];
if (x(j) > x_i_max) { x_i_max = x(j); }
if (x(j) < x_i_min) { x_i_min = x(j); }
}
y_min(i) = x_i_min;
y_max(i) = x_i_max;
}
for (int i = 0; i < dofs; i++)
{
y_min(i) = (y_min(i) - x(i) ) / dt;
y_max(i) = (y_max(i) - x(i) ) / dt;
}
// Compute the high-order solution y = M^{-1} (K x + b).
K.Mult(x, z);
z += b;
M_solver.Mult(z, y);
// The solution y is an increment; it should not introduce new mass.
const double mass_y = 0.0;
// Perform optimization.
Vector y_out(dofs);
const int max_iter = 500;
const double rtol = 1.e-7;
double atol = 1.e-7;
OptimizationSolver *optsolver = NULL;
if (optimizer_type == 2)
{
#ifdef MFEM_USE_HIOP
HiopNlpOptimizer *tmp_opt_ptr = new HiopNlpOptimizer();
optsolver = tmp_opt_ptr;
#else
MFEM_ABORT("MFEM is not built with HiOp support!");
#endif
}
else
{
SLBQPOptimizer *slbqp = new SLBQPOptimizer();
slbqp->SetBounds(y_min, y_max);
slbqp->SetLinearConstraint(M_rowsums, mass_y);
atol = 1.e-15;
optsolver = slbqp;
}
OptimizedTransportProblem ot_prob(y, M_rowsums, mass_y, y_min, y_max);
optsolver->SetOptimizationProblem(ot_prob);
optsolver->SetMaxIter(max_iter);
optsolver->SetAbsTol(atol);
optsolver->SetRelTol(rtol);
optsolver->SetPrintLevel(0);
optsolver->Mult(y, y_out);
y = y_out;
delete optsolver;
}
// Velocity coefficient
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = (invert_velocity) ? -1.0 : 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const double w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const double w = M_PI/2;
double d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
double u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return (X(0) > -0.15 && X(0) < 0.15) ? 1.0 : 0.0;
//return exp(-40.*pow(X(0)-0.0,2));
case 2:
case 3:
{
double rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const double s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( erfc(w*(X(0)-cx-rx))*erfc(-w*(X(0)-cx+rx)) *
erfc(w*(X(1)-cy-ry))*erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
double x_ = X(0), y_ = X(1), rho, phi;
rho = hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const double f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
// Inflow boundary condition (zero for the problems considered in this example)
double inflow_function(const Vector &x)
{
switch (problem)
{
case 0:
case 1:
case 2:
case 3: return 0.0;
}
return 0.0;
}
+802
View File
@@ -0,0 +1,802 @@
// MFEM Example 9 with Nonlinear Constrained Optimization - Parallel Version
//
// Compile with: make ex9p
//
// Sample runs:
//
// mpirun -np 4 ex9p -m ../../data/periodic-segment.mesh -rs 3 -p 0 -o 2 -dt 0.002 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-segment.mesh -rs 3 -p 0 -o 2 -dt 0.002 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 0 -rs 2 -dt 0.01 -tf 10 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 0 -rs 2 -dt 0.01 -tf 10 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/amr-quad.mesh -p 1 -rs 1 -dt 0.002 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/amr-quad.mesh -p 1 -rs 1 -dt 0.002 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 1 -rs 2 -dt 0.005 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 2 -rs 2 -dt 0.01 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/disc-nurbs.mesh -p 2 -rs 2 -dt 0.01 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 3 -rs 3 -dt 0.0025 -tf 9 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-square.mesh -p 3 -rs 3 -dt 0.0025 -tf 9 -opt 2
//
// mpirun -np 4 ex9p -m ../../data/periodic-cube.mesh -p 0 -rs 2 -o 2 -dt 0.02 -tf 8 -opt 1
// mpirun -np 4 ex9p -m ../../data/periodic-cube.mesh -p 0 -rs 2 -o 2 -dt 0.02 -tf 8 -opt 2
// Description: This example modifies the standard MFEM ex9 by adding nonlinear
// constrained optimization capabilities through the SLBQP and
// HIOP solvers. It demonstrates how a user can define a custom
// class OptimizationProblem that includes linear/nonlinear
// equality/inequality constraints. This optimization is applied
// as post-processing to the solution of the transport equation.
//
// Description of ex9:
// This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
// u0(x)=u(0,x) is a given initial condition.
//
// The example demonstrates the use of Discontinuous Galerkin (DG)
// bilinear forms in MFEM (face integrators), the use of explicit
// ODE time integrators, the definition of periodic boundary
// conditions through periodic meshes, as well as the use of GLVis
// for persistent visualization of a time-evolving solution. The
// saving of time-dependent data files for external visualization
// with VisIt (visit.llnl.gov) is also illustrated.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
// Choice for the problem setup. The fluid velocity, initial condition and
// inflow boundary condition are chosen based on this parameter.
int problem;
// Nonlinear optimizer.
int optimizer_type;
// Velocity coefficient
bool invert_velocity = false;
void velocity_function(const Vector &x, Vector &v);
// Initial condition
double u0_function(const Vector &x);
// Inflow boundary condition
double inflow_function(const Vector &x);
// Mesh bounding box
Vector bb_min, bb_max;
/// Computes C(x) = sum w_i x_i, where w is a given Vector.
class LinearScaleOperator : public Operator
{
private:
ParFiniteElementSpace &pfes;
// Local weights.
const Vector &w;
// Gradient for the tdofs.
mutable DenseMatrix grad;
public:
LinearScaleOperator(ParFiniteElementSpace &space, const Vector &weight)
: Operator(1, space.TrueVSize()),
pfes(space), w(weight), grad(1, width)
{
Vector w_glob(width);
pfes.Dof_TrueDof_Matrix()->MultTranspose(w, w_glob);
for (int i = 0; i < width; i++) { grad(0, i) = w_glob(i); }
}
virtual void Mult(const Vector &x, Vector &y) const
{
Vector x_loc(w.Size());
pfes.GetProlongationMatrix()->Mult(x, x_loc);
const double loc_res = w * x_loc;
MPI_Allreduce(&loc_res, &y(0), 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
}
virtual Operator &GetGradient(const Vector &x) const
{
return grad;
}
};
/// Nonlinear monotone bounded operator to test nonlinear ineq constraints.
/// Computes D(x) = tanh(sum(x_i)).
class TanhSumOperator : public Operator
{
private:
// Gradient for the tdofs.
mutable DenseMatrix grad;
public:
TanhSumOperator(ParFiniteElementSpace &space)
: Operator(1, space.TrueVSize()), grad(1, width) { }
virtual void Mult(const Vector &x, Vector &y) const
{
double sum_loc = x.Sum();
MPI_Allreduce(&sum_loc, &y(0), 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
y(0) = std::tanh(y(0));
}
virtual Operator &GetGradient(const Vector &x) const
{
double sum_loc = x.Sum();
double dtanh;
MPI_Allreduce(&sum_loc, &dtanh, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
dtanh = 1.0 - pow(std::tanh(dtanh), 2);
for (int i = 0; i < width; i++) { grad(0, i) = dtanh; }
return grad;
}
};
/** Monotone and conservative a-posteriori correction for transport solutions:
* Find x that minimizes 0.5 || x - x_HO ||^2, subject to
* sum w_i x_i = mass,
* tanh(sum(x_i_min)) <= tanh(sum(x_i)) <= tanh(sum(x_i_max)),
* x_i_min <= x_i <= x_i_max,
*/
class OptimizedTransportProblem : public OptimizationProblem
{
private:
const Vector &x_HO;
Vector massvec, d_lo, d_hi;
const LinearScaleOperator LSoper;
const TanhSumOperator TSoper;
public:
OptimizedTransportProblem(ParFiniteElementSpace &space,
const Vector &xho, const Vector &w, double mass,
const Vector &xmin, const Vector &xmax)
: OptimizationProblem(xho.Size(), NULL, NULL),
x_HO(xho), massvec(1), d_lo(1), d_hi(1),
LSoper(space, w), TSoper(space)
{
C = &LSoper;
massvec(0) = mass;
SetEqualityConstraint(massvec);
D = &TSoper;
double lsums[2], gsums[2];
lsums[0] = xmin.Sum();
lsums[1] = xmax.Sum();
MPI_Allreduce(lsums, gsums, 2, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
d_lo(0) = std::tanh(gsums[0]);
d_hi(0) = std::tanh(gsums[1]);
MFEM_ASSERT(d_lo(0) < d_hi(0),
"The bounds produce an infeasible optimization problem");
SetInequalityConstraint(d_lo, d_hi);
SetSolutionBounds(xmin, xmax);
}
virtual double CalcObjective(const Vector &x) const
{
double loc_res = 0.0;
for (int i = 0; i < input_size; i++)
{
const double d = x(i) - x_HO(i);
loc_res += d * d;
}
loc_res *= 0.5;
double res;
MPI_Allreduce(&loc_res, &res, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
return res;
}
virtual void CalcObjectiveGrad(const Vector &x, Vector &grad) const
{
for (int i = 0; i < input_size; i++) { grad(i) = x(i) - x_HO(i); }
}
};
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
{
private:
HypreParMatrix &M, &K;
const Vector &b;
HypreSmoother M_prec;
CGSolver M_solver;
mutable Vector z;
double dt;
ParBilinearForm &pbf;
Vector &M_rowsums;
public:
FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
const Vector &_b, ParBilinearForm &_pbf, Vector &M_rs);
void SetTimeStep(double _dt) { dt = _dt; }
void SetK(HypreParMatrix &_K) { K = _K; }
virtual void Mult(const Vector &x, Vector &y) const;
virtual ~FE_Evolution() { }
};
int main(int argc, char *argv[])
{
// 1. Initialize MPI.
int num_procs, myid;
MPI_Init(&argc, &argv);
MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
MPI_Comm_rank(MPI_COMM_WORLD, &myid);
// 2. Parse command-line options.
problem = 0;
optimizer_type = 1;
const char *mesh_file = "../../data/periodic-hexagon.mesh";
int ser_ref_levels = 2;
int par_ref_levels = 0;
int order = 3;
int ode_solver_type = 3;
double t_final = 1.0;
double dt = 0.01;
bool visualization = true;
bool visit = false;
bool binary = false;
int vis_steps = 5;
int precision = 8;
cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&problem, "-p", "--problem",
"Problem setup to use. See options in velocity_function().");
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
"Number of times to refine the mesh uniformly in serial.");
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
"Number of times to refine the mesh uniformly in parallel.");
args.AddOption(&order, "-o", "--order",
"Order (degree) of the finite elements.");
args.AddOption(&optimizer_type, "-opt", "--optimizer",
"Nonlinear optimizer: 1 - SLBQP,\n\t"
" 2 - HIOP.");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
"ODE solver: 1 - Forward Euler,\n\t"
" 2 - RK2 SSP, 3 - RK3 SSP.");
args.AddOption(&t_final, "-tf", "--t-final",
"Final time; start time is 0.");
args.AddOption(&dt, "-dt", "--time-step",
"Time step.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&visit, "-visit", "--visit-datafiles", "-no-visit",
"--no-visit-datafiles",
"Save data files for VisIt (visit.llnl.gov) visualization.");
args.AddOption(&binary, "-binary", "--binary-datafiles", "-ascii",
"--ascii-datafiles",
"Use binary (Sidre) or ascii format for VisIt data files.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.Parse();
if (!args.Good())
{
if (myid == 0) { args.PrintUsage(cout); }
MPI_Finalize();
return 1;
}
if (myid == 0) { args.PrintOptions(cout); }
// 3. Read the serial mesh from the given mesh file on all processors. We can
// handle geometrically periodic meshes in this code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Define the ODE solver used for time integration. Several explicit
// Runge-Kutta methods are available.
ODESolver *ode_solver = NULL;
switch (ode_solver_type)
{
case 1: ode_solver = new ForwardEulerSolver; break;
case 2: ode_solver = new RK2Solver(1.0); break;
case 3: ode_solver = new RK3SSPSolver; break;
case 4: ode_solver = new RK4Solver; break;
case 6: ode_solver = new RK6Solver; break;
default:
if (myid == 0)
{
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
}
delete mesh;
MPI_Finalize();
return 3;
}
// 5. Refine the mesh in serial to increase the resolution. In this example
// we do 'ser_ref_levels' of uniform refinement, where 'ser_ref_levels' is
// a command-line parameter. If the mesh is of NURBS type, we convert it
// to a (piecewise-polynomial) high-order mesh.
for (int lev = 0; lev < ser_ref_levels; lev++)
{
mesh->UniformRefinement();
}
if (mesh->NURBSext)
{
mesh->SetCurvature(max(order, 1));
}
mesh->GetBoundingBox(bb_min, bb_max, max(order, 1));
// 6. Define the parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
for (int lev = 0; lev < par_ref_levels; lev++)
{
pmesh->UniformRefinement();
}
// 7. Define the parallel discontinuous DG finite element space on the
// parallel refined mesh of the given polynomial order.
DG_FECollection fec(order, dim, BasisType::Positive);
ParFiniteElementSpace *fes = new ParFiniteElementSpace(pmesh, &fec);
HYPRE_Int global_vSize = fes->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of unknowns: " << global_vSize << endl;
}
// 8. Set up and assemble the parallel bilinear and linear forms (and the
// parallel hypre matrices) corresponding to the DG discretization. The
// DGTraceIntegrator involves integrals over mesh interior faces.
VectorFunctionCoefficient velocity(dim, velocity_function);
FunctionCoefficient inflow(inflow_function);
FunctionCoefficient u0(u0_function);
ParBilinearForm *m = new ParBilinearForm(fes);
m->AddDomainIntegrator(new MassIntegrator);
ParBilinearForm *k = new ParBilinearForm(fes);
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k->AddInteriorFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
k->AddBdrFaceIntegrator(
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
ParLinearForm *b = new ParLinearForm(fes);
b->AddBdrFaceIntegrator(
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
m->Assemble();
m->Finalize();
int skip_zeros = 0;
k->Assemble(skip_zeros);
k->Finalize(skip_zeros);
b->Assemble();
HypreParMatrix *M = m->ParallelAssemble();
HypreParMatrix *K = k->ParallelAssemble();
HypreParVector *B = b->ParallelAssemble();
// 9. Define the initial conditions, save the corresponding grid function to
// a file and (optionally) save data in the VisIt format and initialize
// GLVis visualization.
ParGridFunction *u = new ParGridFunction(fes);
u->ProjectCoefficient(u0);
HypreParVector *U = u->GetTrueDofs();
{
ostringstream mesh_name, sol_name;
mesh_name << "ex9-mesh." << setfill('0') << setw(6) << myid;
sol_name << "ex9-init." << setfill('0') << setw(6) << myid;
ofstream omesh(mesh_name.str().c_str());
omesh.precision(precision);
pmesh->Print(omesh);
ofstream osol(sol_name.str().c_str());
osol.precision(precision);
u->Save(osol);
}
// Create data collection for solution output: either VisItDataCollection for
// ascii data files, or SidreDataCollection for binary data files.
DataCollection *dc = NULL;
if (visit)
{
if (binary)
{
#ifdef MFEM_USE_SIDRE
dc = new SidreDataCollection("Example9-Parallel", pmesh);
#else
MFEM_ABORT("Must build with MFEM_USE_SIDRE=YES for binary output.");
#endif
}
else
{
dc = new VisItDataCollection("Example9-Parallel", pmesh);
dc->SetPrecision(precision);
// To save the mesh using MFEM's parallel mesh format:
// dc->SetFormat(DataCollection::PARALLEL_FORMAT);
}
dc->RegisterField("solution", u);
dc->SetCycle(0);
dc->SetTime(0.0);
dc->Save();
}
socketstream sout;
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
sout.open(vishost, visport);
if (!sout)
{
if (myid == 0)
cout << "Unable to connect to GLVis server at "
<< vishost << ':' << visport << endl;
visualization = false;
if (myid == 0)
{
cout << "GLVis visualization disabled.\n";
}
}
else
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout.precision(precision);
sout << "solution\n" << *pmesh << *u;
sout << "pause\n";
sout << flush;
if (myid == 0)
cout << "GLVis visualization paused."
<< " Press space (in the GLVis window) to resume it.\n";
}
}
Vector M_rowsums(m->Size());
m->SpMat().GetRowSums(M_rowsums);
// 10. Define the time-dependent evolution operator describing the ODE
// right-hand side, and perform time-integration (looping over the time
// iterations, ti, with a time-step dt).
FE_Evolution adv(*M, *K, *B, *k, M_rowsums);
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
*u = *U;
// Compute initial volume.
const double vol0_loc = M_rowsums * (*u);
double vol0;
MPI_Allreduce(&vol0_loc, &vol0, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
bool done = false;
for (int ti = 0; !done; )
{
double dt_real = min(dt, t_final - t);
adv.SetTimeStep(dt_real);
ode_solver->Step(*U, t, dt_real);
ti++;
done = (t >= t_final - 1e-8*dt);
if (done || ti % vis_steps == 0)
{
if (myid == 0)
{
cout << "time step: " << ti << ", time: " << t << endl;
}
// 11. Extract the parallel grid function corresponding to the finite
// element approximation U (the local solution on each processor).
*u = *U;
if (visualization)
{
sout << "parallel " << num_procs << " " << myid << "\n";
sout << "solution\n" << *pmesh << *u << flush;
}
if (visit)
{
dc->SetCycle(ti);
dc->SetTime(t);
dc->Save();
}
}
}
// Print the error vs exact solution.
const double max_error = u->ComputeMaxError(u0),
l1_error = u->ComputeL1Error(u0),
l2_error = u->ComputeL2Error(u0);
if (myid == 0)
{
std::cout << "Linf error = " << max_error << endl
<< "L1 error = " << l1_error << endl
<< "L2 error = " << l2_error << endl;
}
// Print error in volume.
const double vol_loc = M_rowsums * (*u);
double vol;
MPI_Allreduce(&vol_loc, &vol, 1, MPI_DOUBLE, MPI_SUM, MPI_COMM_WORLD);
if (myid == 0)
{
std::cout << "Vol error = " << vol - vol0 << endl;
}
// 12. Save the final solution in parallel. This output can be viewed later
// using GLVis: "glvis -np <np> -m ex9-mesh -g ex9-final".
{
*u = *U;
ostringstream sol_name;
sol_name << "ex9-final." << setfill('0') << setw(6) << myid;
ofstream osol(sol_name.str().c_str());
osol.precision(precision);
u->Save(osol);
}
// 13. Free the used memory.
delete U;
delete u;
delete B;
delete b;
delete K;
delete k;
delete M;
delete m;
delete fes;
delete pmesh;
delete ode_solver;
delete dc;
MPI_Finalize();
return 0;
}
// Implementation of class FE_Evolution
FE_Evolution::FE_Evolution(HypreParMatrix &_M, HypreParMatrix &_K,
const Vector &_b, ParBilinearForm &_pbf,
Vector &M_rs)
: TimeDependentOperator(_M.Height()),
M(_M), K(_K), b(_b), M_solver(M.GetComm()), z(_M.Height()),
pbf(_pbf), M_rowsums(M_rs)
{
M_prec.SetType(HypreSmoother::Jacobi);
M_solver.SetPreconditioner(M_prec);
M_solver.SetOperator(M);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
M_solver.SetAbsTol(0.0);
M_solver.SetMaxIter(100);
M_solver.SetPrintLevel(0);
}
void FE_Evolution::Mult(const Vector &x, Vector &y) const
{
// Get values on the ldofs.
ParFiniteElementSpace *pfes = pbf.ParFESpace();
ParGridFunction x_gf(pfes);
pfes->GetProlongationMatrix()->Mult(x, x_gf);
// Compute bounds y_min, y_max for y from from x on the ldofs.
const int ldofs = x_gf.Size();
Vector y_min(ldofs), y_max(ldofs);
x_gf.ExchangeFaceNbrData();
Vector &x_nd = x_gf.FaceNbrData();
const int *In = pbf.SpMat().GetI(), *Jn = pbf.SpMat().GetJ();
for (int i = 0, k = 0; i < ldofs; i++)
{
double x_i_min = +std::numeric_limits<double>::infinity();
double x_i_max = -std::numeric_limits<double>::infinity();
for (int end = In[i+1]; k < end; k++)
{
const int j = Jn[k];
const double x_j = (j < ldofs) ? x(j): x_nd(j-ldofs);
if (x_j > x_i_max) { x_i_max = x_j; }
if (x_j < x_i_min) { x_i_min = x_j; }
}
y_min(i) = x_i_min;
y_max(i) = x_i_max;
}
for (int i = 0; i < ldofs; i++)
{
y_min(i) = (y_min(i) - x_gf(i) ) / dt;
y_max(i) = (y_max(i) - x_gf(i) ) / dt;
}
Vector y_min_tdofs(y.Size()), y_max_tdofs(y.Size());
// Move the bounds to the tdofs.
pfes->GetRestrictionMatrix()->Mult(y_min, y_min_tdofs);
pfes->GetRestrictionMatrix()->Mult(y_max, y_max_tdofs);
// Compute the high-order solution y = M^{-1} (K x + b) on the tdofs.
K.Mult(x, z);
z += b;
M_solver.Mult(z, y);
// The solution y is an increment; it should not introduce new mass.
const double mass_y = 0.0;
// Perform optimization on the tdofs.
Vector y_out(y.Size());
const int max_iter = 500;
const double rtol = 1.e-7;
double atol = 1.e-7;
OptimizationSolver* optsolver = NULL;
if (optimizer_type == 2)
{
#ifdef MFEM_USE_HIOP
HiopNlpOptimizer *tmp_opt_ptr = new HiopNlpOptimizer(MPI_COMM_WORLD);
optsolver = tmp_opt_ptr;
#else
MFEM_ABORT("MFEM is not built with HiOp support!");
#endif
}
else
{
SLBQPOptimizer *slbqp = new SLBQPOptimizer(MPI_COMM_WORLD);
slbqp->SetBounds(y_min_tdofs, y_max_tdofs);
slbqp->SetLinearConstraint(M_rowsums, mass_y);
atol = 1.e-15;
optsolver = slbqp;
}
OptimizedTransportProblem ot_prob(*pfes, y, M_rowsums, mass_y,
y_min_tdofs, y_max_tdofs);
optsolver->SetOptimizationProblem(ot_prob);
optsolver->SetMaxIter(max_iter);
optsolver->SetAbsTol(atol);
optsolver->SetRelTol(rtol);
optsolver->SetPrintLevel(0);
optsolver->Mult(y, y_out);
y = y_out;
delete optsolver;
}
// Velocity coefficient
void velocity_function(const Vector &x, Vector &v)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
{
// Translations in 1D, 2D, and 3D
switch (dim)
{
case 1: v(0) = (invert_velocity) ? -1.0 : 1.0; break;
case 2: v(0) = sqrt(2./3.); v(1) = sqrt(1./3.); break;
case 3: v(0) = sqrt(3./6.); v(1) = sqrt(2./6.); v(2) = sqrt(1./6.);
break;
}
break;
}
case 1:
case 2:
{
// Clockwise rotation in 2D around the origin
const double w = M_PI/2;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = w*X(1); v(1) = -w*X(0); break;
case 3: v(0) = w*X(1); v(1) = -w*X(0); v(2) = 0.0; break;
}
break;
}
case 3:
{
// Clockwise twisting rotation in 2D around the origin
const double w = M_PI/2;
double d = max((X(0)+1.)*(1.-X(0)),0.) * max((X(1)+1.)*(1.-X(1)),0.);
d = d*d;
switch (dim)
{
case 1: v(0) = 1.0; break;
case 2: v(0) = d*w*X(1); v(1) = -d*w*X(0); break;
case 3: v(0) = d*w*X(1); v(1) = -d*w*X(0); v(2) = 0.0; break;
}
break;
}
}
}
// Initial condition
double u0_function(const Vector &x)
{
int dim = x.Size();
// map to the reference [-1,1] domain
Vector X(dim);
for (int i = 0; i < dim; i++)
{
double center = (bb_min[i] + bb_max[i]) * 0.5;
X(i) = 2 * (x(i) - center) / (bb_max[i] - bb_min[i]);
}
switch (problem)
{
case 0:
case 1:
{
switch (dim)
{
case 1:
return (X(0) > -0.15 && X(0) < 0.15) ? 1.0 : 0.0;
//return exp(-40.*pow(X(0)-0.0,2));
case 2:
case 3:
{
double rx = 0.45, ry = 0.25, cx = 0., cy = -0.2, w = 10.;
if (dim == 3)
{
const double s = (1. + 0.25*cos(2*M_PI*X(2)));
rx *= s;
ry *= s;
}
return ( erfc(w*(X(0)-cx-rx))*erfc(-w*(X(0)-cx+rx)) *
erfc(w*(X(1)-cy-ry))*erfc(-w*(X(1)-cy+ry)) )/16;
}
}
}
case 2:
{
double x_ = X(0), y_ = X(1), rho, phi;
rho = hypot(x_, y_);
phi = atan2(y_, x_);
return pow(sin(M_PI*rho),2)*sin(3*phi);
}
case 3:
{
const double f = M_PI;
return sin(f*X(0))*sin(f*X(1));
}
}
return 0.0;
}
// Inflow boundary condition (zero for the problems considered in this example)
double inflow_function(const Vector &x)
{
switch (problem)
{
case 0:
case 1:
case 2:
case 3: return 0.0;
}
return 0.0;
}
+66
View File
@@ -0,0 +1,66 @@
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
# See file COPYRIGHT for details.
#
# This file is part of the MFEM library. For more information and source code
# availability see http://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the GNU Lesser General Public License (as published by the Free
# Software Foundation) version 2.1 dated February 1999.
# Use the MFEM build directory
MFEM_DIR ?= ../..
MFEM_BUILD_DIR ?= ../..
SRC = $(if $(MFEM_DIR:../..=),$(MFEM_DIR)/examples/hiop/,)
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
# Use the MFEM install directory
# MFEM_INSTALL_DIR = ../../mfem
# CONFIG_MK = $(MFEM_INSTALL_DIR)/share/mfem/config.mk
MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_EXAMPLES = ex9
PAR_EXAMPLES = ex9p
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
else
EXAMPLES = $(PAR_EXAMPLES) $(SEQ_EXAMPLES)
endif
.SUFFIXES:
.SUFFIXES: .o .cpp .mk
.PHONY: all clean clean-build clean-exec
# Remove built-in rule
%: %.cpp
# Replace the default implicit rule for *.cpp files
%: $(SRC)%.cpp $(MFEM_LIB_FILE) $(CONFIG_MK)
$(MFEM_CXX) $(MFEM_FLAGS) $< -o $@ $(MFEM_LIBS)
all: $(EXAMPLES)
ifeq ($(MFEM_USE_HIOP),NO)
$(EXAMPLES):
$(error MFEM is not configured with HIOP)
endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
# Generate an error message if the MFEM library is not built and exit
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.* Example9*
+11 -4
View File
@@ -22,9 +22,9 @@ MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_EXAMPLES = ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 ex17\
ex18 ex19 ex20 ex21
ex18 ex19 ex20 ex21 ex22
PAR_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p ex12p\
ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p
ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p
ifeq ($(MFEM_USE_MPI),NO)
EXAMPLES = $(SEQ_EXAMPLES)
@@ -41,6 +41,13 @@ endif
ifeq ($(MFEM_USE_PUMI),YES)
SUBDIRS += pumi
endif
ifeq ($(MFEM_USE_HIOP),YES)
SUBDIRS += hiop
endif
ifeq ($(MFEM_USE_GINKGO),YES)
SUBDIRS += ginkgo
endif
SUBDIRS_ALL = $(addsuffix /all,$(SUBDIRS))
SUBDIRS_TEST = $(addsuffix /test,$(SUBDIRS))
SUBDIRS_CLEAN = $(addsuffix /clean,$(SUBDIRS))
@@ -117,8 +124,8 @@ clean-build:
clean-exec:
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh
@rm -rf Example5* Example9* Example15* Example16*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.*
@rm -rf Example5* Example9* Example15* Example16* PVExample*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f deformed.* velocity.* elastic_energy.* mode_*
@rm -f ex16.mesh ex16-mesh.* ex16-init.* ex16-final.*
+22
View File
@@ -14,8 +14,13 @@ set(SRCS
bilinearform_ext.cpp
bilininteg.cpp
bilininteg_diffusion.cpp
bilininteg_divergence.cpp
bilininteg_gradient.cpp
bilininteg_mass.cpp
bilininteg_vecdiffusion.cpp
bilininteg_vecmass.cpp
coefficient.cpp
complex_fem.cpp
datacollection.cpp
eltrans.cpp
estimators.cpp
@@ -29,10 +34,13 @@ set(SRCS
linearform.cpp
lininteg.cpp
nonlinearform.cpp
nonlinearform_ext.cpp
nonlininteg.cpp
nonlininteg_vectorconvection.cpp
staticcond.cpp
tmop.cpp
tmop_tools.cpp
gslib.cpp
)
set(HDRS
@@ -40,6 +48,7 @@ set(HDRS
bilinearform_ext.hpp
bilininteg.hpp
coefficient.hpp
complex_fem.hpp
datacollection.hpp
eltrans.hpp
estimators.hpp
@@ -54,6 +63,7 @@ set(HDRS
linearform.hpp
lininteg.hpp
nonlinearform.hpp
nonlinearform_ext.hpp
nonlininteg.hpp
staticcond.hpp
tbilinearform.hpp
@@ -66,6 +76,7 @@ set(HDRS
tintrules.hpp
tmop.hpp
tmop_tools.hpp
gslib.hpp
)
if (MFEM_USE_SIDRE)
@@ -95,6 +106,17 @@ if (MFEM_USE_MPI)
pnonlinearform.hpp)
endif()
if (MFEM_USE_CEED)
list(APPEND SRCS
libceed/ceed.cpp
libceed/diffusion.cpp
libceed/mass.cpp)
list(APPEND HDRS
libceed/ceed.hpp
libceed/diffusion.hpp
libceed/mass.hpp)
endif()
convert_filenames_to_full_paths(SRCS)
convert_filenames_to_full_paths(HDRS)
+33
View File
@@ -0,0 +1,33 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "fem.hpp"
#include "../general/forall.hpp"
#include "adnonlininteg.hpp"
namespace mfem
{
ADNonlinearFormIntegrator::ADNonlinearFormIntegrator()
{
}
ADNonlinearFormIntegrator::~ADNonlinearFormIntegrator()
{
}
}
+78
View File
@@ -0,0 +1,78 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_ADNONLININTEG
#define MFEM_ADNONLININTEG
#include "../config/config.hpp"
#include "fe.hpp"
#include "coefficient.hpp"
#include "fespace.hpp"
#include "nonlininteg.hpp"
#include "tadvectro.hpp"
#include "taddensmat.hpp"
namespace mfem
{
#if define(MFEM_USE_ADEPT)||define(MFEM_USE_CODIPACK)
/** The abstract base class ADNonlinearFormIntegrator is
a generalization of the NonlinearFormIntegrator class suitable
for algorithmic differentiation.
All derived classes must implement ADAssembleElementVector(...);
and ADGetElementEnergy(...); */
class ADNonlinearFormIntegrator: public NonlinearFormIntegrator
{
protected:
#ifdef MFEM_USE_ADEPT
#elseif MFEM_USE_CODIPACK
#endif
public:
ADNonlinearFormIntegrator();
virtual ~ADNonlinearFormIntegrator();
/// Methods called by the AD routines
virtual void ADGetElementEnergy(const mfem::FiniteElement & el,
mfem::ElementTransformation & Tr,
const mfem::TADVector<adouble> & elfun);
virtual void ADAssembleElementVector(const mfem::FiniteElement & el,
mfem::ElementTransformation & Tr,
const mfem::TADVector<adouble> & elfun,
mfem::TADVector<adouble> &elvec);
/// Perform the local action of the NonlinearFormIntegrator
virtual void AssembleElementVector(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun, Vector &elvect) override;
virtual void AssembleElementGrad(const mfem::FiniteElement & el,
mfem::ElementTransformation & Tr,
const mfem::Vector & elfun,
mfem::DenseMatrix & elmat) override;
virtual double GetElementEnergy(const mfem::FiniteElement & el,
mfem::ElementTransformation & Tr,
const mfem::Vector & elfun) override;
};
#endif
}
+205 -14
View File
@@ -204,7 +204,7 @@ void BilinearForm::UseSparsity(SparseMatrix &A)
<< A.Height() << " x " << A.Width());
MFEM_ASSERT(A.Finalized(), "matrix A must be Finalized");
UseSparsity(A.GetI(), A.GetJ(), A.areColumnsSorted());
UseSparsity(A.GetI(), A.GetJ(), A.ColumnsAreSorted());
}
double& BilinearForm::Elem (int i, int j)
@@ -608,18 +608,41 @@ void BilinearForm::ConformingAssemble()
width = mat->Width();
}
void BilinearForm::AssembleDiagonal(Vector &diag) const
{
if (ext)
{
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
"Vector for holding diagonal has wrong size!");
const Operator *P = fes->GetProlongationMatrix();
if (!IsIdentityProlongation(P))
{
Vector local_diag(P->Height());
ext->AssembleDiagonal(local_diag);
P->MultTranspose(local_diag, diag);
}
else
{
ext->AssembleDiagonal(diag);
}
}
else
{
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
"matrix and use SparseMatrix::GetDiag?");
}
}
void BilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x,
Vector &b, OperatorHandle &A, Vector &X,
Vector &B, int copy_interior)
{
const SparseMatrix *P = fes->GetConformingProlongation();
if (ext)
{
ext->FormLinearSystem(ess_tdof_list, x, b, A, X, B, copy_interior);
return;
}
const SparseMatrix *P = fes->GetConformingProlongation();
FormSystemMatrix(ess_tdof_list, A);
// Transform the system and perform the elimination in B, based on the
@@ -960,6 +983,18 @@ void BilinearForm::EliminateVDofsInRHS(
mat->PartMult(vdofs, x, b);
}
void BilinearForm::Mult(const Vector &x, Vector &y) const
{
if (ext)
{
ext->Mult(x, y);
}
else
{
mat->Mult(x, y);
}
}
void BilinearForm::Update(FiniteElementSpace *nfes)
{
bool full_update;
@@ -1035,7 +1070,11 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
trial_fes = tr_fes;
test_fes = te_fes;
mat = NULL;
mat_e = NULL;
extern_bfs = 0;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
@@ -1046,6 +1085,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
trial_fes = tr_fes;
test_fes = te_fes;
mat = NULL;
mat_e = NULL;
extern_bfs = 1;
// Copy the pointers to the integrators
@@ -1056,6 +1096,38 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
bbfi_marker = mbf->bbfi_marker;
btfbfi_marker = mbf->btfbfi_marker;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
{
if (ext)
{
MFEM_ABORT("the assembly level has already been set!");
}
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::FULL:
// ext = new FAMixedBilinearFormExtension(this);
// Use the original BilinearForm implementation for now
break;
case AssemblyLevel::ELEMENT:
mfem_error("Element assembly not supported yet... stay tuned!");
// ext = new EAMixedBilinearFormExtension(this);
break;
case AssemblyLevel::PARTIAL:
ext = new PAMixedBilinearFormExtension(this);
break;
case AssemblyLevel::NONE:
mfem_error("Matrix-free action not supported yet... stay tuned!");
// ext = new MFMixedBilinearFormExtension(this);
break;
default:
mfem_error("Unknown assembly level");
}
}
double & MixedBilinearForm::Elem (int i, int j)
@@ -1068,31 +1140,63 @@ const double & MixedBilinearForm::Elem (int i, int j) const
return (*mat)(i, j);
}
void MixedBilinearForm::Mult (const Vector & x, Vector & y) const
void MixedBilinearForm::Mult(const Vector & x, Vector & y) const
{
mat -> Mult (x, y);
y = 0.0;
AddMult(x, y);
}
void MixedBilinearForm::AddMult (const Vector & x, Vector & y,
const double a) const
void MixedBilinearForm::AddMult(const Vector & x, Vector & y,
const double a) const
{
mat -> AddMult (x, y, a);
if (ext)
{
ext->AddMult(x, y, a);
}
else
{
mat->AddMult(x, y, a);
}
}
void MixedBilinearForm::AddMultTranspose (const Vector & x, Vector & y,
const double a) const
void MixedBilinearForm::MultTranspose(const Vector & x, Vector & y) const
{
mat -> AddMultTranspose (x, y, a);
y = 0.0;
AddMultTranspose(x, y);
}
void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
const double a) const
{
if (ext)
{
ext->AddMultTranspose(x, y, a);
}
else
{
mat->AddMultTranspose(x, y, a);
}
}
MatrixInverse * MixedBilinearForm::Inverse() const
{
return mat -> Inverse ();
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
return NULL;
}
else
{
return mat -> Inverse ();
}
}
void MixedBilinearForm::Finalize (int skip_zeros)
{
mat -> Finalize (skip_zeros);
if (assembly == AssemblyLevel::FULL)
{
mat -> Finalize (skip_zeros);
}
}
void MixedBilinearForm::GetBlocks(Array2D<SparseMatrix *> &blocks) const
@@ -1145,6 +1249,12 @@ void MixedBilinearForm::AddBdrTraceFaceIntegrator(BilinearFormIntegrator *bfi,
void MixedBilinearForm::Assemble (int skip_zeros)
{
if (ext)
{
ext->Assemble();
return;
}
Array<int> tr_vdofs, te_vdofs;
ElementTransformation *eltrans;
DenseMatrix elemmat;
@@ -1312,6 +1422,12 @@ void MixedBilinearForm::Assemble (int skip_zeros)
void MixedBilinearForm::ConformingAssemble()
{
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
return;
}
Finalize();
const SparseMatrix *P2 = test_fes->GetConformingProlongation();
@@ -1470,17 +1586,91 @@ void MixedBilinearForm::EliminateTestDofs (const Array<int> &bdr_attr_is_ess)
}
}
void MixedBilinearForm::FormRectangularSystemMatrix(const Array<int>
&trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A)
{
if (ext)
{
ext->FormRectangularSystemOperator(trial_tdof_list, test_tdof_list, A);
return;
}
const SparseMatrix *test_P = test_fes->GetConformingProlongation();
const SparseMatrix *trial_P = trial_fes->GetConformingProlongation();
mat->Finalize();
if (test_P) // TODO: Must actually check for trial_P too
{
SparseMatrix *m = RAP(*test_P, *mat, *trial_P);
delete mat;
mat = m;
}
Array<int> ess_trial_tdof_marker, ess_test_tdof_marker;
FiniteElementSpace::ListToMarker(trial_tdof_list, trial_fes->GetTrueVSize(),
ess_trial_tdof_marker);
FiniteElementSpace::ListToMarker(test_tdof_list, test_fes->GetTrueVSize(),
ess_test_tdof_marker);
mat_e = new SparseMatrix(mat->Height(), mat->Width());
mat->EliminateCols(ess_trial_tdof_marker, *mat_e);
for (int i=0; i<test_tdof_list.Size(); ++i)
{
mat->EliminateRow(test_tdof_list[i]);
}
mat_e->Finalize();
A.Reset(mat, false);
}
void MixedBilinearForm::FormRectangularLinearSystem(const Array<int>
&trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B)
{
if (ext)
{
ext->FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b, A, X,
B);
return;
}
const Operator *Po = this->GetOutputProlongation();
const Operator *Ri = this->GetRestriction();
InitTVectors(Po, Ri, x, b, X, B);
if (!mat_e)
{
FormRectangularSystemMatrix(trial_tdof_list, test_tdof_list,
A); // Set A = mat_e
}
// Eliminate essential BCs with B -= Ab xb
mat_e->AddMult(X, B, -1.0);
B.SetSubVector(test_tdof_list, 0.0);
}
void MixedBilinearForm::Update()
{
delete mat;
mat = NULL;
delete mat_e;
mat_e = NULL;
height = test_fes->GetVSize();
width = trial_fes->GetVSize();
if (ext) { ext->Update(); }
}
MixedBilinearForm::~MixedBilinearForm()
{
if (mat) { delete mat; }
if (mat_e) { delete mat_e; }
if (!extern_bfs)
{
int i;
@@ -1489,6 +1679,7 @@ MixedBilinearForm::~MixedBilinearForm()
for (i = 0; i < tfbfi.Size(); i++) { delete tfbfi[i]; }
for (i = 0; i < btfbfi.Size(); i++) { delete btfbfi[i]; }
}
delete ext;
}
+120 -6
View File
@@ -58,7 +58,7 @@ protected:
/// FE space on which the form lives. Not owned.
FiniteElementSpace *fes;
/// The form assembly level (full, partial, etc.)
/// The assembly level of the form (full, partial, etc.)
AssemblyLevel assembly;
/// Element batch size used in the form action (1, 8, num_elems, etc.)
int batch;
@@ -227,7 +227,7 @@ public:
virtual const double &Elem(int i, int j) const;
/// Matrix vector multiplication.
virtual void Mult(const Vector &x, Vector &y) const { mat->Mult(x, y); }
virtual void Mult(const Vector &x, Vector &y) const;
void FullMult(const Vector &x, Vector &y) const
{ mat->Mult(x, y); mat_e->AddMult(x, y); }
@@ -319,12 +319,26 @@ public:
/// Assembles the form i.e. sums over all domain/bdr integrators.
void Assemble(int skip_zeros = 1);
/** @brief Assemble the diagonal of the bilinear form into diag
For adaptively refined meshes, this returns P^T d_e, where d_e is the
locally assembled diagonal on each element and P^T is the transpose of
the conforming prolongation. In general this is not the correct diagonal
for an AMR mesh. */
void AssembleDiagonal(Vector &diag) const;
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const
{ return fes->GetConformingProlongation(); }
/// Get the finite element space restriction matrix
virtual const Operator *GetRestriction() const
{ return fes->GetConformingRestriction(); }
/// Get the output finite element space prolongation matrix
virtual const Operator *GetOutputProlongation() const
{ return GetProlongation(); }
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const
{ return GetRestriction(); }
/** @brief Form the linear system A X = B, corresponding to this bilinear
form and the linear form @a b(.). */
@@ -525,6 +539,9 @@ public:
/// Sets diagonal policy used upon construction of the linear system
void SetDiagonalPolicy(DiagonalPolicy policy);
/// Indicate that integrators are not owned by the BilinearForm
void UseExternalIntegrators() { extern_bfs = 1; };
/// Destroys bilinear form.
virtual ~BilinearForm();
};
@@ -549,10 +566,17 @@ class MixedBilinearForm : public Matrix
{
protected:
SparseMatrix *mat; ///< Owned.
SparseMatrix *mat_e; ///< Owned.
FiniteElementSpace *trial_fes, ///< Not owned
*test_fes; ///< Not owned
/// The form assembly level (full, partial, etc.)
AssemblyLevel assembly;
/** Extension for supporting Full Assembly (FA), Element Assembly (EA),
Partial Assembly (PA), or Matrix Free assembly (MF). */
MixedBilinearFormExtension *ext;
/** @brief Indicates the BilinearFormIntegrator%s stored in #dbfi, #bbfi,
#tfbfi and #btfbfi are owned by another MixedBilinearForm. */
int extern_bfs;
@@ -607,16 +631,13 @@ public:
virtual const double &Elem(int i, int j) const;
virtual void Mult(const Vector & x, Vector & y) const;
virtual void AddMult(const Vector & x, Vector & y,
const double a = 1.0) const;
virtual void MultTranspose(const Vector & x, Vector & y) const;
virtual void AddMultTranspose(const Vector & x, Vector & y,
const double a = 1.0) const;
virtual void MultTranspose(const Vector & x, Vector & y) const
{ y = 0.0; AddMultTranspose (x, y); }
virtual MatrixInverse *Inverse() const;
virtual void Finalize(int skip_zeros = 1);
@@ -676,8 +697,28 @@ public:
void operator=(const double a) { *mat = a; }
/// Set the desired assembly level. The default is AssemblyLevel::FULL.
/** This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level);
void Assemble(int skip_zeros = 1);
/// Get the input finite element space prolongation matrix
virtual const Operator *GetProlongation() const
{ return trial_fes->GetProlongationMatrix(); }
/// Get the input finite element space restriction matrix
virtual const Operator *GetRestriction() const
{ return trial_fes->GetRestrictionMatrix(); }
/// Get the test finite element space prolongation matrix
virtual const Operator *GetOutputProlongation() const
{ return test_fes->GetProlongationMatrix(); }
/// Get the test finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const
{ return test_fes->GetRestrictionMatrix(); }
/** For partially conforming trial and/or test FE spaces, complete the
assembly process by performing A := P2^t A P1 where A is the internal
sparse matrix; P1 and P2 are the conforming prolongation matrices of the
@@ -739,8 +780,81 @@ public:
virtual void EliminateTestDofs(const Array<int> &bdr_attr_is_ess);
/** @brief Return in @a A a parallel (on truedofs) version of this operator.
This returns the same operator as FormRectangularLinearSystem(), but does
without the transformations of the right-hand side. */
void FormRectangularSystemMatrix(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A);
/** @brief Form the column-constrained linear system matrix A.
See FormRectangularSystemMatrix() for details.
Version of the method FormRectangularSystemMatrix() where the system matrix is
returned in the variable @a A, of type OpType, holding a *reference* to
the system matrix (created with the method OpType::MakeRef()). The
reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called.
Currently, this method can be used only with AssemblyLevel::FULL. */
template <typename OpType>
void FormRectangularSystemMatrix(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list, OpType &A)
{
OperatorHandle Ah;
FormRectangularSystemMatrix(trial_tdof_list, test_tdof_list, Ah);
OpType *A_ptr = Ah.Is<OpType>();
MFEM_VERIFY(A_ptr, "invalid OpType used");
A.MakeRef(*A_ptr);
}
/** @brief Form the linear system A X = B, corresponding to this mixed bilinear
form and the linear form @a b(.).
Return in @a A a *reference* to the system matrix that is column-constrained.
The reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called. */
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B);
/** @brief Form the linear system A X = B, corresponding to this bilinear
form and the linear form @a b(.).
Version of the method FormRectangularLinearSystem() where the system matrix is
returned in the variable @a A, of type OpType, holding a *reference* to
the system matrix (created with the method OpType::MakeRef()). The
reference will be invalidated when SetOperatorType(), Update(), or the
destructor is called.
Currently, this method can be used only with AssemblyLevel::FULL. */
template <typename OpType>
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OpType &A, Vector &X, Vector &B)
{
OperatorHandle Ah;
FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b, Ah, X, B);
OpType *A_ptr = Ah.Is<OpType>();
MFEM_VERIFY(A_ptr, "invalid OpType used");
A.MakeRef(*A_ptr);
}
void Update();
/// Return the trial FE space associated with the BilinearForm.
FiniteElementSpace *TrialFESpace() { return trial_fes; }
/// Read-only access to the associated trial FiniteElementSpace.
const FiniteElementSpace *TrialFESpace() const { return trial_fes; }
/// Return the test FE space associated with the BilinearForm.
FiniteElementSpace *TestFESpace() { return test_fes; }
/// Read-only access to the associated test FiniteElementSpace.
const FiniteElementSpace *TestFESpace() const { return test_fes; }
virtual ~MixedBilinearForm();
};
+240 -17
View File
@@ -14,6 +14,7 @@
#include "../general/forall.hpp"
#include "bilinearform.hpp"
#include "libceed/ceed.hpp"
namespace mfem
{
@@ -38,7 +39,8 @@ const Operator *BilinearFormExtension::GetRestriction() const
// Data and methods for partially-assembled bilinear forms
PABilinearFormExtension::PABilinearFormExtension(BilinearForm *form)
: BilinearFormExtension(form),
trialFes(a->FESpace()), testFes(a->FESpace())
trialFes(a->FESpace()),
testFes(a->FESpace())
{
elem_restrict_lex = trialFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
@@ -60,6 +62,31 @@ void PABilinearFormExtension::Assemble()
}
}
void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
{
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AssembleDiagonalPA(localY);
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AssembleDiagonalPA(y);
}
}
}
void PABilinearFormExtension::Update()
{
FiniteElementSpace *fes = a->FESpace();
@@ -78,12 +105,9 @@ void PABilinearFormExtension::Update()
void PABilinearFormExtension::FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A)
{
const Operator* trialP = trialFes->GetProlongationMatrix();
const Operator* testP = testFes->GetProlongationMatrix();
Operator *rap = this;
if (trialP) { rap = new RAPOperator(*testP, *this, *trialP); }
const bool own_A = (rap!=this);
A.Reset(new ConstrainedOperator(rap, ess_tdof_list, own_A));
Operator *oper;
Operator::FormSystemOperator(ess_tdof_list, oper);
A.Reset(oper); // A will own oper
}
void PABilinearFormExtension::FormLinearSystem(const Array<int> &ess_tdof_list,
@@ -102,7 +126,16 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
if (DeviceCanUseCeed() || !elem_restrict_lex)
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
else
{
elem_restrict_lex->Mult(x, localX);
localY = 0.0;
@@ -112,15 +145,6 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
}
void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
@@ -148,4 +172,203 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
}
}
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
: Operator(form->Height(), form->Width()), a(form)
{
// empty
}
const Operator *MixedBilinearFormExtension::GetProlongation() const
{
return a->GetProlongation();
}
const Operator *MixedBilinearFormExtension::GetRestriction() const
{
return a->GetRestriction();
}
const Operator *MixedBilinearFormExtension::GetOutputProlongation() const
{
return a->GetOutputProlongation();
}
const Operator *MixedBilinearFormExtension::GetOutputRestriction() const
{
return a->GetOutputRestriction();
}
// Data and methods for partially-assembled bilinear forms
PAMixedBilinearFormExtension::PAMixedBilinearFormExtension(
MixedBilinearForm *form)
: MixedBilinearFormExtension(form),
trialFes(form->TrialFESpace()),
testFes(form->TestFESpace()),
elem_restrict_trial(NULL),
elem_restrict_test(NULL)
{
Update();
}
void PAMixedBilinearFormExtension::Assemble()
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int integratorCount = integrators.Size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->AssemblePA(*trialFes, *testFes);
}
}
void PAMixedBilinearFormExtension::Update()
{
trialFes = a->TrialFESpace();
testFes = a->TestFESpace();
height = testFes->GetVSize();
width = trialFes->GetVSize();
elem_restrict_trial = trialFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
elem_restrict_test = testFes->GetElementRestriction(
ElementDofOrdering::LEXICOGRAPHIC);
if (elem_restrict_trial)
{
localTrial.UseDevice(true);
localTrial.SetSize(elem_restrict_trial->Height(), Device::GetMemoryType());
}
if (elem_restrict_test)
{
localTest.UseDevice(true); // ensure 'localY = 0.0' is done on device
localTest.SetSize(elem_restrict_test->Height(), Device::GetMemoryType());
}
}
void PAMixedBilinearFormExtension::FormRectangularSystemOperator(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A)
{
Operator * oper;
Operator::FormRectangularSystemOperator(trial_tdof_list, test_tdof_list, oper);
A.Reset(oper); // A will own oper
}
void PAMixedBilinearFormExtension::FormRectangularLinearSystem(
const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B)
{
Operator *oper;
Operator::FormRectangularLinearSystem(trial_tdof_list, test_tdof_list, x, b,
oper, X, B);
A.Reset(oper); // A will own oper
}
void PAMixedBilinearFormExtension::SetupMultInputs(const Operator
*elem_restrict_x,
const Vector &x,
Vector &localX,
const Operator *elem_restrict_y,
Vector &y,
Vector &localY,
const double c) const
{
// * G operation: localX = c*local(x)
if (elem_restrict_x)
{
elem_restrict_x->Mult(x, localX);
if (c != 1.0)
{
localX *= c;
}
}
else
{
if (c == 1.0)
{
localX.SyncAliasMemory(x);
}
else
{
localX.Set(c, x);
}
}
if (elem_restrict_y)
{
localY = 0.0;
}
else
{
y.UseDevice(true);
localY.SyncAliasMemory(y);
}
}
void PAMixedBilinearFormExtension::Mult(const Vector &x, Vector &y) const
{
y = 0.0;
AddMult(x, y);
}
void PAMixedBilinearFormExtension::AddMult(const Vector &x, Vector &y,
const double c) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
// * G operation
SetupMultInputs(elem_restrict_trial, x, localTrial,
elem_restrict_test, y, localTest, c);
// * B^TDB operation
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(localTrial, localTest);
}
// * G^T operation
if (elem_restrict_test)
{
tempY.SetSize(y.Size());
elem_restrict_test->MultTranspose(localTest, tempY);
y += tempY;
}
}
void PAMixedBilinearFormExtension::MultTranspose(const Vector &x,
Vector &y) const
{
y = 0.0;
AddMultTranspose(x, y);
}
void PAMixedBilinearFormExtension::AddMultTranspose(const Vector &x, Vector &y,
const double c) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
// * G operation
SetupMultInputs(elem_restrict_test, x, localTest,
elem_restrict_trial, y, localTrial, c);
// * B^TD^TB operation
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultTransposePA(localTest, localTrial);
}
// * G^T operation
if (elem_restrict_trial)
{
tempY.SetSize(y.Size());
elem_restrict_trial->MultTranspose(localTrial, tempY);
y += tempY;
}
}
} // namespace mfem
+99
View File
@@ -20,6 +20,7 @@ namespace mfem
{
class BilinearForm;
class MixedBilinearForm;
/** @brief Class extending the BilinearForm class to support the different
@@ -42,6 +43,10 @@ public:
virtual const Operator *GetRestriction() const;
virtual void Assemble() = 0;
virtual void AssembleDiagonal(Vector &diag) const
{
MFEM_ABORT("AssembleDiagonal not implemented for this assembly level!");
}
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A) = 0;
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
@@ -103,6 +108,7 @@ public:
PABilinearFormExtension(BilinearForm*);
void Assemble();
void AssembleDiagonal(Vector &diag) const;
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A);
void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
@@ -114,6 +120,7 @@ public:
void Update();
};
/// Data and methods for matrix-free bilinear forms
class MFBilinearFormExtension : public BilinearFormExtension
{
@@ -134,6 +141,98 @@ public:
~MFBilinearFormExtension() {}
};
/** @brief Class extending the MixedBilinearForm class to support the different
AssemblyLevel%s. */
class MixedBilinearFormExtension : public Operator
{
protected:
MixedBilinearForm *a; ///< Not owned
public:
MixedBilinearFormExtension(MixedBilinearForm *form);
virtual MemoryClass GetMemoryClass() const
{ return Device::GetMemoryClass(); }
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const;
/// Get the finite element space restriction matrix
virtual const Operator *GetRestriction() const;
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputProlongation() const;
/// Get the output finite element space restriction matrix
virtual const Operator *GetOutputRestriction() const;
virtual void Assemble() = 0;
virtual void FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A) = 0;
virtual void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B) = 0;
virtual void AddMult(const Vector &x, Vector &y, const double c=1.0) const = 0;
virtual void AddMultTranspose(const Vector &x, Vector &y,
const double c=1.0) const = 0;
virtual void Update() = 0;
};
/// Data and methods for partially-assembled mixed bilinear forms
class PAMixedBilinearFormExtension : public MixedBilinearFormExtension
{
protected:
const FiniteElementSpace *trialFes, *testFes; // Not owned
mutable Vector localTrial, localTest, tempY;
const Operator *elem_restrict_trial; // Not owned
const Operator *elem_restrict_test; // Not owned
private:
/// Helper function to set up inputs/outputs for Mult or MultTranspose
void SetupMultInputs(const Operator *elem_restrict_x,
const Vector &x, Vector &localX,
const Operator *elem_restrict_y,
Vector &y, Vector &localY, const double c) const;
public:
PAMixedBilinearFormExtension(MixedBilinearForm *form);
/// Partial assembly of all internal integrators
void Assemble();
/**
@brief Setup OperatorHandle A to contain constrained linear operator
OperatorHandle A contains matrix-free constrained operator formed for RAP
system where ess_tdof_list are in trial space and eliminated from
"columns" of A.
*/
void FormRectangularSystemOperator(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
OperatorHandle &A);
/**
Setup OperatorHandle A to contain constrained linear operator and
eliminate columns corresponding to essential dofs from system,
updating RHS B vector with the results.
*/
void FormRectangularLinearSystem(const Array<int> &trial_tdof_list,
const Array<int> &test_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B);
/// y = A*x
void Mult(const Vector &x, Vector &y) const;
/// y += c*A*x
void AddMult(const Vector &x, Vector &y, const double c=1.0) const;
/// y = A^T*x
void MultTranspose(const Vector &x, Vector &y) const;
/// y += c*A^T*x
void AddMultTranspose(const Vector &x, Vector &y, const double c=1.0) const;
/// Update internals for when a new MixedBilinearForm is given to this class
void Update();
};
}
#endif
+91 -7
View File
@@ -22,10 +22,23 @@ namespace mfem
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
{
mfem_error ("BilinearFormIntegrator::Assemble (...)\n"
mfem_error ("BilinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&,
const FiniteElementSpace&)
{
mfem_error ("BilinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AssembleDiagonalPA(Vector &)
{
MFEM_ABORT("BilinearFormIntegrator::AssembleDiagonalPA (...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
{
mfem_error ("BilinearFormIntegrator::MultAssembled (...)\n"
@@ -380,6 +393,72 @@ void MixedScalarVectorIntegrator::AssembleElementMatrix2(
}
void GradientIntegrator::AssembleElementMatrix2(
const FiniteElement &trial_fe, const FiniteElement &test_fe,
ElementTransformation &Trans, DenseMatrix &elmat)
{
int dim = test_fe.GetDim();
int trial_dof = trial_fe.GetDof();
int test_dof = test_fe.GetDof();
double c;
Vector d_col;
dshape.SetSize(trial_dof, dim);
gshape.SetSize(trial_dof, dim);
Jadj.SetSize(dim);
shape.SetSize(test_dof);
elmat.SetSize(dim * test_dof, trial_dof);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
Trans);
elmat = 0.0;
elmat_comp.SetSize(test_dof, trial_dof);
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
trial_fe.CalcDShape(ip, dshape);
test_fe.CalcShape(ip, shape);
Trans.SetIntPoint(&ip);
CalcAdjugate(Trans.Jacobian(), Jadj);
Mult(dshape, Jadj, gshape);
c = ip.weight;
if (Q)
{
c *= Q->Eval(Trans, ip);
}
shape *= c;
for (int d = 0; d < dim; ++d)
{
gshape.GetColumnReference(d, d_col);
MultVWt(shape, d_col, elmat_comp);
for (int jj = 0; jj < trial_dof; ++jj)
{
for (int ii = 0; ii < test_dof; ++ii)
{
elmat(d * test_dof + ii, jj) += elmat_comp(ii, jj);
}
}
}
}
}
const IntegrationRule &GradientIntegrator::GetRule(const FiniteElement
&trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder() + Trans.OrderJ();
return IntRules.Get(trial_fe.GetGeomType(), order);
}
void DiffusionIntegrator::AssembleElementMatrix
( const FiniteElement &el, ElementTransformation &Trans,
DenseMatrix &elmat )
@@ -1944,12 +2023,8 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
elmat.SetSize (test_dof, dim*trial_dof);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder();
ir = &IntRules.Get(trial_fe.GetGeomType(), order);
}
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
Trans);
elmat = 0.0;
@@ -1979,6 +2054,15 @@ void VectorDivergenceIntegrator::AssembleElementMatrix2(
}
}
const IntegrationRule &VectorDivergenceIntegrator::GetRule(
const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans)
{
int order = Trans.OrderGrad(&trial_fe) + test_fe.GetOrder() + Trans.OrderJ();
return IntRules.Get(trial_fe.GetGeomType(), order);
}
void DivDivIntegrator::AssembleElementMatrix(
const FiniteElement &el,
+185 -10
View File
@@ -15,6 +15,7 @@
#include "../config/config.hpp"
#include "nonlininteg.hpp"
#include "fespace.hpp"
#include "libceed/ceed.hpp"
namespace mfem
{
@@ -39,10 +40,18 @@ public:
// make sense for the action of the nonlinear operator (but they all make
// sense for its Jacobian).
using NonlinearFormIntegrator::AssemblePA;
/// Method defining partial assembly.
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA() and AddMultTransposePA(). */
virtual void AssemblePA(const FiniteElementSpace &fes);
/** Used with BilinearFormIntegrators that have different spaces. */
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
/// Assemble diagonal and add it to Vector @a diag.
virtual void AssembleDiagonalPA(Vector &diag);
/// Method for partially assembled action.
/** Perform the action of integrator on the input @a x and add the result to
@@ -1658,6 +1667,55 @@ protected:
}
};
/** Class for integrating the bilinear form a(u,v) := (Q grad u, v) where Q is a
scalar coefficient, and v is a vector with components v_i in the same space
as u. */
class GradientIntegrator : public BilinearFormIntegrator
{
protected:
Coefficient *Q;
private:
Vector shape;
DenseMatrix dshape;
DenseMatrix gshape;
DenseMatrix Jadj;
DenseMatrix elmat_comp;
// PA extension
Vector pa_data;
const DofToQuad *trial_maps, *test_maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
int trial_dofs1D, test_dofs1D, quad1D;
public:
GradientIntegrator() :
Q{NULL}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
GradientIntegrator(Coefficient *_q) :
Q{_q}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
GradientIntegrator(Coefficient &q) :
Q{&q}, trial_maps{NULL}, test_maps{NULL}, geom{NULL}
{ }
virtual void AssembleElementMatrix2(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
};
/** Class for integrating the bilinear form a(u,v) := (Q grad u, grad v) where Q
can be a scalar or a matrix coefficient. */
class DiffusionIntegrator: public BilinearFormIntegrator
@@ -1674,22 +1732,60 @@ private:
#endif
// PA extension
const FiniteElementSpace *fespace;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D;
Vector pa_data;
#ifdef MFEM_USE_CEED
// CEED extension
CeedData* ceedDataPtr;
#endif
public:
/// Construct a diffusion integrator with coefficient Q = 1
DiffusionIntegrator() { Q = NULL; MQ = NULL; maps = NULL; geom = NULL; }
DiffusionIntegrator()
{
Q = NULL;
MQ = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
/// Construct a diffusion integrator with a scalar coefficient q
DiffusionIntegrator(Coefficient &q)
: Q(&q) { MQ = NULL; maps = NULL; geom = NULL; }
: Q(&q)
{
MQ = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
/// Construct a diffusion integrator with a matrix coefficient q
DiffusionIntegrator(MatrixCoefficient &q)
: MQ(&q) { Q = NULL; maps = NULL; geom = NULL; }
: MQ(&q)
{
Q = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
virtual ~DiffusionIntegrator()
{
#ifdef MFEM_USE_CEED
delete ceedDataPtr;
#endif
}
/** Given a particular Finite Element
computes the element stiffness matrix elmat. */
@@ -1717,12 +1813,18 @@ public:
ElementTransformation &Trans,
Vector &flux, Vector *d_energy = NULL);
virtual void AssemblePA(const FiniteElementSpace&);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleDiagonalPA(Vector &diag);
virtual void AddMultPA(const Vector&, Vector&) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
@@ -1734,19 +1836,46 @@ protected:
#endif
Coefficient *Q;
// PA extension
const FiniteElementSpace *fespace;
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
#ifdef MFEM_USE_CEED
// CEED extension
CeedData* ceedDataPtr;
#endif
public:
MassIntegrator(const IntegrationRule *ir = NULL)
: BilinearFormIntegrator(ir) { Q = NULL; maps = NULL; geom = NULL; }
: BilinearFormIntegrator(ir)
{
Q = NULL;
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
/// Construct a mass integrator with coefficient q
MassIntegrator(Coefficient &q, const IntegrationRule *ir = NULL)
: BilinearFormIntegrator(ir), Q(&q) { maps = NULL; geom = NULL; }
: BilinearFormIntegrator(ir), Q(&q)
{
maps = NULL;
geom = NULL;
#ifdef MFEM_USE_CEED
ceedDataPtr = NULL;
#endif
}
virtual ~MassIntegrator()
{
#ifdef MFEM_USE_CEED
delete ceedDataPtr;
#endif
}
/** Given a particular Finite Element
computes the element mass matrix elmat. */
virtual void AssembleElementMatrix(const FiniteElement &el,
@@ -1757,13 +1886,19 @@ public:
ElementTransformation &Trans,
DenseMatrix &elmat);
virtual void AssemblePA(const FiniteElementSpace&);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleDiagonalPA(Vector &diag);
virtual void AddMultPA(const Vector&, Vector&) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
class BoundaryMassIntegrator : public MassIntegrator
@@ -1835,6 +1970,11 @@ protected:
Coefficient *Q;
VectorCoefficient *VQ;
MatrixCoefficient *MQ;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
public:
/// Construct an integrator with coefficient 1.0
@@ -1866,6 +2006,9 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
@@ -2089,16 +2232,39 @@ private:
DenseMatrix dshape;
DenseMatrix gshape;
DenseMatrix Jadj;
// PA extension
Vector pa_data;
const DofToQuad *trial_maps, *test_maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
int trial_dofs1D, test_dofs1D, quad1D;
public:
VectorDivergenceIntegrator() { Q = NULL; }
VectorDivergenceIntegrator(Coefficient *_q) { Q = _q; }
VectorDivergenceIntegrator(Coefficient &q) { Q = &q; }
VectorDivergenceIntegrator() :
Q(NULL), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
VectorDivergenceIntegrator(Coefficient *_q) :
Q(_q), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
VectorDivergenceIntegrator(Coefficient &q) :
Q(&q), trial_maps(NULL), test_maps(NULL), geom(NULL)
{ }
virtual void AssembleElementMatrix2(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans);
};
/// (Q div u, div v) for RT elements
@@ -2132,6 +2298,12 @@ class VectorDiffusionIntegrator : public BilinearFormIntegrator
protected:
Coefficient *Q;
// PA extension
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D;
Vector pa_data;
private:
DenseMatrix Jinv;
DenseMatrix dshape;
@@ -2148,6 +2320,9 @@ public:
virtual void AssembleElementVector(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun, Vector &elvect);
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
/** Integrator for the linear elasticity form:
+604 -136
View File
@@ -12,6 +12,7 @@
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
#include "libceed/diffusion.hpp"
using namespace std;
@@ -90,7 +91,6 @@ static void PADiffusionSetup2D(const int Q1D,
const int NQ = Q1D*Q1D;
const bool const_c = c.Size() == 1;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 3, NE);
@@ -199,12 +199,25 @@ static void PADiffusionSetup(const int dim,
}
}
void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
const bool force)
{
// Assumes tensor-product elements
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
ceedDataPtr = ptr;
InitCeedCoeff(Q, ptr);
return CeedPADiffusionAssemble(fes, *ir, *ptr);
}
#endif
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
@@ -229,7 +242,7 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
else
{
coeff.SetSize(nq * ne);
auto C = Reshape(coeff.Write(), nq, ne);
auto C = Reshape(coeff.HostWrite(), nq, ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
@@ -243,6 +256,435 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
pa_data);
}
void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
SetupPA(fes);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionDiagonal2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
// note the different shape for D, this is a (symmetric) matrix so we only
// store necessary entries
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
// gradphi \cdot Q \gradphi has four terms
double QD0[MQ1][MD1];
double QD1[MQ1][MD1];
double QD2[MQ1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dy = 0; dy < D1D; ++dy)
{
QD0[qx][dy] = 0.0;
QD1[qx][dy] = 0.0;
QD2[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const double D0 = D(q,0,e);
const double D1 = D(q,1,e);
const double D2 = D(q,2,e);
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
}
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
Y(dx,dy,e) += G(qx, dx) * G(qx, dx) * QD0[qx][dy];
Y(dx,dy,e) += G(qx, dx) * B(qx, dx) * QD1[qx][dy];
Y(dx,dy,e) += B(qx, dx) * G(qx, dx) * QD1[qx][dy];
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD2[qx][dy];
}
}
}
});
}
// Shared memory PA Diffusion Diagonal 2D kernel
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPADiffusionDiagonal2D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double BG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (BG+0);
double (*G)[MD1] = (double (*)[MD1]) (BG+1);
MFEM_SHARED double QD[4][NBZ][MD1][MQ1];
double (*QD0)[MD1] = (double (*)[MD1])(QD[0] + tidz);
double (*QD1)[MD1] = (double (*)[MD1])(QD[1] + tidz);
double (*QD2)[MD1] = (double (*)[MD1])(QD[3] + tidz);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD0[qx][dy] = 0.0;
QD1[qx][dy] = 0.0;
QD2[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const double D0 = D(q,0,e);
const double D1 = D(q,1,e);
const double D2 = D(q,2,e);
const double By = B[qy][dy];
const double Gy = G[qy][dy];
const double BB = By * By;
const double BG = By * Gy;
const double GG = Gy * Gy;
QD0[qx][dy] += BB * D0;
QD1[qx][dy] += BG * D1;
QD2[qx][dy] += GG * D2;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B[qx][dx];
const double Gx = G[qx][dx];
const double BB = Bx * Bx;
const double BG = Bx * Gx;
const double GG = Gx * Gx;
Y(dx,dy,e) += GG * QD0[qx][dy];
Y(dx,dy,e) += BG * QD1[qx][dy];
Y(dx,dy,e) += BG * QD1[qx][dy];
Y(dx,dy,e) += BB * QD2[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionDiagonal3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QQD[MQ1][MQ1][MD1];
double QDD[MQ1][MD1][MD1];
for (int i = 0; i < DIM; ++i)
{
for (int j = 0; j < DIM; ++j)
{
// first tensor contraction, along z direction
for (int qx = 0; qx < Q1D; ++qx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int dz = 0; dz < D1D; ++dz)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
const double O = Q(q,k,e);
const double Bz = B(qz,dz);
const double Gz = G(qz,dz);
const double L = i==2 ? Gz : Bz;
const double R = j==2 ? Gz : Bz;
QQD[qx][qy][dz] += L * O * R;
}
}
}
}
// second tensor contraction, along y direction
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy,dy);
const double Gy = G(qy,dy);
const double L = i==1 ? Gy : By;
const double R = j==1 ? Gy : By;
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
}
}
}
}
// third tensor contraction, along x direction
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx,dx);
const double Gx = G(qx,dx);
const double L = i==0 ? Gx : Bx;
const double R = j==0 ? Gx : Bx;
Y(dx, dy, dz, e) += L * QDD[qx][dy][dz] * R;
}
}
}
}
}
}
});
}
// Shared memory PA Diffusion Diagonal 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPADiffusionDiagonal3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double BG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (BG+0);
double (*G)[MD1] = (double (*)[MD1]) (BG+1);
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
MFEM_SHARED double QDD[MQ1][MD1][MD1];
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int i = 0; i < DIM; ++i)
{
for (int j = 0; j < DIM; ++j)
{
// first tensor contraction, along z direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
const double O = D(q,k,e);
const double Bz = B[qz][dz];
const double Gz = G[qz][dz];
const double L = i==2 ? Gz : Bz;
const double R = j==2 ? Gz : Bz;
QQD[qx][qy][dz] += L * O * R;
}
}
}
}
MFEM_SYNC_THREAD;
// second tensor contraction, along y direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B[qy][dy];
const double Gy = G[qy][dy];
const double L = i==1 ? Gy : By;
const double R = j==1 ? Gy : By;
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
}
}
}
}
MFEM_SYNC_THREAD;
// third tensor contraction, along x direction
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B[qx][dx];
const double Gx = G[qx][dx];
const double L = i==0 ? Gx : Bx;
const double R = j==0 ? Gx : Bx;
Y(dx, dy, dz, e) += L * QDD[qx][dy][dz] * R;
}
}
}
}
}
}
});
}
static void PADiffusionAssembleDiagonal(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Vector &D,
Vector &Y)
{
if (dim == 2)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPADiffusionDiagonal2D<2,2,8>(NE,B,G,D,Y);
case 0x33: return SmemPADiffusionDiagonal2D<3,3,8>(NE,B,G,D,Y);
case 0x44: return SmemPADiffusionDiagonal2D<4,4,4>(NE,B,G,D,Y);
case 0x55: return SmemPADiffusionDiagonal2D<5,5,4>(NE,B,G,D,Y);
case 0x66: return SmemPADiffusionDiagonal2D<6,6,2>(NE,B,G,D,Y);
case 0x77: return SmemPADiffusionDiagonal2D<7,7,2>(NE,B,G,D,Y);
case 0x88: return SmemPADiffusionDiagonal2D<8,8,1>(NE,B,G,D,Y);
case 0x99: return SmemPADiffusionDiagonal2D<9,9,1>(NE,B,G,D,Y);
default: return PADiffusionDiagonal2D(NE,B,G,D,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,B,G,D,Y);
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,B,G,D,Y);
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,B,G,D,Y);
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,B,G,D,Y);
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,B,G,D,Y);
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,B,G,D,Y);
case 0x89: return SmemPADiffusionDiagonal3D<8,9>(NE,B,G,D,Y);
case 0x9A: return SmemPADiffusionDiagonal3D<9,10>(NE,B,G,D,Y);
default: return PADiffusionDiagonal3D(NE,B,G,D,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
}
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
#ifdef MFEM_USE_OCCA
// OCCA PA Diffusion Apply 2D kernel
static void OccaPADiffusionApply2D(const int D1D,
@@ -252,9 +694,9 @@ static void OccaPADiffusionApply2D(const int D1D,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
occa::properties props;
props["defines/D1D"] = D1D;
@@ -263,9 +705,9 @@ static void OccaPADiffusionApply2D(const int D1D,
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -277,7 +719,7 @@ static void OccaPADiffusionApply2D(const int D1D,
"DiffusionApply2D_CPU", props);
OccaDiffApply2D_cpu.emplace(id, DiffusionApply2D_CPU);
}
OccaDiffApply2D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
OccaDiffApply2D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
}
else
{
@@ -289,7 +731,7 @@ static void OccaPADiffusionApply2D(const int D1D,
"DiffusionApply2D_GPU", props);
OccaDiffApply2D_gpu.emplace(id, DiffusionApply2D_GPU);
}
OccaDiffApply2D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
OccaDiffApply2D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
}
}
@@ -301,9 +743,9 @@ static void OccaPADiffusionApply3D(const int D1D,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
occa::properties props;
props["defines/D1D"] = D1D;
@@ -312,9 +754,9 @@ static void OccaPADiffusionApply3D(const int D1D,
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -326,7 +768,7 @@ static void OccaPADiffusionApply3D(const int D1D,
"DiffusionApply3D_CPU", props);
OccaDiffApply3D_cpu.emplace(id, DiffusionApply3D_CPU);
}
OccaDiffApply3D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
OccaDiffApply3D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
}
else
{
@@ -338,35 +780,35 @@ static void OccaPADiffusionApply3D(const int D1D,
"DiffusionApply3D_GPU", props);
OccaDiffApply3D_gpu.emplace(id, DiffusionApply3D_GPU);
}
OccaDiffApply3D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_op, o_x, o_y);
OccaDiffApply3D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
}
}
#endif // MFEM_USE_OCCA
// PA Diffusion Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0> static
void PADiffusionApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionApply2D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, NE);
auto B = Reshape(b_.Read(), Q1D, D1D);
auto G = Reshape(g_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto Gt = Reshape(gt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto X = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -394,7 +836,7 @@ void PADiffusionApply2D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,e);
const double s = X(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
@@ -419,9 +861,9 @@ void PADiffusionApply2D(const int NE,
{
const int q = qx + qy * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O22 = D(q,2,e);
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
@@ -456,7 +898,7 @@ void PADiffusionApply2D(const int NE,
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
Y(dx,dy,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
}
}
}
@@ -464,17 +906,15 @@ void PADiffusionApply2D(const int NE,
}
// Shared memory PA Diffusion Apply 2D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0,
const int T_NBZ = 0>
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPADiffusionApply2D(const int NE,
const Array<double> &_b,
const Array<double> &_g,
const Array<double> &_bt,
const Array<double> &_gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
@@ -485,11 +925,11 @@ static void SmemPADiffusionApply2D(const int NE,
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(_b.Read(), Q1D, D1D);
auto g = Reshape(_g.Read(), Q1D, D1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, NE);
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -520,12 +960,12 @@ static void SmemPADiffusionApply2D(const int NE,
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
B[q][dy] = b(q,dy);
G[q][dy] = g(q,dy);
}
}
}
@@ -568,9 +1008,9 @@ static void SmemPADiffusionApply2D(const int NE,
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = (qx + ((qy) * Q1D));
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O22 = D(q,2,e);
const double gX = QQ0[qy][qx];
const double gY = QQ1[qy][qx];
QQ0[qy][qx] = (O11 * gX) + (O12 * gY);
@@ -580,12 +1020,12 @@ static void SmemPADiffusionApply2D(const int NE,
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = b(q,d);
Gt[d][q] = g(q,d);
Bt[dy][q] = b(q,dy);
Gt[dy][q] = g(q,dy);
}
}
}
@@ -617,25 +1057,23 @@ static void SmemPADiffusionApply2D(const int NE,
u += DQ0[qy][dx] * Bt[dy][qy];
v += DQ1[qy][dx] * Gt[dy][qy];
}
y(dx,dy,e) += (u + v);
Y(dx,dy,e) += (u + v);
}
}
});
}
// PA Diffusion Apply 3D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0> static
void PADiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int d1d = 0, int q1d = 0)
template<int T_D1D = 0, int T_Q1D = 0>
static void PADiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &d_,
const Vector &x_,
Vector &y_,
int d1d = 0, int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -645,9 +1083,9 @@ void PADiffusionApply3D(const int NE,
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, NE);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto X = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -689,7 +1127,7 @@ void PADiffusionApply3D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,e);
const double s = X(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
@@ -733,12 +1171,12 @@ void PADiffusionApply3D(const int NE,
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double O11 = D(q,0,e);
const double O12 = D(q,1,e);
const double O13 = D(q,2,e);
const double O22 = D(q,3,e);
const double O23 = D(q,4,e);
const double O33 = D(q,5,e);
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
@@ -803,7 +1241,7 @@ void PADiffusionApply3D(const int NE,
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,e) +=
Y(dx,dy,dz,e) +=
((gradXY[dy][dx][0] * wz) +
(gradXY[dy][dx][1] * wz) +
(gradXY[dy][dx][2] * wDz));
@@ -815,16 +1253,15 @@ void PADiffusionApply3D(const int NE,
}
// Shared memory PA Diffusion Apply 3D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0>
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPADiffusionApply3D(const int NE,
const Array<double> &_b,
const Array<double> &_g,
const Array<double> &_bt,
const Array<double> &_gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Array<double> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
@@ -834,11 +1271,11 @@ static void SmemPADiffusionApply3D(const int NE,
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(_b.Read(), Q1D, D1D);
auto g = Reshape(_g.Read(), Q1D, D1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, NE);
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -962,12 +1399,12 @@ static void SmemPADiffusionApply3D(const int NE,
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double O11 = d(q,0,e);
const double O12 = d(q,1,e);
const double O13 = d(q,2,e);
const double O22 = d(q,3,e);
const double O23 = d(q,4,e);
const double O33 = d(q,5,e);
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
@@ -1064,21 +1501,21 @@ static void PADiffusionApply(const int dim,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
if (dim == 2)
{
OccaPADiffusionApply2D(D1D, Q1D, NE, B, G, Bt, Gt, op, x, y);
OccaPADiffusionApply2D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
return;
}
if (dim == 3)
{
OccaPADiffusionApply3D(D1D, Q1D, NE, B, G, Bt, Gt, op, x, y);
OccaPADiffusionApply3D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
return;
}
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
@@ -1088,29 +1525,29 @@ static void PADiffusionApply(const int dim,
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,Bt,Gt,op,x,y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,Bt,Gt,op,x,y);
case 0x44: return SmemPADiffusionApply2D<4,4,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x55: return SmemPADiffusionApply2D<5,5,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x66: return SmemPADiffusionApply2D<6,6,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x77: return SmemPADiffusionApply2D<7,7,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x88: return SmemPADiffusionApply2D<8,8,2>(NE,B,G,Bt,Gt,op,x,y);
case 0x99: return SmemPADiffusionApply2D<9,9,2>(NE,B,G,Bt,Gt,op,x,y);
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,Bt,Gt,D,X,Y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,Bt,Gt,D,X,Y);
case 0x44: return SmemPADiffusionApply2D<4,4,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x55: return SmemPADiffusionApply2D<5,5,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x66: return SmemPADiffusionApply2D<6,6,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x77: return SmemPADiffusionApply2D<7,7,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x88: return SmemPADiffusionApply2D<8,8,2>(NE,B,G,Bt,Gt,D,X,Y);
case 0x99: return SmemPADiffusionApply2D<9,9,2>(NE,B,G,Bt,Gt,D,X,Y);
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,Bt,Gt,op,x,y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,Bt,Gt,op,x,y);
case 0x56: return SmemPADiffusionApply3D<5,6>(NE,B,G,Bt,Gt,op,x,y);
case 0x67: return SmemPADiffusionApply3D<6,7>(NE,B,G,Bt,Gt,op,x,y);
case 0x78: return SmemPADiffusionApply3D<7,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x89: return SmemPADiffusionApply3D<8,9>(NE,B,G,Bt,Gt,op,x,y);
default: return PADiffusionApply3D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,Bt,Gt,D,X,Y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,Bt,Gt,D,X,Y);
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,Bt,Gt,D,X,Y);
case 0x56: return SmemPADiffusionApply3D<5,6>(NE,B,G,Bt,Gt,D,X,Y);
case 0x67: return SmemPADiffusionApply3D<6,7>(NE,B,G,Bt,Gt,D,X,Y);
case 0x78: return SmemPADiffusionApply3D<7,8>(NE,B,G,Bt,Gt,D,X,Y);
case 0x89: return SmemPADiffusionApply3D<8,9>(NE,B,G,Bt,Gt,D,X,Y);
default: return PADiffusionApply3D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
@@ -1119,9 +1556,40 @@ static void PADiffusionApply(const int dim,
// PA Diffusion Apply kernel
void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PADiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
{
PADiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
}
} // namespace mfem
File diff suppressed because it is too large Load Diff
+823
View File
@@ -0,0 +1,823 @@
// Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Gradient Integrator
/* Description of the *SetupND functions
Inputs are as follows
\b Q1D number of quadrature points in one dimension.
\b w quadrature weights.
\b j element Jacobians.
\b COEFF coefficient at quadrature points.
The function is used precompute data needed at quadrature points during
the action. */
/* Description of the *ApplyND functions
The template parameters are
\b T_D1D number of degrees of freedom in one dimension,
\b T_Q1D number of quadrature points in one dimension,
and are necessary to allow for compiler optimizations inside the kernel.
Inputs are as follows
\b NE number of elements.
\b B matrix of basis functions.
\b G matrix of derivatives of the basis functions.
\b Bt transpose of matrix of basis functions.
\b Gt transpose matrix of derivatives of the basis functions.
\b op data used during action of the element matrix in the tensor
product application.
\b x input vector of degrees of freedom on the element.
\b y output vector of degrees of freedom on the element.
The function computes the kernel for one dimension that is suitable for
tensor product action to form ND operators.
Most of the ND inputs are reshaped as NQ*(ND*ND)*NE data structure, i.e
to allow indexing such as op(qpt,i,j,el).
The output data structure is dependent on the kernel and layout of the
dimension ND and element number, but in general resembles the action of the
element matrix in the tensor product application. */
/* Description of the Smem*ApplyND functions
The shared memory (Smem) versions of the kernels differ from the regular
versions in the following properties.
\b MFEM_FORALL is using only one level of parallelism.
\b MFEM_FORALL_ND uses an additional level of parallelism
\b MFEM_FOREACH_THREAD
These macros allow automatic mapping of manually defined blocks to
underlying hardware threads. These threads can share memory by using
the \b MFEM_SHARED keyword for local arrays. */
// PA Gradient Assemble 2D kernel
static void PAGradientSetup2D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 2, 2, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J12 = J(q,0,1,e);
const double J21 = J(q,1,0,e);
const double J22 = J(q,1,1,e);
// Store wq * Q * adj(J)
y(q,0,0,e) = W[q] * COEFF * J22; // 1,1
y(q,0,1,e) = W[q] * COEFF * -J12; // 1,2
y(q,1,0,e) = W[q] * COEFF * -J21; // 2,1
y(q,1,1,e) = W[q] * COEFF * J11; // 2,2
}
});
}
// PA Gradient Assemble 3D kernel
static void PAGradientSetup3D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 3, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double cw = W[q] * COEFF;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// Store wq * Q * adj(J)
y(q,0,0,e) = cw * A11; // 1,1
y(q,0,1,e) = cw * A12; // 1,2
y(q,0,2,e) = cw * A13; // 1,3
y(q,1,0,e) = cw * A21; // 2,1
y(q,1,1,e) = cw * A22; // 2,2
y(q,1,2,e) = cw * A23; // 2,3
y(q,2,0,e) = cw * A31; // 3,1
y(q,2,1,e) = cw * A32; // 3,2
y(q,2,2,e) = cw * A33; // 3,3
}
});
}
static void PAGradientSetup(const int dim,
const int TR_D1D,
const int TE_D1D,
const int Q1D,
const int NE,
const Array<double> &W,
const Vector &J,
const double COEFF,
Vector &op)
{
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PAGradientSetup"); }
if (dim == 2)
{
PAGradientSetup2D(Q1D, NE, W, J, COEFF, op);
}
if (dim == 3)
{
PAGradientSetup3D(Q1D, NE, W, J, COEFF, op);
}
}
void GradientIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes)
{
// Assumes tensor-product elements ordered by nodes
MFEM_ASSERT(trial_fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
Mesh *mesh = trial_fes.GetMesh();
const FiniteElement &trial_fe = *trial_fes.GetFE(0);
const FiniteElement &test_fe = *test_fes.GetFE(0);
ElementTransformation *trans = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(trial_fe, test_fe,
*trans);
const int dims = trial_fe.GetDim();
const int dimsToStore = dims * dims;
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = trial_fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
trial_maps = &trial_fe.GetDofToQuad(*ir, DofToQuad::TENSOR);
trial_dofs1D = trial_maps->ndof;
quad1D = trial_maps->nqpt;
test_maps = &test_fe.GetDofToQuad(*ir, DofToQuad::TENSOR);
test_dofs1D = test_maps->ndof;
MFEM_ASSERT(quad1D == test_maps->nqpt,
"PA requires test and trial space to have same number of quadrature points!");
pa_data.SetSize(nq * dimsToStore * ne, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
coeff = cQ->constant;
}
PAGradientSetup(dim, trial_dofs1D, test_dofs1D, quad1D,
ne, ir->GetWeights(), geom->J, coeff, pa_data);
}
// PA Gradient Apply 2D kernel
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
static void PAGradientApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, TR_D1D);
auto G = Reshape(g.Read(), Q1D, TR_D1D);
auto Bt = Reshape(bt.Read(), TE_D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 2,2, NE);
auto x = Reshape(_x.Read(), TR_D1D, TR_D1D, NE);
auto y = Reshape(_y.ReadWrite(), TE_D1D, TE_D1D, 2, NE);
MFEM_FORALL(e, NE,
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = 2;
// the following variables are evaluated at compile time
constexpr int max_TE_D1D = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double grad[max_Q1D][max_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] = 0.0;
grad[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < TR_D1D; ++dy)
{
double gradX[max_Q1D][VDIM];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < TR_D1D; ++dx)
{
const double s = x(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * G(qx,dx);
gradX[qx][1] += s * B(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] += gradX[qx][0] * wy;
grad[qy][qx][1] += gradX[qx][1] * wDy;
}
}
}
// We've now calculated grad(p) = [Dxy, xDy] in plane
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
grad[qy][qx][0] = gradX*op(q,0,0,e) + gradY*op(q,1,0,e);
grad[qy][qx][1] = gradX*op(q,0,1,e) + gradY*op(q,1,1,e);
}
}
// We've now calculated grad = grad p * op
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_TE_D1D][VDIM];
for (int dx = 0; dx < TE_D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
opX[dx][0] += Bt(dx,qx)*grad[qy][qx][0];
opX[dx][1] += Bt(dx,qx)*grad[qy][qx][1];
}
}
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
y(dx,dy,0,e) += Bt(dy,qy)*opX[dx][0];
y(dx,dy,1,e) += Bt(dy,qy)*opX[dx][1];
}
}
}
// We've now calculated y = u * grad
});
}
// PA Gradient Apply 2D kernel transpose
template<int T_TR_D1D = 0, int T_TE_D1D = 0, int T_Q1D = 0>
static void PAGradientApplyTranspose2D(const int NE,
const Array<double> &bt,
const Array<double> &gt,
const Array<double> &b,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
// TODO
MFEM_ASSERT(false, "GradientPAApplyTranspose 3D not implemented.");
}
// PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void PAGradientApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int tr_d1d = 0,
int te_d1d = 0,
int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, TR_D1D);
auto G = Reshape(g.Read(), Q1D, TR_D1D);
auto Bt = Reshape(bt.Read(), TE_D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 3,3, NE);
auto x = Reshape(_x.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
auto y = Reshape(_y.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
MFEM_FORALL(e, NE,
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = 3;
// the following variables are evaluated at compile time
constexpr int max_TE_D1D = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double grad[max_Q1D][max_Q1D][max_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] = 0.0;
grad[qz][qy][qx][1] = 0.0;
grad[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < TR_D1D; ++dz)
{
double gradXY[max_Q1D][max_Q1D][3];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
gradXY[qy][qx][0] = 0.0;
gradXY[qy][qx][1] = 0.0;
gradXY[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < TR_D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < TR_D1D; ++dx)
{
const double s = x(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
const double wx = gradX[qx][0];
const double wDx = gradX[qx][1];
gradXY[qy][qx][0] += wDx * wy;
gradXY[qy][qx][1] += wx * wDy;
gradXY[qy][qx][2] += wx * wy;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
const double wDz = G(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
}
}
}
}
// We've now calculated grad(p) = [Dxyz, xDyz, xyDz] in plane
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
grad[qz][qy][qx][0] = gradX*op(q,0,0,e) + gradY*op(q,1,0,e) + gradZ*op(q,2,0,e);
grad[qz][qy][qx][1] = gradX*op(q,0,1,e) + gradY*op(q,1,1,e) + gradZ*op(q,2,1,e);
grad[qz][qy][qx][2] = gradX*op(q,0,2,e) + gradY*op(q,1,2,e) + gradZ*op(q,2,2,e);
}
}
}
// We've now calculated grad = grad p * op
for (int qz = 0; qz < Q1D; ++qz)
{
double opXY[max_TE_D1D][max_TE_D1D][VDIM];
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
opXY[dy][dx][0] = 0.0;
opXY[dy][dx][1] = 0.0;
opXY[dy][dx][2] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_TE_D1D][VDIM];
for (int dx = 0; dx < TE_D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
opX[dx][2] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
opX[dx][0] += Bt(dx,qx)*grad[qz][qy][qx][0];
opX[dx][1] += Bt(dx,qx)*grad[qz][qy][qx][1];
opX[dx][2] += Bt(dx,qx)*grad[qz][qy][qx][2];
}
}
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
opXY[dy][dx][0] += Bt(dy,qy)*opX[dx][0];
opXY[dy][dx][1] += Bt(dy,qy)*opX[dx][1];
opXY[dy][dx][2] += Bt(dy,qy)*opX[dx][2];
}
}
}
for (int dz = 0; dz < TE_D1D; ++dz)
{
for (int dy = 0; dy < TE_D1D; ++dy)
{
for (int dx = 0; dx < TE_D1D; ++dx)
{
y(dx,dy,dz,0,e) += Bt(dz,qz)*opXY[dy][dx][0];
y(dx,dy,dz,1,e) += Bt(dz,qz)*opXY[dy][dx][1];
y(dx,dy,dz,2,e) += Bt(dz,qz)*opXY[dy][dx][2];
}
}
}
}
// We've now calculated y = u * grad
});
}
// PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void PAGradientApplyTranspose3D(const int NE,
const Array<double> &bt,
const Array<double> &gt,
const Array<double> &b,
const Vector &_op,
const Vector &_x,
Vector &_y,
int tr_d1d = 0,
int te_d1d = 0,
int q1d = 0)
{
MFEM_ASSERT(false, "Gradient PA Apply Transpose 3D not implemented.");
}
// Shared memory PA Gradient Apply 3D kernel
template<const int T_TR_D1D = 0, const int T_TE_D1D = 0, const int T_Q1D = 0>
static void SmemPAGradientApply3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Array<double> &bt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int tr_d1d = 0,
const int te_d1d = 0,
const int q1d = 0)
{
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(TR_D1D <= MAX_D1D, "");
MFEM_VERIFY(TE_D1D <= MAX_D1D, "");
MFEM_VERIFY(TR_D1D <= Q1D, "");
MFEM_VERIFY(TE_D1D <= Q1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto b = Reshape(b_.Read(), Q1D, TR_D1D);
auto g = Reshape(g_.Read(), Q1D, TR_D1D);
auto bt = Reshape(bt_.Read(), TE_D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, 3, 3, NE);
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
MFEM_FORALL_3D(e, NE, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1DR = T_TR_D1D ? T_TR_D1D : tr_d1d;
const int D1DE = T_TE_D1D ? T_TE_D1D : te_d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1R = T_TR_D1D ? T_TR_D1D : MAX_D1D;
constexpr int MD1E = T_TE_D1D ? T_TE_D1D : MAX_D1D;
constexpr int MD1 = MD1E > MD1R ? MD1E : MD1R;
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
MFEM_SHARED double sBG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
double (*DDQ0)[MD1][MQ1] = (double (*)[MD1][MQ1]) (sm0+0);
double (*DDQ1)[MD1][MQ1] = (double (*)[MD1][MQ1]) (sm0+1);
double (*DQQ0)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+0);
double (*DQQ1)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+1);
double (*DQQ2)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm1+2);
double (*QQQ0)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+0);
double (*QQQ1)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+1);
double (*QQQ2)[MQ1][MQ1] = (double (*)[MQ1][MQ1]) (sm0+2);
double (*QQD0)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+0);
double (*QQD1)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+1);
double (*QQD2)[MQ1][MD1] = (double (*)[MQ1][MD1]) (sm1+2);
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(dy,y,D1DR)
{
MFEM_FOREACH_THREAD(dx,x,D1DR)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
}
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1DR)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(dy,y,D1DR)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1DR; ++dx)
{
const double coord = X[dz][dy][dx];
u += coord * B[qx][dx];
v += coord * G[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DR)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1DR; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dz = 0; dz < D1DR; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
QQQ0[qz][qy][qx] = (D(q,0,0,e)*gX) + (D(q,1,0,e)*gY) + (D(q,2,0,e)*gZ);
QQQ1[qz][qy][qx] = (D(q,0,1,e)*gX) + (D(q,1,1,e)*gY) + (D(q,2,1,e)*gZ);
QQQ2[qz][qy][qx] = (D(q,0,2,e)*gX) + (D(q,1,2,e)*gY) + (D(q,2,2,e)*gZ);
}
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1DE)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = bt(d,q);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += QQQ0[qz][qy][qx] * Bt[dx][qx];
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
QQD1[qz][qy][dx] = v;
QQD2[qz][qy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1DE)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
v += QQD1[qz][qy][dx] * Bt[dy][qy];
w += QQD2[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
QDD1[qz][dy][dx] = v;
QDD2[qz][dy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1DE)
{
MFEM_FOREACH_THREAD(dy,y,D1DE)
{
MFEM_FOREACH_THREAD(dx,x,D1DE)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
v += QDD1[qz][dy][dx] * Bt[dz][qz];
w += QDD2[qz][dy][dx] * Bt[dz][qz];
}
y(dx,dy,dz,0,e) += u;
y(dx,dy,dz,1,e) += v;
y(dx,dy,dz,2,e) += w;
}
}
}
});
}
static void PAGradientApply(const int dim,
const int TR_D1D,
const int TE_D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y,
bool transpose=false)
{
if (dim == 2)
{
return PAGradientApply2D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
}
if (dim == 3)
{
return PAGradientApply3D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
// PA Gradient Apply kernel
void GradientIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAGradientApply(dim, trial_dofs1D, test_dofs1D, quad1D, ne,
trial_maps->B, trial_maps->G, test_maps->Bt, pa_data, x, y,
false);
}
// PA Gradient Apply kernel
void GradientIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
{
MFEM_ABORT("PA Gradient AddMultTransposePA not implemented.");
}
} // namespace mfem
+441 -86
View File
@@ -12,6 +12,7 @@
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
#include "libceed/mass.hpp"
using namespace std;
@@ -21,14 +22,26 @@ namespace mfem
// PA Mass Integrator
// PA Mass Assemble kernel
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
{
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
ceedDataPtr = ptr;
InitCeedCoeff(Q, ptr);
return CeedPAMassAssemble(fes, *ir, *ptr);
}
#endif
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
@@ -52,7 +65,7 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
else
{
coeff.SetSize(nq * ne);
auto C = Reshape(coeff.Write(), nq, ne);
auto C = Reshape(coeff.HostWrite(), nq, ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
@@ -114,6 +127,324 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
}
}
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
SetupPA(fes);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassAssembleDiagonal2D(const int NE,
const Array<double> &b,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto D = Reshape(d.Read(), Q1D, Q1D, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QD[MQ1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dy = 0; dy < D1D; ++dy)
{
QD[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QD[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
}
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
for (int qx = 0; qx < Q1D; ++qx)
{
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPAMassAssembleDiagonal2D(const int NE,
const Array<double> &b_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_SHARED double B[MQ1][MD1];
MFEM_SHARED double QDZ[NBZ][MQ1][MD1];
double (*QD)[MD1] = (double (*)[MD1])(QDZ + tidz);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QD[qx][dy] += B[qy][dy] * B[qy][dy] * D(qx, qy, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int qx = 0; qx < Q1D; ++qx)
{
// might need absolute values on next line
Y(dx,dy,e) += B[qx][dx] * B[qx][dx] * QD[qx][dy];
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassAssembleDiagonal3D(const int NE,
const Array<double> &b,
const Vector &d,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
double QQD[MQ1][MQ1][MD1];
double QDD[MQ1][MD1][MD1];
for (int qx = 0; qx < Q1D; ++qx)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int dz = 0; dz < D1D; ++dz)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
QQD[qx][qy][dz] += B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
}
}
}
}
for (int qx = 0; qx < Q1D; ++qx)
{
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QDD[qx][dy][dz] += B(qy, dy) * B(qy, dy) * QQD[qx][qy][dz];
}
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
double t = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
t += B(qx, dx) * B(qx, dx) * QDD[qx][dy][dz];
}
Y(dx, dy, dz, e) += t;
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPAMassAssembleDiagonal3D(const int NE,
const Array<double> &b_,
const Vector &d_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double B[MQ1][MD1];
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
MFEM_SHARED double QDD[MQ1][MD1][MD1];
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
QQD[qx][qy][dz] += B[qz][dz] * B[qz][dz] * D(qx, qy, qz, e);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
QDD[qx][dy][dz] += B[qy][dy] * B[qy][dy] * QQD[qx][qy][dz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double t = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
t += B[qx][dx] * B[qx][dx] * QDD[qx][dy][dz];
}
Y(dx, dy, dz, e) += t;
}
}
}
});
}
static void PAMassAssembleDiagonal(const int dim, const int D1D,
const int Q1D, const int NE,
const Array<double> &B,
const Vector &D,
Vector &Y)
{
if (dim == 2)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPAMassAssembleDiagonal2D<2,2,16>(NE,B,D,Y);
case 0x33: return SmemPAMassAssembleDiagonal2D<3,3,16>(NE,B,D,Y);
case 0x44: return SmemPAMassAssembleDiagonal2D<4,4,8>(NE,B,D,Y);
case 0x55: return SmemPAMassAssembleDiagonal2D<5,5,8>(NE,B,D,Y);
case 0x66: return SmemPAMassAssembleDiagonal2D<6,6,4>(NE,B,D,Y);
case 0x77: return SmemPAMassAssembleDiagonal2D<7,7,4>(NE,B,D,Y);
case 0x88: return SmemPAMassAssembleDiagonal2D<8,8,2>(NE,B,D,Y);
case 0x99: return SmemPAMassAssembleDiagonal2D<9,9,2>(NE,B,D,Y);
default: return PAMassAssembleDiagonal2D(NE,B,D,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
case 0x89: return SmemPAMassAssembleDiagonal3D<8,9>(NE,B,D,Y);
default: return PAMassAssembleDiagonal3D(NE,B,D,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
}
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
#ifdef MFEM_USE_OCCA
// OCCA PA Mass Apply 2D kernel
static void OccaPAMassApply2D(const int D1D,
@@ -121,18 +452,18 @@ static void OccaPAMassApply2D(const int D1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
occa::properties props;
props["defines/D1D"] = D1D;
props["defines/Q1D"] = Q1D;
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -144,7 +475,7 @@ static void OccaPAMassApply2D(const int D1D,
"MassApply2D_CPU", props);
OccaMassApply2D_cpu.emplace(id, MassApply2D_CPU);
}
OccaMassApply2D_cpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
OccaMassApply2D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
}
else
{
@@ -156,7 +487,7 @@ static void OccaPAMassApply2D(const int D1D,
"MassApply2D_GPU", props);
OccaMassApply2D_gpu.emplace(id, MassApply2D_GPU);
}
OccaMassApply2D_gpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
OccaMassApply2D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
}
}
@@ -166,18 +497,18 @@ static void OccaPAMassApply3D(const int D1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
occa::properties props;
props["defines/D1D"] = D1D;
props["defines/Q1D"] = Q1D;
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
const occa::memory o_op = OccaMemoryRead(op.GetMemory(), op.Size());
const occa::memory o_x = OccaMemoryRead(x.GetMemory(), x.Size());
occa::memory o_y = OccaMemoryReadWrite(y.GetMemory(), y.Size());
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
const occa_id_t id = std::make_pair(D1D,Q1D);
if (!Device::Allows(Backend::OCCA_CUDA))
{
@@ -189,7 +520,7 @@ static void OccaPAMassApply3D(const int D1D,
"MassApply3D_CPU", props);
OccaMassApply3D_cpu.emplace(id, MassApply3D_CPU);
}
OccaMassApply3D_cpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
OccaMassApply3D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
}
else
{
@@ -201,17 +532,16 @@ static void OccaPAMassApply3D(const int D1D,
"MassApply3D_GPU", props);
OccaMassApply3D_gpu.emplace(id, MassApply3D_GPU);
}
OccaMassApply3D_gpu.at(id)(NE, o_B, o_Bt, o_op, o_x, o_y);
OccaMassApply3D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
}
}
#endif // MFEM_USE_OCCA
template<const int T_D1D = 0,
const int T_Q1D = 0>
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassApply2D(const int NE,
const Array<double> &B_,
const Array<double> &Bt_,
const Vector &op_,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -221,11 +551,11 @@ static void PAMassApply2D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto B = Reshape(b_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto X = Reshape(x_.Read(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
@@ -250,7 +580,7 @@ static void PAMassApply2D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,e);
const double s = X(dx,dy,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
@@ -269,7 +599,7 @@ static void PAMassApply2D(const int NE,
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= op(qx,qy,e);
sol_xy[qy][qx] *= D(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
@@ -292,20 +622,18 @@ static void PAMassApply2D(const int NE,
const double q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,e) += q2d * sol_x[dx];
Y(dx,dy,e) += q2d * sol_x[dx];
}
}
}
});
}
template<const int T_D1D = 0,
const int T_Q1D = 0,
const int T_NBZ = 0>
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
static void SmemPAMassApply2D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &op_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -319,9 +647,9 @@ static void SmemPAMassApply2D(const int NE,
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int tidz = MFEM_THREAD_ID(z);
@@ -349,11 +677,11 @@ static void SmemPAMassApply2D(const int NE,
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
B[q][dy] = b(q,dy);
}
}
}
@@ -380,17 +708,17 @@ static void SmemPAMassApply2D(const int NE,
{
qq += DQ[dy][qx] * B[qy][dy];
}
QQ[qy][qx] = qq * op(qx, qy, e);
QQ[qy][qx] = qq * D(qx, qy, e);
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = b(q,d);
Bt[dy][q] = b(q,dy);
}
}
}
@@ -417,18 +745,17 @@ static void SmemPAMassApply2D(const int NE,
{
dd += (QD[qy][dx] * Bt[dy][qy]);
}
y(dx, dy, e) += dd;
Y(dx, dy, e) += dd;
}
}
});
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
template<int T_D1D = 0, int T_Q1D = 0>
static void PAMassApply3D(const int NE,
const Array<double> &B_,
const Array<double> &Bt_,
const Vector &op_,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -438,11 +765,11 @@ static void PAMassApply3D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
auto B = Reshape(b_.Read(), Q1D, D1D);
auto Bt = Reshape(bt_.Read(), D1D, Q1D);
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto X = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -479,7 +806,7 @@ static void PAMassApply3D(const int NE,
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,e);
const double s = X(dx,dy,dz,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
@@ -512,7 +839,7 @@ static void PAMassApply3D(const int NE,
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
sol_xyz[qz][qy][qx] *= D(qx,qy,qz,e);
}
}
}
@@ -557,7 +884,7 @@ static void PAMassApply3D(const int NE,
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,e) += wz * sol_xy[dy][dx];
Y(dx,dy,dz,e) += wz * sol_xy[dy][dx];
}
}
}
@@ -565,12 +892,11 @@ static void PAMassApply3D(const int NE,
});
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPAMassApply3D(const int NE,
const Array<double> &b_,
const Array<double> &bt_,
const Vector &op_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
@@ -583,7 +909,7 @@ static void SmemPAMassApply3D(const int NE,
MFEM_VERIFY(D1D <= M1D, "");
MFEM_VERIFY(Q1D <= M1Q, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
@@ -669,7 +995,7 @@ static void SmemPAMassApply3D(const int NE,
{
u += DQQ[dz][qy][qx] * B[qz][dz];
}
QQQ[qz][qy][qx] = u * op(qx,qy,qz,e);
QQQ[qz][qy][qx] = u * d(qx,qy,qz,e);
}
}
}
@@ -741,22 +1067,20 @@ static void PAMassApply(const int dim,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
const Vector &D,
const Vector &X,
Vector &Y)
{
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
if (dim == 2)
{
OccaPAMassApply2D(D1D, Q1D, NE, B, Bt, op, x, y);
return;
return OccaPAMassApply2D(D1D,Q1D,NE,B,Bt,D,X,Y);
}
if (dim == 3)
{
OccaPAMassApply3D(D1D, Q1D, NE, B, Bt, op, x, y);
return;
return OccaPAMassApply3D(D1D,Q1D,NE,B,Bt,D,X,Y);
}
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
}
@@ -765,29 +1089,29 @@ static void PAMassApply(const int dim,
{
switch ((D1D << 4) | Q1D)
{
case 0x22: return SmemPAMassApply2D<2,2,16>(NE, B, Bt, op, x, y);
case 0x33: return SmemPAMassApply2D<3,3,16>(NE, B, Bt, op, x, y);
case 0x44: return SmemPAMassApply2D<4,4,8>(NE, B, Bt, op, x, y);
case 0x55: return SmemPAMassApply2D<5,5,8>(NE, B, Bt, op, x, y);
case 0x66: return SmemPAMassApply2D<6,6,4>(NE, B, Bt, op, x, y);
case 0x77: return SmemPAMassApply2D<7,7,4>(NE, B, Bt, op, x, y);
case 0x88: return SmemPAMassApply2D<8,8,2>(NE, B, Bt, op, x, y);
case 0x99: return SmemPAMassApply2D<9,9,2>(NE, B, Bt, op, x, y);
default: return PAMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
case 0x22: return SmemPAMassApply2D<2,2,16>(NE,B,Bt,D,X,Y);
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
case 0x88: return SmemPAMassApply2D<8,8,2>(NE,B,Bt,D,X,Y);
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
}
}
else if (dim == 3)
{
switch ((D1D << 4) | Q1D)
{
case 0x23: return SmemPAMassApply3D<2,3>(NE, B, Bt, op, x, y);
case 0x34: return SmemPAMassApply3D<3,4>(NE, B, Bt, op, x, y);
case 0x45: return SmemPAMassApply3D<4,5>(NE, B, Bt, op, x, y);
case 0x56: return SmemPAMassApply3D<5,6>(NE, B, Bt, op, x, y);
case 0x67: return SmemPAMassApply3D<6,7>(NE, B, Bt, op, x, y);
case 0x78: return SmemPAMassApply3D<7,8>(NE, B, Bt, op, x, y);
case 0x89: return SmemPAMassApply3D<8,9>(NE, B, Bt, op, x, y);
default: return PAMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
case 0x56: return SmemPAMassApply3D<5,6>(NE,B,Bt,D,X,Y);
case 0x67: return SmemPAMassApply3D<6,7>(NE,B,Bt,D,X,Y);
case 0x78: return SmemPAMassApply3D<7,8>(NE,B,Bt,D,X,Y);
case 0x89: return SmemPAMassApply3D<8,9>(NE,B,Bt,D,X,Y);
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
}
}
MFEM_ABORT("Unknown kernel.");
@@ -795,7 +1119,38 @@ static void PAMassApply(const int dim,
void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
{
PAMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
}
} // namespace mfem
+503
View File
@@ -0,0 +1,503 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Vector Diffusion Integrator
// PA Diffusion Assemble 2D kernel
static void PAVectorDiffusionSetup2D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double c_detJ = W[q] * COEFF / ((J11*J22)-(J21*J12));
y(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
y(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
y(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
});
}
// PA Diffusion Assemble 3D kernel
static void PAVectorDiffusionSetup3D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
const double COEFF,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 6, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double c_detJ = W[q] * COEFF / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
y(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
y(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
y(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
y(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
y(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
y(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
static void PAVectorDiffusionSetup(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &W,
const Vector &J,
const double COEFF,
Vector &op)
{
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
PAVectorDiffusionSetup2D(Q1D, NE, W, J, COEFF, op);
}
if (dim == 3)
{
PAVectorDiffusionSetup3D(Q1D, NE, W, J, COEFF, op);
}
}
void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir
= IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
coeff = cQ->constant;
}
PAVectorDiffusionSetup(dim, dofs1D, quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
// PA Diffusion Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0> static
void PAVectorDiffusionApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D, 3, NE);
auto x = Reshape(_x.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
for (int c = 0; c < VDIM; ++ c)
{
double grad[max_Q1D][max_Q1D][2];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] = 0.0;
grad[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qy][qx][0] += gradX[qx][1] * wy;
grad[qy][qx][1] += gradX[qx][0] * wDy;
}
}
}
// Calculate Dxy, xDy in plane
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O22 = op(q,2,e);
const double gradX = grad[qy][qx][0];
const double gradY = grad[qy][qx][1];
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double gradX[max_D1D][2];
for (int dx = 0; dx < D1D; ++dx)
{
gradX[dx][0] = 0;
gradX[dx][1] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double gX = grad[qy][qx][0];
const double gY = grad[qy][qx][1];
for (int dx = 0; dx < D1D; ++dx)
{
const double wx = Bt(dx,qx);
const double wDx = Gt(dx,qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += ((gradX[dx][0] * wy) + (gradX[dx][1] * wDy));
}
}
}
}
});
}
// PA Diffusion Apply 3D kernel
template<const int T_D1D = 0,
const int T_Q1D = 0> static
void PAVectorDiffusionApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Array<double> &gt,
const Vector &_op,
const Vector &_x,
Vector &_y,
int d1d = 0, int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Gt = Reshape(gt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
for (int c = 0; c < VDIM; ++ c)
{
double grad[max_Q1D][max_Q1D][max_Q1D][3];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] = 0.0;
grad[qz][qy][qx][1] = 0.0;
grad[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double gradXY[max_Q1D][max_Q1D][3];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
gradXY[qy][qx][0] = 0.0;
gradXY[qy][qx][1] = 0.0;
gradXY[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double gradX[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] = 0.0;
gradX[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
gradX[qx][0] += s * B(qx,dx);
gradX[qx][1] += s * G(qx,dx);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
const double wDy = G(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
const double wx = gradX[qx][0];
const double wDx = gradX[qx][1];
gradXY[qy][qx][0] += wDx * wy;
gradXY[qy][qx][1] += wx * wDy;
gradXY[qy][qx][2] += wx * wy;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
const double wDz = G(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
}
}
}
}
// Calculate Dxyz, xDyz, xyDz in plane
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double O11 = op(q,0,e);
const double O12 = op(q,1,e);
const double O13 = op(q,2,e);
const double O22 = op(q,3,e);
const double O23 = op(q,4,e);
const double O33 = op(q,5,e);
const double gradX = grad[qz][qy][qx][0];
const double gradY = grad[qz][qy][qx][1];
const double gradZ = grad[qz][qy][qx][2];
grad[qz][qy][qx][0] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
grad[qz][qy][qx][1] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
grad[qz][qy][qx][2] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double gradXY[max_D1D][max_D1D][3];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
gradXY[dy][dx][0] = 0;
gradXY[dy][dx][1] = 0;
gradXY[dy][dx][2] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double gradX[max_D1D][3];
for (int dx = 0; dx < D1D; ++dx)
{
gradX[dx][0] = 0;
gradX[dx][1] = 0;
gradX[dx][2] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double gX = grad[qz][qy][qx][0];
const double gY = grad[qz][qy][qx][1];
const double gZ = grad[qz][qy][qx][2];
for (int dx = 0; dx < D1D; ++dx)
{
const double wx = Bt(dx,qx);
const double wDx = Gt(dx,qx);
gradX[dx][0] += gX * wDx;
gradX[dx][1] += gY * wx;
gradX[dx][2] += gZ * wx;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
const double wDy = Gt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
gradXY[dy][dx][0] += gradX[dx][0] * wy;
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
gradXY[dy][dx][2] += gradX[dx][2] * wy;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const double wz = Bt(dz,qz);
const double wDz = Gt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) +=
((gradXY[dy][dx][0] * wz) +
(gradXY[dy][dx][1] * wz) +
(gradXY[dy][dx][2] * wDz));
}
}
}
}
}
});
}
static void PAVectorDiffusionApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &G,
const Array<double> &Bt,
const Array<double> &Gt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorDiffusionApply2D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
if (dim == 3)
{
return PAVectorDiffusionApply3D(NE,B,G,Bt,Gt,op,x,y,D1D,Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAVectorDiffusionApply(dim, dofs1D, quad1D, ne,
maps->B, maps->G, maps->Bt, maps->Gt,
pa_data, x, y);
}
} // namespace mfem
+367
View File
@@ -0,0 +1,367 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
#include "gridfunc.hpp"
using namespace std;
namespace mfem
{
// PA Mass Integrator
// PA Mass Assemble kernel
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
Mesh *mesh = fes.GetMesh();
if (mesh->GetNE() == 0) { return; }
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir
= IntRule ? IntRule : &MassIntegrator::GetRule(el, el, *T);
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, Device::GetMemoryType());
double coeff = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "Only ConstantCoefficient is supported.");
coeff = cQ->constant;
}
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
const double constant = coeff;
const int NE = ne;
const int NQ = nq;
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e);
const double J12 = J(q,1,0,e);
const double J21 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double detJ = (J11*J22)-(J21*J12);
v(q,e) = w[q] * constant * detJ;
}
});
}
if (dim == 3)
{
const double constant = coeff;
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
v(q,e) = W[q] * constant * detJ;
}
});
}
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAVectorMassApply2D(const int NE,
const Array<double> &_B,
const Array<double> &_Bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(_B.Read(), Q1D, D1D);
auto Bt = Reshape(_Bt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double sol_xy[max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++c)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double sol_x[max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
sol_x[qy] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double d2q = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += d2q * sol_x[qx];
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= op(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double s = sol_xy[qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += q2d * sol_x[dx];
}
}
}
}
});
}
template<const int T_D1D = 0,
const int T_Q1D = 0>
static void PAVectorMassApply3D(const int NE,
const Array<double> &_B,
const Array<double> &_Bt,
const Vector &_op,
const Vector &_x,
Vector &_y,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(_B.Read(), Q1D, D1D);
auto Bt = Reshape(_Bt.Read(), D1D, Q1D);
auto op = Reshape(_op.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(_x.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(_y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double sol_xyz[max_Q1D][max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++ c)
{
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double sol_xy[max_Q1D][max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double sol_x[max_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] = 0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double wy = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += wy * sol_x[qx];
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double wz = B(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double sol_xy[max_D1D][max_D1D];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const double s = sol_xyz[qz][qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const double wy = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] += wy * sol_x[dx];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const double wz = Bt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) += wz * sol_xy[dy][dx];
}
}
}
}
}
});
}
static void PAVectorMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<double> &B,
const Array<double> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
}
if (dim == 3)
{
return PAVectorMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
PAVectorMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
} // namespace mfem
+891
View File
@@ -0,0 +1,891 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "complex_fem.hpp"
using namespace std;
namespace mfem
{
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
: Vector(2*(fes->GetVSize()))
{
gfr = new GridFunction(fes, &data[0]);
gfi = new GridFunction(fes, &data[fes->GetVSize()]);
}
void
ComplexGridFunction::Update()
{
FiniteElementSpace * fes = gfr->FESpace();
int vsize = fes->GetVSize();
const Operator *T = fes->GetUpdateOperator();
if (T)
{
// Update the individual GridFunction objects. This will allocate new data
// arrays for each GridFunction.
gfr->Update();
gfi->Update();
// Our data array now contains old data as well as being the wrong size so
// reallocate it.
this->SetSize(2 * vsize);
// Create temporary vectors which point to the new data array
Vector gf_r(&data[0], vsize);
Vector gf_i(&data[vsize], vsize);
// Copy the updated GridFunctions into the new data array
gf_r = *gfr;
gf_i = *gfi;
// Replace the individual data arrays with pointers into the new data
// array
gfr->NewDataAndSize(&data[0], vsize);
gfi->NewDataAndSize(&data[vsize], vsize);
}
else
{
// The existing data will not be transferred to the new GridFunctions so
// delete it a allocate a new array
this->SetSize(2 * vsize);
// Point the individual GridFunctions to the new data array
gfr->NewDataAndSize(&data[0], vsize);
gfi->NewDataAndSize(&data[vsize], vsize);
// These updates will only set the proper 'sequence' value within
// the individual GridFunction objects because their sizes are
// already correct
gfr->Update();
gfi->Update();
}
}
void
ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
{
gfr->ProjectCoefficient(real_coeff);
gfi->ProjectCoefficient(imag_coeff);
}
void
ComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff)
{
gfr->ProjectCoefficient(real_vcoeff);
gfi->ProjectCoefficient(imag_vcoeff);
}
void
ComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr)
{
gfr->ProjectBdrCoefficient(real_coeff, attr);
gfi->ProjectBdrCoefficient(imag_coeff, attr);
}
void
ComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff,
Array<int> &attr)
{
gfr->ProjectBdrCoefficientNormal(real_vcoeff, attr);
gfi->ProjectBdrCoefficientNormal(imag_vcoeff, attr);
}
void
ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
VectorCoefficient
&imag_vcoeff,
Array<int> &attr)
{
gfr->ProjectBdrCoefficientTangent(real_vcoeff, attr);
gfi->ProjectBdrCoefficientTangent(imag_vcoeff, attr);
}
ComplexLinearForm::ComplexLinearForm(FiniteElementSpace *f,
ComplexOperator::Convention convention)
: Vector(2*(f->GetVSize())),
conv(convention)
{
lfr = new LinearForm(f, &data[0]);
lfi = new LinearForm(f, &data[f->GetVSize()]);
}
ComplexLinearForm::~ComplexLinearForm()
{
delete lfr;
delete lfi;
}
void
ComplexLinearForm::AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { lfr->AddDomainIntegrator(lfi_real); }
if ( lfi_imag ) { lfi->AddDomainIntegrator(lfi_imag); }
}
void
ComplexLinearForm::AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { lfr->AddBoundaryIntegrator(lfi_real); }
if ( lfi_imag ) { lfi->AddBoundaryIntegrator(lfi_imag); }
}
void
ComplexLinearForm::AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker)
{
if ( lfi_real ) { lfr->AddBoundaryIntegrator(lfi_real, bdr_attr_marker); }
if ( lfi_imag ) { lfi->AddBoundaryIntegrator(lfi_imag, bdr_attr_marker); }
}
void
ComplexLinearForm::AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { lfr->AddBdrFaceIntegrator(lfi_real); }
if ( lfi_imag ) { lfi->AddBdrFaceIntegrator(lfi_imag); }
}
void
ComplexLinearForm::AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker)
{
if ( lfi_real ) { lfr->AddBdrFaceIntegrator(lfi_real, bdr_attr_marker); }
if ( lfi_imag ) { lfi->AddBdrFaceIntegrator(lfi_imag, bdr_attr_marker); }
}
void
ComplexLinearForm::Update()
{
FiniteElementSpace *fes = lfr->FESpace();
this->Update(fes);
}
void
ComplexLinearForm::Update(FiniteElementSpace *fes)
{
int vsize = fes->GetVSize();
SetSize(2 * vsize);
Vector vlfr(&data[0], vsize);
Vector vlfi(&data[vsize], vsize);
lfr->Update(fes, vlfr, 0);
lfi->Update(fes, vlfi, 0);
}
void
ComplexLinearForm::Assemble()
{
lfr->Assemble();
lfi->Assemble();
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
{
*lfi *= -1.0;
}
}
complex<double>
ComplexLinearForm::operator()(const ComplexGridFunction &gf) const
{
double s = (conv == ComplexOperator::HERMITIAN)?1.0:-1.0;
return complex<double>((*lfr)(gf.real()) - s * (*lfi)(gf.imag()),
(*lfr)(gf.imag()) + s * (*lfi)(gf.real()));
}
SesquilinearForm::SesquilinearForm(FiniteElementSpace *f,
ComplexOperator::Convention convention)
: conv(convention),
blfr(new BilinearForm(f)),
blfi(new BilinearForm(f))
{}
SesquilinearForm::~SesquilinearForm()
{
delete blfr;
delete blfi;
}
void SesquilinearForm::AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { blfr->AddDomainIntegrator(bfi_real); }
if (bfi_imag) { blfi->AddDomainIntegrator(bfi_imag); }
}
void
SesquilinearForm::AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { blfr->AddBoundaryIntegrator(bfi_real); }
if (bfi_imag) { blfi->AddBoundaryIntegrator(bfi_imag); }
}
void
SesquilinearForm::AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> & bdr_marker)
{
if (bfi_real) { blfr->AddBoundaryIntegrator(bfi_real, bdr_marker); }
if (bfi_imag) { blfi->AddBoundaryIntegrator(bfi_imag, bdr_marker); }
}
void
SesquilinearForm::AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { blfr->AddInteriorFaceIntegrator(bfi_real); }
if (bfi_imag) { blfi->AddInteriorFaceIntegrator(bfi_imag); }
}
void SesquilinearForm::AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { blfr->AddBdrFaceIntegrator(bfi_real); }
if (bfi_imag) { blfi->AddBdrFaceIntegrator(bfi_imag); }
}
void SesquilinearForm::AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker)
{
if (bfi_real) { blfr->AddBdrFaceIntegrator(bfi_real, bdr_marker); }
if (bfi_imag) { blfi->AddBdrFaceIntegrator(bfi_imag, bdr_marker); }
}
void
SesquilinearForm::Assemble(int skip_zeros)
{
blfr->Assemble(skip_zeros);
blfi->Assemble(skip_zeros);
}
void
SesquilinearForm::Finalize(int skip_zeros)
{
blfr->Finalize(skip_zeros);
blfi->Finalize(skip_zeros);
}
ComplexSparseMatrix *
SesquilinearForm::AssembleComplexSparseMatrix()
{
return new ComplexSparseMatrix(&blfr->SpMat(),
&blfi->SpMat(),
false, false, conv);
}
void
SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B,
int ci)
{
FiniteElementSpace * fes = blfr->FESpace();
int vsize = fes->GetVSize();
double s = (conv == ComplexOperator::HERMITIAN)?1.0:-1.0;
// Allocate temporary vectors
Vector b_0(vsize); b_0 = 0.0;
// Extract the real and imaginary parts of the input vectors
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
Vector x_r(x.GetData(), vsize);
Vector x_i(&(x.GetData())[vsize], vsize);
MFEM_ASSERT(b.Size() == 2 * vsize, "Input LinearForm of incorrect size!");
Vector b_r(b.GetData(), vsize);
Vector b_i(&(b.GetData())[vsize], vsize);
b_i *= s;
SparseMatrix * A_r = new SparseMatrix;
SparseMatrix * A_i = new SparseMatrix;
Vector X_0, B_0;
b_0 = b_r;
blfr->FormLinearSystem(ess_tdof_list, x_r, b_r, *A_r, X_0, B_0, ci);
int tvsize = B_0.Size();
X.SetSize(2 * tvsize);
B.SetSize(2 * tvsize);
Vector X_r(X.GetData(), tvsize);
Vector X_i(&(X.GetData())[tvsize], tvsize);
Vector B_r(B.GetData(), tvsize);
Vector B_i(&(B.GetData())[tvsize], tvsize);
X_r = X_0; B_r = B_0;
b_0 = 0.0;
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, false);
B_r -= B_0;
b_0 = b_i;
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_r, X_0, B_0, ci);
X_i = X_0; B_i = B_0;
b_0 = 0.0;
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, false);
B_i += B_0;
B_i *= s;
b_i *= s;
// A = A_r + i A_i
A.Clear();
ComplexSparseMatrix * A_sp =
new ComplexSparseMatrix(A_r, A_i, true, true, conv);
A.Reset<ComplexSparseMatrix>(A_sp, true);
}
void
SesquilinearForm::RecoverFEMSolution(const Vector &X, const Vector &b,
Vector &x)
{
FiniteElementSpace * fes = blfr->FESpace();
const SparseMatrix *P = fes->GetConformingProlongation();
int vsize = fes->GetVSize();
int tvsize = X.Size() / 2;
Vector X_r(X.GetData(), tvsize);
Vector X_i(&(X.GetData())[tvsize], tvsize);
Vector x_r(x.GetData(), vsize);
Vector x_i(&(x.GetData())[vsize], vsize);
if (!P)
{
x = X;
}
else
{
// Apply conforming prolongation
P->Mult(X_r, x_r);
P->Mult(X_i, x_i);
}
}
void
SesquilinearForm::Update(FiniteElementSpace *nfes)
{
if ( blfr ) { blfr->Update(nfes); }
if ( blfi ) { blfi->Update(nfes); }
}
#ifdef MFEM_USE_MPI
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
: Vector(2*(pfes->GetVSize()))
{
pgfr = new ParGridFunction(pfes, &data[0]);
pgfi = new ParGridFunction(pfes, &data[pfes->GetVSize()]);
}
void
ParComplexGridFunction::Update()
{
ParFiniteElementSpace * pfes = pgfr->ParFESpace();
int vsize = pfes->GetVSize();
const Operator *T = pfes->GetUpdateOperator();
if (T)
{
// Update the individual GridFunction objects. This will allocate new data
// arrays for each GridFunction.
pgfr->Update();
pgfi->Update();
// Our data array now contains old data as well as being the wrong size
// so reallocate it.
this->SetSize(2 * vsize);
// Create temporary vectors which point to the new data array
Vector gf_r(&data[0], vsize);
Vector gf_i(&data[vsize], vsize);
// Copy the updated GridFunctions into the new data array
gf_r = *pgfr;
gf_i = *pgfi;
// Replace the individual data arrays with pointers into the new data
// array
pgfr->NewDataAndSize(&data[0], vsize);
pgfi->NewDataAndSize(&data[vsize], vsize);
}
else
{
// The existing data will not be transferred to the new GridFunctions so
// delete it a allocate a new array
this->SetSize(2 * vsize);
// Point the individual GridFunctions to the new data array
pgfr->NewDataAndSize(&data[0], vsize);
pgfi->NewDataAndSize(&data[vsize], vsize);
// These updates will only set the proper 'sequence' value within the
// individual GridFunction objects because their sizes are already correct
pgfr->Update();
pgfi->Update();
}
}
void
ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
{
pgfr->ProjectCoefficient(real_coeff);
pgfi->ProjectCoefficient(imag_coeff);
}
void
ParComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff)
{
pgfr->ProjectCoefficient(real_vcoeff);
pgfi->ProjectCoefficient(imag_vcoeff);
}
void
ParComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr)
{
pgfr->ProjectBdrCoefficient(real_coeff, attr);
pgfi->ProjectBdrCoefficient(imag_coeff, attr);
}
void
ParComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient
&real_vcoeff,
VectorCoefficient
&imag_vcoeff,
Array<int> &attr)
{
pgfr->ProjectBdrCoefficientNormal(real_vcoeff, attr);
pgfi->ProjectBdrCoefficientNormal(imag_vcoeff, attr);
}
void
ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
VectorCoefficient
&imag_vcoeff,
Array<int> &attr)
{
pgfr->ProjectBdrCoefficientTangent(real_vcoeff, attr);
pgfi->ProjectBdrCoefficientTangent(imag_vcoeff, attr);
}
void
ParComplexGridFunction::Distribute(const Vector *tv)
{
ParFiniteElementSpace * pfes = pgfr->ParFESpace();
HYPRE_Int size = pfes->GetTrueVSize();
double * tvd = tv->GetData();
Vector tvr(tvd, size);
Vector tvi(&tvd[size], size);
pgfr->Distribute(tvr);
pgfi->Distribute(tvi);
}
void
ParComplexGridFunction::ParallelProject(Vector &tv) const
{
ParFiniteElementSpace * pfes = pgfr->ParFESpace();
HYPRE_Int size = pfes->GetTrueVSize();
double * tvd = tv.GetData();
Vector tvr(tvd, size);
Vector tvi(&tvd[size], size);
pgfr->ParallelProject(tvr);
pgfi->ParallelProject(tvi);
}
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ComplexOperator::Convention
convention)
: Vector(2*(pfes->GetVSize())),
conv(convention)
{
plfr = new ParLinearForm(pfes, &data[0]);
plfi = new ParLinearForm(pfes, &data[pfes->GetVSize()]);
HYPRE_Int * tdof_offsets_fes = pfes->GetTrueDofOffsets();
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
tdof_offsets = new HYPRE_Int[n+1];
for (int i=0; i<=n; i++)
{
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
}
}
ParComplexLinearForm::~ParComplexLinearForm()
{
delete plfr;
delete plfi;
delete [] tdof_offsets;
}
void
ParComplexLinearForm::AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { plfr->AddDomainIntegrator(lfi_real); }
if ( lfi_imag ) { plfi->AddDomainIntegrator(lfi_imag); }
}
void
ParComplexLinearForm::AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { plfr->AddBoundaryIntegrator(lfi_real); }
if ( lfi_imag ) { plfi->AddBoundaryIntegrator(lfi_imag); }
}
void
ParComplexLinearForm::AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker)
{
if ( lfi_real ) { plfr->AddBoundaryIntegrator(lfi_real, bdr_attr_marker); }
if ( lfi_imag ) { plfi->AddBoundaryIntegrator(lfi_imag, bdr_attr_marker); }
}
void
ParComplexLinearForm::AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag)
{
if ( lfi_real ) { plfr->AddBdrFaceIntegrator(lfi_real); }
if ( lfi_imag ) { plfi->AddBdrFaceIntegrator(lfi_imag); }
}
void
ParComplexLinearForm::AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker)
{
if ( lfi_real ) { plfr->AddBdrFaceIntegrator(lfi_real, bdr_attr_marker); }
if ( lfi_imag ) { plfi->AddBdrFaceIntegrator(lfi_imag, bdr_attr_marker); }
}
void
ParComplexLinearForm::Update(ParFiniteElementSpace *pf)
{
ParFiniteElementSpace *pfes = (pf!=NULL)?pf:plfr->ParFESpace();
int vsize = pfes->GetVSize();
SetSize(2 * vsize);
Vector vplfr(&data[0], vsize);
Vector vplfi(&data[vsize], vsize);
plfr->Update(pfes, vplfr, 0);
plfi->Update(pfes, vplfi, 0);
}
void
ParComplexLinearForm::Assemble()
{
plfr->Assemble();
plfi->Assemble();
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
{
*plfi *= -1.0;
}
}
void
ParComplexLinearForm::ParallelAssemble(Vector &tv)
{
HYPRE_Int size = plfr->ParFESpace()->GetTrueVSize();
double * tvd = tv.GetData();
Vector tvr(tvd, size);
Vector tvi(&tvd[size], size);
plfr->ParallelAssemble(tvr);
plfi->ParallelAssemble(tvi);
}
HypreParVector *
ParComplexLinearForm::ParallelAssemble()
{
const ParFiniteElementSpace * pfes = plfr->ParFESpace();
HypreParVector * tv = new HypreParVector(pfes->GetComm(),
2*(pfes->GlobalTrueVSize()),
tdof_offsets);
HYPRE_Int size = pfes->GetTrueVSize();
double * tvd = tv->GetData();
Vector tvr(tvd, size);
Vector tvi(&tvd[size], size);
plfr->ParallelAssemble(tvr);
plfi->ParallelAssemble(tvi);
return tv;
}
complex<double>
ParComplexLinearForm::operator()(const ParComplexGridFunction &gf) const
{
double s = (conv == ComplexOperator::HERMITIAN)?1.0:-1.0;
return complex<double>((*plfr)(gf.real()) - s * (*plfi)(gf.imag()),
(*plfr)(gf.imag()) + s * (*plfi)(gf.real()));
}
ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
convention)
: conv(convention),
pblfr(new ParBilinearForm(pf)),
pblfi(new ParBilinearForm(pf))
{}
ParSesquilinearForm::~ParSesquilinearForm()
{
delete pblfr;
delete pblfi;
}
void ParSesquilinearForm::AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { pblfr->AddDomainIntegrator(bfi_real); }
if (bfi_imag) { pblfi->AddDomainIntegrator(bfi_imag); }
}
void
ParSesquilinearForm::AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { pblfr->AddBoundaryIntegrator(bfi_real); }
if (bfi_imag) { pblfi->AddBoundaryIntegrator(bfi_imag); }
}
void
ParSesquilinearForm::AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> & bdr_marker)
{
if (bfi_real) { pblfr->AddBoundaryIntegrator(bfi_real, bdr_marker); }
if (bfi_imag) { pblfi->AddBoundaryIntegrator(bfi_imag, bdr_marker); }
}
void
ParSesquilinearForm::AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { pblfr->AddInteriorFaceIntegrator(bfi_real); }
if (bfi_imag) { pblfi->AddInteriorFaceIntegrator(bfi_imag); }
}
void
ParSesquilinearForm::AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag)
{
if (bfi_real) { pblfr->AddBdrFaceIntegrator(bfi_real); }
if (bfi_imag) { pblfi->AddBdrFaceIntegrator(bfi_imag); }
}
void
ParSesquilinearForm::AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker)
{
if (bfi_real) { pblfr->AddBdrFaceIntegrator(bfi_real, bdr_marker); }
if (bfi_imag) { pblfi->AddBdrFaceIntegrator(bfi_imag, bdr_marker); }
}
void
ParSesquilinearForm::Assemble(int skip_zeros)
{
pblfr->Assemble(skip_zeros);
pblfi->Assemble(skip_zeros);
}
void
ParSesquilinearForm::Finalize(int skip_zeros)
{
pblfr->Finalize(skip_zeros);
pblfi->Finalize(skip_zeros);
}
ComplexHypreParMatrix *
ParSesquilinearForm::ParallelAssemble()
{
return new ComplexHypreParMatrix(pblfr->ParallelAssemble(),
pblfi->ParallelAssemble(),
true, true, conv);
}
void
ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A,
Vector &X, Vector &B,
int ci)
{
int vsize = x.Size() / 2;
double s = (conv == ComplexOperator::HERMITIAN)?1.0:-1.0;
// Allocate temporary vectors
Vector b_0(vsize); b_0 = 0.0;
// Extract the real and imaginary parts of the input vectors
Vector x_r(x.GetData(), vsize);
Vector x_i(&(x.GetData())[vsize], vsize);
MFEM_ASSERT(b.Size() == 2 * vsize, "Input LinearForm of incorrect size!");
Vector b_r(b.GetData(), vsize);
Vector b_i(&(b.GetData())[vsize], vsize);
b_i *= s;
OperatorHandle A_r, A_i;
Vector X_0, B_0;
b_0 = b_r;
pblfr->FormLinearSystem(ess_tdof_list, x_r, b_0, A_r, X_0, B_0, ci);
int tvsize = B_0.Size();
X.SetSize(2 * tvsize);
B.SetSize(2 * tvsize);
Vector X_r(X.GetData(), tvsize);
Vector X_i(&(X.GetData())[tvsize], tvsize);
Vector B_r(B.GetData(), tvsize);
Vector B_i(&(B.GetData())[tvsize], tvsize);
X_r = X_0; B_r = B_0;
b_0 = 0.0;
pblfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, false);
B_r -= B_0;
b_0 = b_i;
pblfr->FormLinearSystem(ess_tdof_list, x_i, b_0, A_r, X_0, B_0, ci);
X_i = X_0; B_i = B_0;
b_0 = 0.0;
pblfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, false);
B_i += B_0;
B_i *= s;
b_i *= s;
// Modify RHS and offdiagonal blocks (Imaginary parts of the matrix) to
// conform with standard essential BC treatment i.e. zero out rows and
// columns and place ones on the diagonal.
if ( A_i.Type() == Operator::Hypre_ParCSR )
{
int n = ess_tdof_list.Size();
int j;
HypreParMatrix * Ah; A_i.Get(Ah);
hypre_ParCSRMatrix * Aih =
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
for (int k=0; k<n; k++)
{
j=ess_tdof_list[k];
Aih->diag->data[Aih->diag->i[j]] = 0.0;
B_r(j) = X_r(j);
B_i(j) = X_i(j);
}
}
// A = A_r + i A_i
A.Clear();
if ( A_r.Type() == Operator::Hypre_ParCSR &&
A_i.Type() == Operator::Hypre_ParCSR )
{
ComplexHypreParMatrix * A_hyp =
new ComplexHypreParMatrix(A_r.As<HypreParMatrix>(),
A_i.As<HypreParMatrix>(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexHypreParMatrix>(A_hyp, true);
}
else
{
ComplexOperator * A_op =
new ComplexOperator(A_r.As<Operator>(),
A_i.As<Operator>(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexOperator>(A_op, true);
}
}
void
ParSesquilinearForm::RecoverFEMSolution(const Vector &X, const Vector &b,
Vector &x)
{
ParFiniteElementSpace * pfes = pblfr->ParFESpace();
const Operator &P = *pfes->GetProlongationMatrix();
int vsize = pfes->GetVSize();
int tvsize = X.Size() / 2;
Vector X_r(X.GetData(), tvsize);
Vector X_i(&(X.GetData())[tvsize], tvsize);
Vector x_r(x.GetData(), vsize);
Vector x_i(&(x.GetData())[vsize], vsize);
// Apply conforming prolongation
P.Mult(X_r, x_r);
P.Mult(X_i, x_i);
}
void
ParSesquilinearForm::Update(FiniteElementSpace *nfes)
{
if ( pblfr ) { pblfr->Update(nfes); }
if ( pblfi ) { pblfi->Update(nfes); }
}
#endif // MFEM_USE_MPI
}
+530
View File
@@ -0,0 +1,530 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_COMPLEX_FEM
#define MFEM_COMPLEX_FEM
#include "../linalg/complex_operator.hpp"
#include "gridfunc.hpp"
#include "linearform.hpp"
#include "bilinearform.hpp"
#ifdef MFEM_USE_MPI
#include "pgridfunc.hpp"
#include "plinearform.hpp"
#include "pbilinearform.hpp"
#endif
#include <complex>
namespace mfem
{
/// Class for complex-valued grid function - real + imaginary part Vector with
/// associated FE space.
class ComplexGridFunction : public Vector
{
private:
GridFunction * gfr;
GridFunction * gfi;
protected:
void Destroy() { delete gfr; delete gfi; }
public:
/* @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
ComplexGridFunction(FiniteElementSpace *f);
void Update();
/// Assign constant values to the ComplexGridFunction data.
ComplexGridFunction &operator=(const std::complex<double> & value)
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
virtual void ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff);
virtual void ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff);
virtual void ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientNormal(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
FiniteElementSpace *FESpace() { return gfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return gfr->FESpace(); }
GridFunction & real() { return *gfr; }
GridFunction & imag() { return *gfi; }
const GridFunction & real() const { return *gfr; }
const GridFunction & imag() const { return *gfi; }
/// Destroys the grid function.
virtual ~ComplexGridFunction() { Destroy(); }
};
/** Class for a complex-valued linear form
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ComplexLinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued field is equal to zero.
*/
class ComplexLinearForm : public Vector
{
private:
ComplexOperator::Convention conv;
protected:
LinearForm * lfr;
LinearForm * lfi;
public:
ComplexLinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
virtual ~ComplexLinearForm();
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
/// Adds new Boundary Face Integrator. Assumes ownership of @a lfi.
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Face Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
FiniteElementSpace *FESpace() const { return lfr->FESpace(); }
LinearForm & real() { return *lfr; }
LinearForm & imag() { return *lfi; }
const LinearForm & real() const { return *lfr; }
const LinearForm & imag() const { return *lfi; }
void Update();
void Update(FiniteElementSpace *f);
/// Assembles the linear form i.e. sums over all domain/bdr integrators.
void Assemble();
std::complex<double> operator()(const ComplexGridFunction &gf) const;
};
/** Class for sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but the first
argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the SesquilinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class SesquilinearForm
{
private:
ComplexOperator::Convention conv;
BilinearForm *blfr;
BilinearForm *blfi;
public:
SesquilinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
BilinearForm & real() { return *blfr; }
BilinearForm & imag() { return *blfi; }
const BilinearForm & real() const { return *blfr; }
const BilinearForm & imag() const { return *blfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator, restricted to specific boundary attributes.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
/** The returned matrix has to be deleted by the caller. */
ComplexSparseMatrix *AssembleComplexSparseMatrix();
/// Return the parallel FE space associated with the ParBilinearForm.
FiniteElementSpace *FESpace() const { return blfr->FESpace(); }
void FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0);
/** Call this method after solving a linear system constructed using the
FormLinearSystem method to recover the solution as a ParGridFunction-size
vector in x. Use the same arguments as in the FormLinearSystem call. */
virtual void RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x);
virtual void Update(FiniteElementSpace *nfes = NULL);
virtual ~SesquilinearForm();
};
#ifdef MFEM_USE_MPI
/// Class for parallel complex-valued grid function - real + imaginary part
/// Vector with associated parallel FE space.
class ParComplexGridFunction : public Vector
{
private:
ParGridFunction * pgfr;
ParGridFunction * pgfi;
protected:
void Destroy() { delete pgfr; delete pgfi; }
public:
/* @brief Construct a ParComplexGridFunction associated with the
ParFiniteElementSpace @a *f. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
void Update();
/// Assign constant values to the ParComplexGridFunction data.
ParComplexGridFunction &operator=(const std::complex<double> & value)
{ *pgfr = value.real(); *pgfi = value.imag(); return *this; }
virtual void ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff);
virtual void ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff);
virtual void ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientNormal(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &real_coeff,
VectorCoefficient &imag_coeff,
Array<int> &attr);
void Distribute(const Vector *tv);
void Distribute(const Vector &tv) { Distribute(&tv); }
/// Returns the vector restricted to the true dofs.
void ParallelProject(Vector &tv) const;
FiniteElementSpace *FESpace() { return pgfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return pgfr->FESpace(); }
ParFiniteElementSpace *ParFESpace() { return pgfr->ParFESpace(); }
const ParFiniteElementSpace *ParFESpace() const { return pgfr->ParFESpace(); }
ParGridFunction & real() { return *pgfr; }
ParGridFunction & imag() { return *pgfi; }
const ParGridFunction & real() const { return *pgfr; }
const ParGridFunction & imag() const { return *pgfi; }
virtual double ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
double err_r = pgfr->ComputeL2Error(exsolr, irs);
double err_i = pgfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
}
virtual double ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
{
double err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
double err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
/// Destroys grid function.
virtual ~ParComplexGridFunction() { Destroy(); }
};
/** Class for a complex-valued, parallel linear form
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ParComplexLinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued field is equal to zero.
*/
class ParComplexLinearForm : public Vector
{
private:
ComplexOperator::Convention conv;
protected:
ParLinearForm * plfr;
ParLinearForm * plfi;
HYPRE_Int * tdof_offsets;
public:
ParComplexLinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
virtual ~ParComplexLinearForm();
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBoundaryIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
/// Adds new Boundary Face Integrator. Assumes ownership of @a lfi.
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag);
/** @brief Add new Boundary Face Integrator, restricted to the given boundary
attributes.
Assumes ownership of @a lfi_real and @a lfi_imag.
The array @a bdr_attr_marker is stored internally as a pointer to the
given Array<int> object. */
void AddBdrFaceIntegrator(LinearFormIntegrator *lfi_real,
LinearFormIntegrator *lfi_imag,
Array<int> &bdr_attr_marker);
ParFiniteElementSpace *ParFESpace() const { return plfr->ParFESpace(); }
ParLinearForm & real() { return *plfr; }
ParLinearForm & imag() { return *plfi; }
const ParLinearForm & real() const { return *plfr; }
const ParLinearForm & imag() const { return *plfi; }
void Update(ParFiniteElementSpace *pf = NULL);
/// Assembles the linear form i.e. sums over all domain/bdr integrators.
void Assemble();
/// Assemble the vector on the true dofs, i.e. P^t v.
void ParallelAssemble(Vector &tv);
/// Returns the vector assembled on the true dofs, i.e. P^t v.
HypreParVector *ParallelAssemble();
std::complex<double> operator()(const ParComplexGridFunction &gf) const;
};
/** Class for a parallel sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but but the
first argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
The @a convention argument in the class's constructor is documented in the
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
When supplying integrators to the ParSesquilinearForm either the real or
imaginary integrator can be NULL. This indicates that the corresponding
portion of the complex-valued material coefficient is equal to zero.
*/
class ParSesquilinearForm
{
private:
ComplexOperator::Convention conv;
ParBilinearForm *pblfr;
ParBilinearForm *pblfi;
public:
ParSesquilinearForm(ParFiniteElementSpace *pf,
ComplexOperator::Convention
convention = ComplexOperator::HERMITIAN);
ComplexOperator::Convention GetConvention() const { return conv; }
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
ParBilinearForm & real() { return *pblfr; }
ParBilinearForm & imag() { return *pblfi; }
const ParBilinearForm & real() const { return *pblfr; }
const ParBilinearForm & imag() const { return *pblfi; }
/// Adds new Domain Integrator.
void AddDomainIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new Boundary Integrator.
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBoundaryIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
void AddInteriorFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag);
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
attributes.
Assumes ownership of @a bfi.
The array @a bdr_marker is stored internally as a pointer to the given
Array<int> object. */
void AddBdrFaceIntegrator(BilinearFormIntegrator *bfi_real,
BilinearFormIntegrator *bfi_imag,
Array<int> &bdr_marker);
/// Assemble the local matrix
void Assemble(int skip_zeros = 1);
/// Finalizes the matrix initialization.
void Finalize(int skip_zeros = 1);
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
/** The returned matrix has to be deleted by the caller. */
ComplexHypreParMatrix *ParallelAssemble();
/// Return the parallel FE space associated with the ParBilinearForm.
ParFiniteElementSpace *ParFESpace() const { return pblfr->ParFESpace(); }
void FormLinearSystem(const Array<int> &ess_tdof_list, Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0);
/** Call this method after solving a linear system constructed using the
FormLinearSystem method to recover the solution as a ParGridFunction-size
vector in x. Use the same arguments as in the FormLinearSystem call. */
virtual void RecoverFEMSolution(const Vector &X, const Vector &b, Vector &x);
virtual void Update(FiniteElementSpace *nfes = NULL);
virtual ~ParSesquilinearForm();
};
#endif // MFEM_USE_MPI
}
#endif // MFEM_COMPLEX_FEM
+369 -1
View File
@@ -14,7 +14,6 @@
#include "../general/text.hpp"
#include "picojson.h"
#include <fstream>
#include <cerrno> // errno
#include <sstream>
@@ -726,4 +725,373 @@ void VisItDataCollection::ParseVisItRootString(const std::string& json)
}
}
ParaViewDataCollection::~ParaViewDataCollection()
{
if (myrank==0)
{
// Close the data collection
pvd_stream << "</Collection>" << std::endl;
pvd_stream << "</VTKFile>" << std::endl;
pvd_stream.close();
}
}
ParaViewDataCollection::ParaViewDataCollection(const std::string&
collection_name,
mfem::Mesh *mesh_)
:DataCollection(collection_name, mesh_)
{
myrank = 0;
nprocs = 1;
levels_of_detail = 1;
#ifdef MFEM_USE_MPI
lcomm = MPI_COMM_SELF;
#endif
std::string dpath=GenerateCollectionPath();
std::string pvdname=dpath+"/"+GeneratePVDFileName();
create_directory(dpath); // this one is a serial
pvd_stream.open(pvdname.c_str(),std::ios::out);
// initialize the file
pvd_stream << "<?xml version=\"1.0\"?>" << std::endl;
pvd_stream << "<VTKFile type=\"Collection\" version=\"0.1\"" << std::endl;
pvd_stream << " byte_order=\"LittleEndian\"" << std::endl;
pvd_stream << " compressor=\"vtkZLibDataCompressor\">" << std::endl;
pvd_stream << "<Collection>" << std::endl;
}
void ParaViewDataCollection::SetMesh(mfem::Mesh * new_mesh)
{
DataCollection::SetMesh(new_mesh);
}
void ParaViewDataCollection::RegisterField(const std::string& field_name,
mfem::GridFunction *gf)
{
DataCollection::RegisterField(field_name,gf);
}
void ParaViewDataCollection::SetLevelsOfDetail(int levels_of_detail_)
{
levels_of_detail = levels_of_detail_;
}
void ParaViewDataCollection::Load(int )
{
MFEM_WARNING("ParaViewDataCollection::Load() is not implemented!");
}
std::string ParaViewDataCollection::GenerateCollectionPath()
{
std::string out = "";
out=DataCollection::GetPrefixPath() + DataCollection::GetCollectionName();
return out;
}
std::string ParaViewDataCollection::GeneratePVTUPath()
{
std::string out = "Cycle" + to_padded_string(cycle,pad_digits_cycle);
return out;
}
std::string ParaViewDataCollection::GenerateVTUPath()
{
std::string out = GeneratePVTUPath();
return out;
}
std::string ParaViewDataCollection::GeneratePVDFileName()
{
std::string out = GetCollectionName()+".pvd";
return out;
}
std::string ParaViewDataCollection::GeneratePVTUFileName()
{
std::string out = "data.pvtu";
return out;
}
std::string ParaViewDataCollection::GenerateVTUFileName()
{
std::string out = "proc" + to_padded_string(myrank,pad_digits_rank)+".vtu";
return out;
}
std::string ParaViewDataCollection::GenerateVTUFileName(int crank)
{
std::string out = "proc" + to_padded_string(crank,pad_digits_rank)+".vtu";
return out;
}
void ParaViewDataCollection::Save()
{
// add a new collection to the PDV file
// check if the directories are created
{
std::string path = GenerateCollectionPath()+"/"+GenerateVTUPath();
#ifndef MFEM_USE_MPI
int err = create_directory(path);
#else
int err;
if (nprocs==1)
{
err = create_directory(path);
}
else
{
err = create_directory(path,myrank,lcomm);
}
#endif
if (err)
{
error = WRITE_ERROR;
MFEM_WARNING("Error creating directory: " << path);
return; // do not even try to write the mesh
}
}
// the directory is created
// define the vtu file
{
std::string fname = GenerateCollectionPath()+"/"+GenerateVTUPath()+"/"
+GenerateVTUFileName();
std::fstream out; out.open(fname.c_str(), std::ios::out);
SaveDataVTU(out,levels_of_detail);
out.close();
}
// define the pvtu file only on process 0
if (myrank==0)
{
std::string fname = GenerateCollectionPath()+"/"+GeneratePVTUPath()+"/"
+GeneratePVTUFileName();
std::fstream out; out.open(fname.c_str(), std::ios::out);
out << "<?xml version=\"1.0\"?>" << std::endl;
out << "<VTKFile type=\"PUnstructuredGrid\"";
out << " version =\"0.1\" byte_order=\"LittleEndian\"> " << std::endl;
out << "<PUnstructuredGrid GhostLevel=\"0\">" << std::endl ;
out << "<PPoints>" << std::endl;
out << "\t<PDataArray type=\"Float64\" ";
out << " Name=\"Points\" NumberOfComponents=\"3\"/>" << std::endl;
out << "</PPoints>" << std::endl;
out << "<PCells>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" ";
out << " Name=\"connectivity\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" ";
out << " Name=\"offsets\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "\t<PDataArray type=\"UInt8\" ";
out << " Name=\"types\" NumberOfComponents=\"1\"/>" << std::endl ;
out << "</PCells>" << std::endl ;
out << "<PPointData>" << std::endl ;
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
out << "<PDataArray type=\"Float64\" Name=\"" << it->first;
int vec_dim=it->second->VectorDim();
out<<"\" NumberOfComponents=\""<< vec_dim <<"\" format=\"ascii\" />" <<
std::endl;
}
out << "</PPointData>" << std::endl ;
// CELL DATA
out << "<PCellData>" << std::endl ;
out << "\t<PDataArray type=\"Int32\" Name=\"" << "material"
<<"\" NumberOfComponents=\"1\"/> " << std::endl ;
out << "</PCellData>" << std::endl ;
for (int ii=0; ii<nprocs; ii++)
{
// this one is generated without the path
std::string nfname=GenerateVTUFileName(ii);
out << "<Piece Source=\"" << nfname << "\"/>" << std::endl;
}
out << "</PUnstructuredGrid>" << std::endl;
out << "</VTKFile>" << std::endl;
out.close();
fname = GeneratePVTUPath()+"/"+GeneratePVTUFileName();
// add the pvtu file to the pvd_stream
pvd_stream << "<DataSet timestep=\"" << GetTime(); // GetCycle();
pvd_stream << "\" group=\"\" part=\"" << 0 << "\" file=\"";
pvd_stream << fname << "\"/>" << std::endl;
}
}
void ParaViewDataCollection::SaveDataVTU(std::ostream &out, int ref)
{
out << "<VTKFile type=\"UnstructuredGrid\" ";
out << " version=\"0.1\" byte_order=\"LittleEndian\">" << std::endl;
out << "<UnstructuredGrid>" << std::endl;
mesh->PrintVTU(out,ref);
// dump out the grid functions as point data
out << "<PointData >" << std::endl;
// save the grid functions
// iterate over all grid functions
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
SaveGFieldVTU(out,ref,it);
}
// iterate over all quadrature functions
// if the Quadrature functions are dumped as cell data
// the cycle should be moved before the grid functions
// and the PrintVTU CellData section should be open in the mesh dump
for (QFieldMapIterator it=q_field_map.begin(); it!=q_field_map.end(); ++it)
{
// save the quadrature functions
// this one is not implemented yet
SaveQFieldVTU(out,ref,it);
}
out << "</PointData>" << std::endl;
// close the mesh
out << "</Piece>" << std::endl; // close the piece open in the PrintVTU method
out << "</UnstructuredGrid>" << std::endl;
out << "</VTKFile>" << std::endl;
}
void ParaViewDataCollection::SaveQFieldVTU(std::ostream &out, int ref,
const QFieldMapIterator& it )
{
MFEM_WARNING("SaveQFieldVTU is wotk in progress - field name:"<<it->second);
}
void ParaViewDataCollection::SaveGFieldVTU(std::ostream &out, int ref_,
const FieldMapIterator& it)
{
RefinedGeometry *RefG;
Vector val;
DenseMatrix vval, pmat;
int vec_dim = it->second->VectorDim();
if (vec_dim == 1)
{
// scalar data
out << "<DataArray type=\"Float64\" Name=\"" << it->first;
out << "\" NumberOfComponents=\"1\" format=\"ascii\" >" << std::endl;
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
it->second->GetValues(i, RefG->RefPts, val, pmat);
for (int j = 0; j < val.Size(); j++)
{
out << val(j) << '\n';
}
}
}
else
{
// vector data
out << "<DataArray type=\"Float64\" Name=\"" << it->first;
out << "\" NumberOfComponents=\"" << vec_dim << "\" format=\"ascii\" >" <<
std::endl;
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
it->second->GetVectorValues(i, RefG->RefPts, vval, pmat);
for (int jj = 0; jj < vval.Width(); jj++)
{
for (int ii = 0; ii < vval.Height(); ii++)
{
out << vval(ii, jj) << ' ';
}
out << std::endl;
}
}
}
out << "</DataArray>" << std::endl;
out.flush();
}
int ParaViewDataCollection::create_directory(const std::string &dir_name)
{
// create directories recursively
const char path_delim = '/';
std::string::size_type pos = 0;
int err;
do
{
pos = dir_name.find(path_delim, pos+1);
std::string subdir = dir_name.substr(0, pos);
err = mkdir(subdir.c_str(), 0777);
err = (err && (errno != EEXIST)) ? 1 : 0;
}
while ( pos != std::string::npos );
return err;
}
#ifdef MFEM_USE_MPI
ParaViewDataCollection::ParaViewDataCollection(const std::string&
collection_name,
mfem::ParMesh *mesh_)
:DataCollection(collection_name,mesh_)
{
lcomm = mesh_->GetComm();
MPI_Comm_rank(lcomm, &myrank);
MPI_Comm_size(lcomm, &nprocs);
levels_of_detail = 1;
std::string dpath = GenerateCollectionPath();
std::string pvdname = dpath+"/"+GeneratePVDFileName();
int err = create_directory(dpath,myrank,lcomm);
if (err) { MFEM_ABORT("Cannot create the directory:"<<dpath);}
if (myrank==0)
{
pvd_stream.open(pvdname.c_str(),std::ios::out);
pvd_stream << "<?xml version=\"1.0\"?>" << std::endl;
pvd_stream << "<VTKFile type=\"Collection\" version=\"0.1\"" << std::endl;
pvd_stream << " byte_order=\"LittleEndian\"" << std::endl;
pvd_stream << " compressor=\"vtkZLibDataCompressor\">" << std::endl;
pvd_stream << "<Collection>" << std::endl;
}
}
int ParaViewDataCollection::create_directory(const std::string &dir_name,
int myid,
MPI_Comm lcomm_)
{
// create directories recursively
const char path_delim = '/';
std::string::size_type pos = 0;
int err;
// create the directories only on process 0
if (myid==0)
{
do
{
pos = dir_name.find(path_delim, pos+1);
std::string subdir = dir_name.substr(0, pos);
err = mkdir(subdir.c_str(), 0777);
err = (err && (errno != EEXIST)) ? 1 : 0;
}
while ( pos != std::string::npos );
}
// broadcast the error
MPI_Bcast(&err, 1, MPI_INT, 0, lcomm_);
return err;
}
void ParaViewDataCollection::SetMesh(MPI_Comm comm, mfem::Mesh *new_mesh)
{
DataCollection::SetMesh(new_mesh);
lcomm = comm;
MPI_Comm_rank(comm, &myrank);
MPI_Comm_size(comm, &nprocs);
}
#endif
} // end namespace MFEM
+74 -1
View File
@@ -19,6 +19,7 @@
#endif
#include <string>
#include <map>
#include <fstream>
namespace mfem
{
@@ -467,6 +468,78 @@ public:
virtual ~VisItDataCollection() {}
};
}
/// Helper class for ParaView visualization data
class ParaViewDataCollection : public DataCollection
{
private:
#ifdef MFEM_USE_MPI
MPI_Comm lcomm;
#endif
int myrank;
int nprocs;
int levels_of_detail;
std::fstream pvd_stream;
protected:
void SaveDataVTU(std::ostream &out, int ref);
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
void SaveQFieldVTU(std::ostream &out, int ref, const QFieldMapIterator& it);
std::string GenerateCollectionPath();
std::string GenerateVTUFileName();
std::string GenerateVTUFileName(int rank);
std::string GenerateVTUPath();
std::string GeneratePVDFileName();
std::string GeneratePVTUFileName();
std::string GeneratePVTUPath();
public:
/// Constructor. The collection name is used when saving the data.
/** If @a mesh_ is NULL, then the mesh can be set later by calling SetMesh().
The constructor works only in serial. */
ParaViewDataCollection(const std::string& collection_name,
mfem::Mesh *mesh_ = NULL);
#ifdef MFEM_USE_MPI
/// Construct a parallel ParaViewDataCollection.
/** Before saving the data collection, some parameters in the collection can
be adjusted, e.g. SetPadDigits(), SetPrefixPath(), etc. */
ParaViewDataCollection(const std::string& collection_name,
mfem::ParMesh *mesh_ = NULL);
#endif
virtual ~ParaViewDataCollection() override;
virtual void SetMesh(mfem::Mesh * new_mesh) override;
#ifdef MFEM_USE_MPI
/// Set/change the mesh associated with the collection.
virtual void SetMesh(MPI_Comm comm, mfem::Mesh *new_mesh) override;
#endif
/// Add a grid function to the collection
virtual void RegisterField(const std::string& field_name,
mfem::GridFunction *gf) override;
/// Set refinement levels - every element is uniformly split based on
/// levels_of_detail_
void SetLevelsOfDetail(int levels_of_detail_);
/// Save the collection - the directory name is constructed based on the
/// cycle value
virtual void Save() override;
/// Load the collection - not implemented in the ParaView writer
virtual void Load(int cycle_ = 0) override;
static int create_directory(const std::string &dir_name);
#ifdef MFEM_USE_MPI
static int create_directory(const std::string &dir_name, int myid,
MPI_Comm mycom);
#endif
};
}
#endif
+2
View File
@@ -18,6 +18,7 @@
#include "fe_coll.hpp"
#include "eltrans.hpp"
#include "coefficient.hpp"
#include "complex_fem.hpp"
#include "lininteg.hpp"
#include "nonlininteg.hpp"
#include "bilininteg.hpp"
@@ -32,6 +33,7 @@
#include "staticcond.hpp"
#include "tmop.hpp"
#include "tmop_tools.hpp"
#include "gslib.hpp"
#ifdef MFEM_USE_MPI
#include "pfespace.hpp"
+62 -52
View File
@@ -325,7 +325,7 @@ void FiniteElementSpace::GetEssentialVDofs(const Array<int> &bdr_attr_is_ess,
// mark possible hidden boundary edges in a non-conforming mesh, also
// local DOFs affected by boundary elements on other processors
if (mesh->ncmesh)
if (Nonconforming())
{
Array<int> bdr_verts, bdr_edges;
mesh->ncmesh->GetBoundaryClosure(bdr_attr_is_ess, bdr_verts, bdr_edges);
@@ -567,16 +567,16 @@ bool FiniteElementSpace::DofFinalizable(int dof, const Array<bool>& finalized,
return true;
}
void FiniteElementSpace::GetDegenerateFaceDofs(int index,
Array<int> &dofs) const
void FiniteElementSpace::GetDegenerateFaceDofs(int index, Array<int> &dofs,
Geometry::Type master_geom) const
{
// In NC meshes with prisms, a special constraint occurs where a prism edge
// is slave to a quadrilateral face. Rather than introduce a new edge-face
// constraint type, we handle such cases as degenerate face-face constraints,
// where the point-matrix rectangle has zero height. This method returns
// DOFs for the first edge of the rectangle, duplicated in the orthogonal
// direction, to resemble DOFs for a quadrilateral face. The extra DOFs are
// ignored by FiniteElementSpace::AddDependencies.
// In NC meshes with prisms/tets, a special constraint occurs where a
// prism/tet edge is slave to another element's face. Rather than introduce a
// new edge-face constraint type, we handle such cases as degenerate
// face-face constraints, where the point-matrix rectangle has zero height.
// This method returns DOFs for the first edge of the rectangle, duplicated
// in the orthogonal direction, to resemble DOFs for a quadrilateral face.
// The extra DOFs are ignored by FiniteElementSpace::AddDependencies.
Array<int> edof;
GetEdgeDofs(-1 - index, edof);
@@ -586,6 +586,8 @@ void FiniteElementSpace::GetDegenerateFaceDofs(int index,
int nn = 2*nv + ne;
dofs.SetSize(nn*nn);
if (!dofs.Size()) { return; }
dofs = edof[0];
// copy first two vertex DOFs
@@ -595,21 +597,23 @@ void FiniteElementSpace::GetDegenerateFaceDofs(int index,
dofs[nv+i] = edof[nv+i];
}
// copy first edge DOFs
int face_vert = Geometry::NumVerts[master_geom];
for (int i = 0; i < ne; i++)
{
dofs[4*nv + i] = edof[2*nv + i];
dofs[face_vert*nv + i] = edof[2*nv + i];
}
}
void
FiniteElementSpace::GetEntityDofs(int entity, int index, Array<int> &dofs) const
FiniteElementSpace::GetEntityDofs(int entity, int index, Array<int> &dofs,
Geometry::Type master_geom) const
{
switch (entity)
{
case 0: GetVertexDofs(index, dofs); break;
case 1: GetEdgeDofs(index, dofs); break;
case 2: (index >= 0) ? GetFaceDofs(index, dofs)
/* */ : GetDegenerateFaceDofs(index, dofs);
/* */ : GetDegenerateFaceDofs(index, dofs, master_geom);
}
}
@@ -662,7 +666,7 @@ void FiniteElementSpace::BuildConformingInterpolation() const
for (int si = master.slaves_begin; si < master.slaves_end; si++)
{
const NCMesh::Slave &slave = list.slaves[si];
GetEntityDofs(entity, slave.index, slave_dofs);
GetEntityDofs(entity, slave.index, slave_dofs, master.Geom());
if (!slave_dofs.Size()) { continue; }
slave.OrientedPointMatrix(T.GetPointMat());
@@ -954,7 +958,7 @@ void FiniteElementSpace::GetLocalRefinementMatrices(
const DenseTensor &pmats = rtrans.point_matrices[geom];
int nmat = pmats.SizeK();
int ldof = fe->GetDof(); // assuming the same FE everywhere
int ldof = fe->GetDof();
IsoparametricTransformation isotr;
isotr.SetIdentityTransformation(geom);
@@ -972,7 +976,8 @@ void FiniteElementSpace::GetLocalRefinementMatrices(
SparseMatrix* FiniteElementSpace::RefinementMatrix(int old_ndofs,
const Table* old_elem_dof)
{
MFEM_VERIFY(ndofs >= old_ndofs, "Previous space is not coarser.");
MFEM_VERIFY(GetNE() >= old_elem_dof->Size(),
"Previous mesh is not coarser.");
Mesh::GeometryList elem_geoms(*mesh);
@@ -990,10 +995,8 @@ FiniteElementSpace::RefinementOperator::RefinementOperator
: fespace(fespace)
, old_elem_dof(old_elem_dof)
{
const Mesh* mesh = fespace->GetMesh();
MFEM_VERIFY(mesh->ReduceInt(fespace->GetNDofs()) >=
mesh->ReduceInt(old_ndofs),
"Previous space is not coarser.");
MFEM_VERIFY(fespace->GetNE() >= old_elem_dof->Size(),
"Previous mesh is not coarser.");
width = old_ndofs * fespace->GetVDim();
height = fespace->GetVSize();
@@ -1255,16 +1258,11 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
GetLocalDerefinementMatrices(elem_geoms[i], localR[elem_geoms[i]]);
}
SparseMatrix *R;
if (elem_geoms.Size() == 1)
{
R = new SparseMatrix(ndofs*vdim, old_ndofs*vdim,
localR[elem_geoms[0]].SizeI());
}
else
{
R = new SparseMatrix(ndofs*vdim, old_ndofs*vdim);
}
SparseMatrix *R = (elem_geoms.Size() != 1)
? new SparseMatrix(ndofs*vdim, old_ndofs*vdim) // variable row size
: new SparseMatrix(ndofs*vdim, old_ndofs*vdim,
localR[elem_geoms[0]].SizeI());
Array<int> mark(R->Height());
mark = 0;
@@ -1277,7 +1275,7 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
for (int k = 0; k < dtrans.embeddings.Size(); k++)
{
const Embedding &emb = dtrans.embeddings[k];
const Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
@@ -1290,7 +1288,7 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
for (int i = 0; i < lR.Height(); i++)
{
if (lR(i, 0) == infinity()) { continue; }
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -1308,7 +1306,8 @@ SparseMatrix* FiniteElementSpace::DerefinementMatrix(int old_ndofs,
MFEM_VERIFY(num_marked == R->Height(),
"internal error: not all rows of R were set.");
if (elem_geoms.Size() != 1) { R->Finalize(); }
R->Finalize(); // no-op if fixed width
return R;
}
@@ -2691,57 +2690,68 @@ L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
for (int iel=0; iel<ne; ++iel)
const int NE = ne;
const int VDIM = vdim;
const int NDOF = ndof;
const bool BYVDIM = byvdim;
auto d_x = x.Read();
auto d_y = y.Write();
MFEM_FORALL(iel, NE,
{
for (int vd=0; vd<vdim; ++vd)
for (int vd=0; vd<VDIM; ++vd)
{
for (int idof=0; idof<ndof; ++idof)
for (int idof=0; idof<NDOF; ++idof)
{
// E-vector dimensions (dofs, vdim, elements)
// L-vector dimensions: byVDIM: (vdim, dofs, element)
// byNODES: (dofs, elements, vdim)
int yidx = iel*vdim*ndof + vd*ndof + idof;
int yidx = iel*VDIM*NDOF + vd*NDOF + idof;
int xidx;
if (byvdim)
if (BYVDIM)
{
xidx = iel*ndof*vdim + idof*vdim + vd;
xidx = iel*NDOF*VDIM + idof*VDIM + vd;
}
else
{
xidx = vd*ne*ndof + iel*ndof + idof;
xidx = vd*NE*NDOF + iel*NDOF + idof;
}
y[yidx] = x[xidx];
d_y[yidx] = d_x[xidx];
}
}
}
});
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int NE = ne;
const int VDIM = vdim;
const int NDOF = ndof;
const bool BYVDIM = byvdim;
auto d_x = x.Read();
auto d_y = y.Write();
// Since this restriction is a permutation, the transpose is the inverse
for (int iel=0; iel<ne; ++iel)
MFEM_FORALL(iel, NE,
{
for (int vd=0; vd<vdim; ++vd)
for (int vd=0; vd<VDIM; ++vd)
{
for (int idof=0; idof<ndof; ++idof)
for (int idof=0; idof<NDOF; ++idof)
{
// E-vector dimensions (dofs, vdim, elements)
// L-vector dimensions: byVDIM: (vdim, dofs, element)
// byNODES: (dofs, elements, vdim)
int xidx = iel*vdim*ndof + vd*ndof + idof;
int xidx = iel*VDIM*NDOF + vd*NDOF + idof;
int yidx;
if (byvdim)
if (BYVDIM)
{
yidx = iel*ndof*vdim + idof*vdim + vd;
yidx = iel*NDOF*VDIM + idof*VDIM + vd;
}
else
{
yidx = vd*ne*ndof + iel*ndof + idof;
yidx = vd*NE*NDOF + iel*NDOF + idof;
}
y[yidx] = x[xidx];
d_y[yidx] = d_x[xidx];
}
}
}
});
}
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
+4 -2
View File
@@ -145,9 +145,11 @@ protected:
{ return (dof >= 0) ? (sign = 1, dof) : (sign = -1, (-1 - dof)); }
/// Helper to get vertex, edge or face DOFs (entity=0,1,2 resp.).
void GetEntityDofs(int entity, int index, Array<int> &dofs) const;
void GetEntityDofs(int entity, int index, Array<int> &dofs,
Geometry::Type master_geom = Geometry::INVALID) const;
// Get degenerate face DOFs: see explanation in method implementation.
void GetDegenerateFaceDofs(int index, Array<int> &dofs) const;
void GetDegenerateFaceDofs(int index, Array<int> &dofs,
Geometry::Type master_geom) const;
/// Calculate the cP and cR matrices for a nonconforming mesh.
void BuildConformingInterpolation() const;
+220
View File
@@ -0,0 +1,220 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "gslib.hpp"
#ifdef MFEM_USE_GSLIB
// Ignore warnings from the gslib header (GCC version)
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
FindPointsGSLIB::FindPointsGSLIB()
: mesh(NULL), gsl_mesh(), fdata2D(NULL), fdata3D(NULL), dim(-1)
{
gsl_comm = new comm;
#ifdef MFEM_USE_MPI
MPI_Init(NULL, NULL);
MPI_Comm comm = MPI_COMM_WORLD;;
comm_init(gsl_comm, comm);
#else
comm_init(gsl_comm, 0);
#endif
}
FindPointsGSLIB::~FindPointsGSLIB()
{
delete gsl_comm;
}
#ifdef MFEM_USE_MPI
FindPointsGSLIB::FindPointsGSLIB(MPI_Comm _comm)
: mesh(NULL), gsl_mesh(), fdata2D(NULL), fdata3D(NULL), dim(-1)
{
gsl_comm = new comm;
comm_init(gsl_comm, _comm);
}
#endif
void FindPointsGSLIB::Setup(Mesh &m, double bb_t, double newt_tol, int npt_max)
{
MFEM_VERIFY(m.GetNodes() != NULL, "Mesh nodes are required.");
mesh = &m;
const GridFunction *nodes = mesh->GetNodes();
const FiniteElementSpace *fes = nodes->FESpace();
dim = mesh->Dimension();
const int NE = mesh->GetNE(),
dof_cnt = fes->GetFE(0)->GetDof(),
pts_cnt = NE * dof_cnt;
gsl_mesh.SetSize(dim * pts_cnt);
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fes->GetFE(0));
const Array<int> &dof_map = tbe->GetDofMap();
DenseMatrix pos(dof_cnt, dim);
Vector posV(pos.Data(), dof_cnt * dim);
Array<int> xdofs(dof_cnt * dim);
int pt_id = 0;
for (int i = 0; i < NE; i++)
{
fes->GetElementVDofs(i, xdofs);
nodes->GetSubVector(xdofs, posV);
for (int j = 0; j < dof_cnt; j++)
{
for (int d = 0; d < dim; d++)
{
gsl_mesh(pts_cnt * d + pt_id) = pos(dof_map[j], d);
}
pt_id++;
}
}
const unsigned dof1D = fes->GetFE(0)->GetOrder() + 1;
if (dim == 2)
{
unsigned nr[2] = {dof1D, dof1D};
unsigned mr[2] = {2*dof1D, 2*dof1D};
double * const elx[2] = { &gsl_mesh(0), &gsl_mesh(pts_cnt) };
fdata2D = findpts_setup_2(gsl_comm, elx, nr, NE, mr, bb_t,
pts_cnt, pts_cnt, npt_max, newt_tol);
}
else
{
unsigned nr[3] = {dof1D, dof1D, dof1D};
unsigned mr[3] = {2*dof1D, 2*dof1D, 2*dof1D};
double * const elx[3] =
{ &gsl_mesh(0), &gsl_mesh(pts_cnt), &gsl_mesh(2*pts_cnt) };
fdata3D = findpts_setup_3(gsl_comm, elx, nr, NE, mr, bb_t,
pts_cnt, pts_cnt, npt_max, newt_tol);
}
}
void FindPointsGSLIB::FindPoints(Vector &point_pos, Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist)
{
const int points_cnt = point_pos.Size() / dim;
if (dim == 2)
{
const double *xv_base[2];
xv_base[0] = point_pos.GetData();
xv_base[1] = point_pos.GetData() + points_cnt;
unsigned xv_stride[2];
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
findpts_2(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata2D);
}
else
{
const double *xv_base[3];
xv_base[0] = point_pos.GetData();
xv_base[1] = point_pos.GetData() + points_cnt;
xv_base[2] = point_pos.GetData() + 2*points_cnt;
unsigned xv_stride[3];
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
xv_stride[2] = sizeof(double);
findpts_3(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata3D);
}
}
void FindPointsGSLIB::Interpolate(Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, const GridFunction &field_in,
Vector &field_out)
{
Vector node_vals;
GetNodeValues(field_in, node_vals);
const int points_cnt = ref_pos.Size() / dim;
if (dim==2)
{
findpts_eval_2(field_out.GetData(), sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata2D);
}
else
{
findpts_eval_3(field_out.GetData(), sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata3D);
}
}
void FindPointsGSLIB::FreeData()
{
(dim == 2) ? findpts_free_2(fdata2D) : findpts_free_3(fdata3D);
}
void FindPointsGSLIB::GetNodeValues(const GridFunction &gf_in,
Vector &node_vals)
{
MFEM_ASSERT(gf_in.FESpace()->GetVDim() == 1, "Scalar function expected.");
const GridFunction *nodes = mesh->GetNodes();
const FiniteElementSpace *fes = nodes->FESpace();
const IntegrationRule &ir = fes->GetFE(0)->GetNodes();
const int NE = mesh->GetNE(), dof_cnt = ir.GetNPoints();
node_vals.SetSize(NE * dof_cnt);
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fes->GetFE(0));
const Array<int> &dof_map = tbe->GetDofMap();
int pt_id = 0;
Vector vals_el;
for (int i = 0; i < NE; i++)
{
gf_in.GetValues(i, ir, vals_el);
for (int j = 0; j < dof_cnt; j++)
{
node_vals(pt_id++) = vals_el(dof_map[j]);
}
}
}
} // namespace mfem
#endif // MFEM_USE_GSLIB
+107
View File
@@ -0,0 +1,107 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_GSLIB
#define MFEM_GSLIB
#include "../config/config.hpp"
#include "gridfunc.hpp"
#ifdef MFEM_USE_GSLIB
struct comm;
struct findpts_data_2;
struct findpts_data_3;
namespace mfem
{
class FindPointsGSLIB
{
protected:
Mesh *mesh;
Vector gsl_mesh;
struct findpts_data_2 *fdata2D;
struct findpts_data_3 *fdata3D;
int dim;
struct comm *gsl_comm;
void GetNodeValues(const GridFunction &gf_in, Vector &node_vals);
public:
FindPointsGSLIB();
#ifdef MFEM_USE_MPI
FindPointsGSLIB(MPI_Comm _comm);
#endif
~FindPointsGSLIB();
/** Initializes the internal mesh in gslib, by sending the positions of the
Gauss-Lobatto nodes of the input Mesh object @a m.
Note: not tested with periodic (DG meshes).
Note: the input mesh @a m must have Nodes set.
@param[in] m Input mesh.
@param[in] bb_t Relative size of bounding box around each element.
@param[in] newt_tol Newton tolerance for the gslib search methods.
@param[in] npt_max Number of points for simultaneous iteration. This
alters performance and memory footprint. */
void Setup(Mesh &m, double bb_t, double newt_tol, int npt_max);
/** Searches positions given in physical space by @a point_pos. All output
Arrays and Vectors are expected to have the correct size.
@param[in] point_pos Positions to be found. Must by ordered by nodes
(XXX...,YYY...,ZZZ).
@param[out] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[out] proc_ids MPI proc ids where the points were found.
@param[out] elem_ids Element ids where the points were found.
@param[out] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[out] dist Distance between the seeked and the found point
in physical space. */
void FindPoints(Vector &point_pos, Array<unsigned int> &codes,
Array<unsigned int> &proc_ids, Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist);
/** Interpolation of field values at prescribed reference space positions.
@param[in] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[in] proc_ids MPI proc ids where the points were found.
@param[in] elem_ids Element ids where the points were found.
@param[in] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[in] field_in Function values that will be interpolated on the
reference positions. Note: it is assumed that
@a field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. */
void Interpolate(Array<unsigned int> &codes, Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids, Vector &ref_pos,
const GridFunction &field_in, Vector &field_out);
/** Cleans up memory allocated internally by gslib.
Note that in parallel, this must be called before MPI_Finalize(), as
it calls MPI_Comm_free() for internal gslib communicators. */
void FreeData();
};
} // namespace mfem
#endif //MFEM_USE_GSLIB
#endif //MFEM_GSLIB guard
+12
View File
@@ -48,6 +48,8 @@ IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry)
ip.weight = ipx.weight * ipy.weight;
}
}
SetPointIndices();
}
IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry,
@@ -76,6 +78,8 @@ IntegrationRule::IntegrationRule(IntegrationRule &irx, IntegrationRule &iry,
}
}
}
SetPointIndices();
}
const Array<double> &IntegrationRule::GetWeights() const
@@ -91,6 +95,14 @@ const Array<double> &IntegrationRule::GetWeights() const
return weights;
}
void IntegrationRule::SetPointIndices()
{
for (int i = 0; i < Size(); i++)
{
IntPoint(i).index = i;
}
}
void IntegrationRule::GrundmannMollerSimplexRule(int s, int n)
{
// for pow on older compilers
+10 -2
View File
@@ -26,8 +26,13 @@ class IntegrationPoint
{
public:
double x, y, z, weight;
int index;
void Init() { x = y = z = weight = 0.0; }
void Init(int const i)
{
x = y = z = weight = 0.0;
index = i;
}
void Set(const double *p, const int dim)
{
@@ -91,6 +96,9 @@ private:
by request with the method GetWeights(). */
mutable Array<double> weights;
/// Sets the indices of each quadrature point on initialization.
void SetPointIndices();
/// Define n-simplex rule (triangle/tetrahedron for n=2/3) of order (2s+1)
void GrundmannMollerSimplexRule(int s, int n = 3);
@@ -215,7 +223,7 @@ public:
{
for (int i = 0; i < this->Size(); i++)
{
(*this)[i].Init();
(*this)[i].Init(i);
}
}
+170
View File
@@ -0,0 +1,170 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../fem/gridfunc.hpp"
#include <sys/types.h>
#include <sys/stat.h>
#ifndef _WIN32
typedef struct stat struct_stat;
#else
#define stat(dir, buf) _stat(dir, buf)
#define S_ISDIR(mode) _S_IFDIR(mode)
typedef struct _stat struct_stat;
#endif
namespace mfem
{
namespace internal
{
extern Ceed ceed;
std::string ceed_path;
}
void InitCeedCoeff(Coefficient* Q, CeedData* ptr)
{
if (ConstantCoefficient* coeff = dynamic_cast<ConstantCoefficient*>(Q))
{
CeedConstCoeff* ceedCoeff = new CeedConstCoeff{coeff->constant};
ptr->coeff_type = CeedCoeff::Const;
ptr->coeff = (void*)ceedCoeff;
}
else if (GridFunctionCoefficient* coeff =
dynamic_cast<GridFunctionCoefficient*>(Q))
{
CeedGridCoeff* ceedCoeff = new CeedGridCoeff;
ceedCoeff->coeff = coeff->GetGridFunction();
ptr->coeff_type = CeedCoeff::Grid;
ptr->coeff = (void*)ceedCoeff;
}
else
{
MFEM_ABORT("This type of Coefficient is not supported.");
}
}
void InitCeedTensorBasisAndRestriction(const mfem::FiniteElementSpace &fes,
const mfem::IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr)
{
mfem::Mesh *mesh = fes.GetMesh();
const mfem::FiniteElement *fe = fes.GetFE(0);
const int order = fes.GetOrder(0);
mfem::Array<int> dof_map;
switch (mesh->Dimension())
{
case 1:
{
const mfem::H1_SegmentElement *h1_fe =
dynamic_cast<const mfem::H1_SegmentElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
case 2:
{
const mfem::H1_QuadrilateralElement *h1_fe =
dynamic_cast<const mfem::H1_QuadrilateralElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
case 3:
{
const mfem::H1_HexahedronElement *h1_fe =
dynamic_cast<const mfem::H1_HexahedronElement *>(fe);
MFEM_VERIFY(h1_fe, "invalid FE");
h1_fe->GetDofMap().Copy(dof_map);
break;
}
}
const mfem::FiniteElement *fe1d =
fes.FEColl()->FiniteElementForGeometry(mfem::Geometry::SEGMENT);
mfem::DenseMatrix shape1d(fe1d->GetDof(), ir.GetNPoints());
mfem::DenseMatrix grad1d(fe1d->GetDof(), ir.GetNPoints());
mfem::Vector qref1d(ir.GetNPoints()), qweight1d(ir.GetNPoints());
mfem::Vector shape_i(shape1d.Height());
mfem::DenseMatrix grad_i(grad1d.Height(), 1);
const mfem::H1_SegmentElement *h1_fe1d =
dynamic_cast<const mfem::H1_SegmentElement *>(fe1d);
MFEM_VERIFY(h1_fe1d, "invalid FE");
const mfem::Array<int> &dof_map_1d = h1_fe1d->GetDofMap();
for (int i = 0; i < ir.GetNPoints(); i++)
{
const mfem::IntegrationPoint &ip = ir.IntPoint(i);
qref1d(i) = ip.x;
qweight1d(i) = ip.weight;
fe1d->CalcShape(ip, shape_i);
fe1d->CalcDShape(ip, grad_i);
for (int j = 0; j < shape1d.Height(); j++)
{
shape1d(j, i) = shape_i(dof_map_1d[j]);
grad1d(j, i) = grad_i(dof_map_1d[j], 0);
}
}
CeedBasisCreateTensorH1(ceed, mesh->Dimension(), fes.GetVDim(), order + 1,
ir.GetNPoints(), shape1d.GetData(),
grad1d.GetData(), qref1d.GetData(),
qweight1d.GetData(), basis);
const mfem::Table &el_dof = fes.GetElementToDofTable();
mfem::Array<int> tp_el_dof(el_dof.Size_of_connections());
for (int i = 0; i < mesh->GetNE(); i++)
{
const int el_offset = fe->GetDof() * i;
for (int j = 0; j < fe->GetDof(); j++)
{
tp_el_dof[j + el_offset] = el_dof.GetJ()[dof_map[j] + el_offset];
}
}
CeedElemRestrictionCreate(ceed, mesh->GetNE(), fe->GetDof(),
fes.GetNDofs(), fes.GetVDim(), CEED_MEM_HOST, CEED_COPY_VALUES,
tp_el_dof.GetData(), restr);
}
const std::string &GetCeedPath()
{
if (internal::ceed_path.empty())
{
const char *install_dir = MFEM_INSTALL_DIR "/include/mfem/fem/libceed";
const char *source_dir = MFEM_SOURCE_DIR "/fem/libceed";
struct_stat m_stat;
if (stat(install_dir, &m_stat) == 0 && S_ISDIR(m_stat.st_mode))
{
internal::ceed_path = install_dir;
}
else if (stat(source_dir, &m_stat) == 0 && S_ISDIR(m_stat.st_mode))
{
internal::ceed_path = source_dir;
}
else
{
MFEM_ABORT("Cannot find libCEED kernels in MFEM_INSTALL_DIR or "
"MFEM_SOURCE_DIR");
}
// Could be useful for debugging:
// mfem::out << "Using libCEED dir: " << internal::ceed_path << std::endl;
}
return internal::ceed_path;
}
} // namespace mfem
#endif // MFEM_USE_CEED
+132
View File
@@ -0,0 +1,132 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_HPP
#define MFEM_LIBCEED_HPP
#include "../../config/config.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include <ceed.h>
namespace mfem
{
class FiniteElementSpace;
class GridFunction;
class IntegrationRule;
class Coefficient;
namespace internal { extern Ceed ceed; } // defined in device.cpp
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
enum class CeedCoeff { Const, Grid };
struct CeedConstCoeff
{
double val;
};
struct CeedGridCoeff
{
GridFunction* coeff;
CeedBasis basis;
CeedElemRestriction restr;
CeedVector coeffVector;
};
struct CeedData
{
CeedOperator build_oper, oper;
CeedBasis basis, mesh_basis;
CeedElemRestriction restr, mesh_restr, restr_i, mesh_restr_i;
CeedQFunction apply_qfunc, build_qfunc;
CeedVector node_coords, rho;
CeedCoeff coeff_type;
void* coeff;
BuildContext build_ctx;
CeedVector u, v;
~CeedData()
{
CeedOperatorDestroy(&build_oper);
CeedOperatorDestroy(&oper);
CeedBasisDestroy(&basis);
CeedBasisDestroy(&mesh_basis);
CeedElemRestrictionDestroy(&restr);
CeedElemRestrictionDestroy(&mesh_restr);
CeedElemRestrictionDestroy(&restr_i);
CeedElemRestrictionDestroy(&mesh_restr_i);
CeedQFunctionDestroy(&apply_qfunc);
CeedQFunctionDestroy(&build_qfunc);
CeedVectorDestroy(&node_coords);
CeedVectorDestroy(&rho);
if (coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* c = (CeedGridCoeff*)coeff;
CeedBasisDestroy(&c->basis);
CeedElemRestrictionDestroy(&c->restr);
CeedVectorDestroy(&c->coeffVector);
delete c;
}
else
{
delete (CeedConstCoeff*)coeff;
}
CeedVectorDestroy(&u);
CeedVectorDestroy(&v);
}
};
/** @brief Identifies the type of coefficient of the Integrator to initialize
accordingly the CeedData. */
void InitCeedCoeff(Coefficient* Q, CeedData* ptr);
/// Initialize a tensor CeedBasis and a CeedElemRestriction
void InitCeedTensorBasisAndRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr);
/// Return the path to the libCEED q-function headers.
const std::string &GetCeedPath();
/** @brief Function that determines if a CEED kernel should be used, based on
the current mfem::Device configuration. */
inline bool DeviceCanUseCeed()
{
return Device::Allows(Backend::CEED_CUDA) ||
(Device::Allows(Backend::CEED_CPU) &&
!Device::Allows(Backend::DEVICE_MASK|Backend::OMP_MASK));
}
} // namespace mfem
#else // MFEM_USE_CEED
namespace mfem
{
inline bool DeviceCanUseCeed()
{
return false;
}
} // namespace mfem
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_HPP
+150
View File
@@ -0,0 +1,150 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "diffusion.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../mesh/mesh.hpp"
#include "../../fem/gridfunc.hpp"
#include "ceed.hpp"
#include "diffusion.h"
namespace mfem
{
void CeedPADiffusionAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &irm, CeedData& ceedData)
{
Ceed ceed(internal::ceed);
mfem::Mesh *mesh = fes.GetMesh();
const int ir_order = irm.GetOrder();
const mfem::IntegrationRule &ir =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir_order);
CeedInt nqpts, nelem = mesh->GetNE(), dim = mesh->SpaceDimension();
mesh->EnsureNodes();
InitCeedTensorBasisAndRestriction(fes, ir, ceed, &ceedData.basis,
&ceedData.restr);
const mfem::FiniteElementSpace *mesh_fes = mesh->GetNodalFESpace();
MFEM_VERIFY(mesh_fes, "the Mesh has no nodal FE space");
InitCeedTensorBasisAndRestriction(*mesh_fes, ir, ceed, &ceedData.mesh_basis,
&ceedData.mesh_restr);
CeedBasisGetNumQuadraturePoints(ceedData.basis, &nqpts);
CeedElemRestrictionCreateIdentity(ceed, nelem, nqpts,
nqpts * nelem, dim * (dim + 1) / 2, &ceedData.restr_i);
CeedElemRestrictionCreateIdentity(ceed, nelem, nqpts,
nqpts * nelem, 1, &ceedData.mesh_restr_i);
CeedVectorCreate(ceed, mesh->GetNodes()->Size(), &ceedData.node_coords);
CeedVectorSetArray(ceedData.node_coords, CEED_MEM_HOST, CEED_USE_POINTER,
mesh->GetNodes()->GetData());
CeedVectorCreate(ceed, nelem * nqpts * dim * (dim + 1) / 2, &ceedData.rho);
// Context data to be passed to the 'f_build_diff' Q-function.
ceedData.build_ctx.dim = mesh->Dimension();
ceedData.build_ctx.space_dim = mesh->SpaceDimension();
std::string diff_qf_file = GetCeedPath() + "/diffusion.h";
std::string diff_qf;
// Create the Q-function that builds the diff operator (i.e. computes its
// quadrature data) and set its context data.
switch (ceedData.coeff_type)
{
case CeedCoeff::Const:
diff_qf = diff_qf_file + ":f_build_diff_const";
CeedQFunctionCreateInterior(ceed, 1, f_build_diff_const,
diff_qf.c_str(),
&ceedData.build_qfunc);
ceedData.build_ctx.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
break;
case CeedCoeff::Grid:
diff_qf = diff_qf_file + ":f_build_diff_grid";
CeedQFunctionCreateInterior(ceed, 1, f_build_diff_grid,
diff_qf.c_str(),
&ceedData.build_qfunc);
CeedQFunctionAddInput(ceedData.build_qfunc, "coeff", 1, CEED_EVAL_INTERP);
break;
default:
MFEM_ABORT("This coeff_type is not handled");
}
CeedQFunctionAddInput(ceedData.build_qfunc, "dx", dim * dim, CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.build_qfunc, "weights", 1, CEED_EVAL_WEIGHT);
CeedQFunctionAddOutput(ceedData.build_qfunc, "rho", dim * (dim + 1) / 2,
CEED_EVAL_NONE);
CeedQFunctionSetContext(ceedData.build_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the operator that builds the quadrature data for the diff operator.
CeedOperatorCreate(ceed, ceedData.build_qfunc, NULL, NULL,
&ceedData.build_oper);
CeedTransposeMode lmode = CEED_NOTRANSPOSE;
if (mesh_fes->GetOrdering()==Ordering::byVDIM)
{
lmode = CEED_TRANSPOSE;
}
if (ceedData.coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* ceedCoeff = (CeedGridCoeff*)ceedData.coeff;
InitCeedTensorBasisAndRestriction(*ceedCoeff->coeff->FESpace(), ir, ceed,
&ceedCoeff->basis,
&ceedCoeff->restr);
CeedVectorCreate(ceed, ceedCoeff->coeff->FESpace()->GetNDofs(),
&ceedCoeff->coeffVector);
CeedVectorSetArray(ceedCoeff->coeffVector, CEED_MEM_HOST, CEED_USE_POINTER,
ceedCoeff->coeff->GetData());
CeedOperatorSetField(ceedData.build_oper, "coeff", ceedCoeff->restr,
CEED_NOTRANSPOSE, ceedCoeff->basis, ceedCoeff->coeffVector);
}
CeedOperatorSetField(ceedData.build_oper, "dx", ceedData.mesh_restr, lmode,
ceedData.mesh_basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.build_oper, "weights", ceedData.mesh_restr_i,
CEED_NOTRANSPOSE,
ceedData.mesh_basis, CEED_VECTOR_NONE);
CeedOperatorSetField(ceedData.build_oper, "rho", ceedData.restr_i,
CEED_NOTRANSPOSE,
CEED_BASIS_COLLOCATED, CEED_VECTOR_ACTIVE);
// Compute the quadrature data for the diff operator.
CeedOperatorApply(ceedData.build_oper, ceedData.node_coords, ceedData.rho,
CEED_REQUEST_IMMEDIATE);
// Create the Q-function that defines the action of the diff operator.
diff_qf = diff_qf_file + ":f_apply_diff";
CeedQFunctionCreateInterior(ceed, 1, f_apply_diff,
diff_qf.c_str(),
&ceedData.apply_qfunc);
CeedQFunctionAddInput(ceedData.apply_qfunc, "u", dim, CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.apply_qfunc, "rho", dim * (dim + 1) / 2,
CEED_EVAL_NONE);
CeedQFunctionAddOutput(ceedData.apply_qfunc, "v", dim, CEED_EVAL_GRAD);
CeedQFunctionSetContext(ceedData.apply_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the diff operator.
CeedOperatorCreate(ceed, ceedData.apply_qfunc, NULL, NULL, &ceedData.oper);
CeedOperatorSetField(ceedData.oper, "u", ceedData.restr, CEED_NOTRANSPOSE,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.oper, "rho", ceedData.restr_i, CEED_NOTRANSPOSE,
CEED_BASIS_COLLOCATED, ceedData.rho);
CeedOperatorSetField(ceedData.oper, "v", ceedData.restr, CEED_NOTRANSPOSE,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.u);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
} // namespace mfem
#endif // MFEM_USE_CEED
+200
View File
@@ -0,0 +1,200 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
/// libCEED Q-function for building quadrature data for a diffusion operator with a constant coefficient
CEED_QFUNCTION(f_build_diff_const)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext*)ctx;
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
//
// At every quadrature point, compute qw/det(J).adj(J).adj(J)^T and store
// the symmetric part of the result.
const CeedScalar coeff = bc->coeff;
const CeedScalar *J = in[0], *qw = in[1];
CeedScalar *qd = out[0];
switch (bc->dim + 10 * bc->space_dim)
{
case 11:
for (CeedInt i = 0; i < Q; i++)
{
qd[i] = coeff * qw[i] / J[i];
}
break;
case 22:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 2 qd: 0 1 adj(J): J22 -J12
// 1 3 1 2 -J21 J11
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J12 = J[i + Q * 2];
const CeedScalar J22 = J[i + Q * 3];
const CeedScalar w = qw[i] / (J11 * J22 - J21 * J12);
qd[i + Q * 0] = coeff * w * (J12 * J12 + J22 * J22);
qd[i + Q * 1] = - coeff * w * (J11 * J12 + J21 * J22);
qd[i + Q * 2] = coeff * w * (J11 * J11 + J21 * J21);
}
break;
case 33:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 3 6 qd: 0 1 2
// 1 4 7 1 3 4
// 2 5 8 2 4 5
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J31 = J[i + Q * 2];
const CeedScalar J12 = J[i + Q * 3];
const CeedScalar J22 = J[i + Q * 4];
const CeedScalar J32 = J[i + Q * 5];
const CeedScalar J13 = J[i + Q * 6];
const CeedScalar J23 = J[i + Q * 7];
const CeedScalar J33 = J[i + Q * 8];
const CeedScalar A11 = J22 * J33 - J23 * J32;
const CeedScalar A12 = J13 * J32 - J12 * J33;
const CeedScalar A13 = J12 * J23 - J13 * J22;
const CeedScalar A21 = J23 * J31 - J21 * J33;
const CeedScalar A22 = J11 * J33 - J13 * J31;
const CeedScalar A23 = J13 * J21 - J11 * J23;
const CeedScalar A31 = J21 * J32 - J22 * J31;
const CeedScalar A32 = J12 * J31 - J11 * J32;
const CeedScalar A33 = J11 * J22 - J12 * J21;
const CeedScalar w = qw[i] / (J11 * A11 + J21 * A12 + J31 * A13);
qd[i + Q * 0] = coeff * w * (A11 * A11 + A12 * A12 + A13 * A13);
qd[i + Q * 1] = coeff * w * (A11 * A21 + A12 * A22 + A13 * A23);
qd[i + Q * 2] = coeff * w * (A11 * A31 + A12 * A32 + A13 * A33);
qd[i + Q * 3] = coeff * w * (A21 * A21 + A22 * A22 + A23 * A23);
qd[i + Q * 4] = coeff * w * (A21 * A31 + A22 * A32 + A23 * A33);
qd[i + Q * 5] = coeff * w * (A31 * A31 + A32 * A32 + A33 * A33);
}
break;
}
return 0;
}
/// libCEED Q-function for building quadrature data for a diffusion operator with a grid function coefficient
CEED_QFUNCTION(f_build_diff_grid)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext *)ctx;
// in[1] is Jacobians with shape [dim, nc=dim, Q]
// in[2] is quadrature weights, size (Q)
//
// At every quadrature point, compute qw/det(J).adj(J).adj(J)^T and store
// the symmetric part of the result.
const CeedScalar *c = in[0], *J = in[1], *qw = in[2];
CeedScalar *qd = out[0];
switch (bc->dim + 10 * bc->space_dim)
{
case 11:
for (CeedInt i = 0; i < Q; i++)
{
qd[i] = c[i] * qw[i] / J[i];
}
break;
case 22:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 2 qd: 0 1 adj(J): J22 -J12
// 1 3 1 2 -J21 J11
const CeedScalar coeff = c[i];
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J12 = J[i + Q * 2];
const CeedScalar J22 = J[i + Q * 3];
const CeedScalar w = qw[i] / (J11 * J22 - J21 * J12);
qd[i + Q * 0] = coeff * w * (J12 * J12 + J22 * J22);
qd[i + Q * 1] = - coeff * w * (J11 * J12 + J21 * J22);
qd[i + Q * 2] = coeff * w * (J11 * J11 + J21 * J21);
}
break;
case 33:
for (CeedInt i = 0; i < Q; i++)
{
// J: 0 3 6 qd: 0 1 2
// 1 4 7 1 3 4
// 2 5 8 2 4 5
const CeedScalar coeff = c[i];
const CeedScalar J11 = J[i + Q * 0];
const CeedScalar J21 = J[i + Q * 1];
const CeedScalar J31 = J[i + Q * 2];
const CeedScalar J12 = J[i + Q * 3];
const CeedScalar J22 = J[i + Q * 4];
const CeedScalar J32 = J[i + Q * 5];
const CeedScalar J13 = J[i + Q * 6];
const CeedScalar J23 = J[i + Q * 7];
const CeedScalar J33 = J[i + Q * 8];
const CeedScalar A11 = J22 * J33 - J23 * J32;
const CeedScalar A12 = J13 * J32 - J12 * J33;
const CeedScalar A13 = J12 * J23 - J13 * J22;
const CeedScalar A21 = J23 * J31 - J21 * J33;
const CeedScalar A22 = J11 * J33 - J13 * J31;
const CeedScalar A23 = J13 * J21 - J11 * J23;
const CeedScalar A31 = J21 * J32 - J22 * J31;
const CeedScalar A32 = J12 * J31 - J11 * J32;
const CeedScalar A33 = J11 * J22 - J12 * J21;
const CeedScalar w = qw[i] / (J11 * A11 + J21 * A12 + J31 * A13);
qd[i + Q * 0] = coeff * w * (A11 * A11 + A12 * A12 + A13 * A13);
qd[i + Q * 1] = coeff * w * (A11 * A21 + A12 * A22 + A13 * A23);
qd[i + Q * 2] = coeff * w * (A11 * A31 + A12 * A32 + A13 * A33);
qd[i + Q * 3] = coeff * w * (A21 * A21 + A22 * A22 + A23 * A23);
qd[i + Q * 4] = coeff * w * (A21 * A31 + A22 * A32 + A23 * A33);
qd[i + Q * 5] = coeff * w * (A31 * A31 + A32 * A32 + A33 * A33);
}
break;
}
return 0;
}
/// libCEED Q-function for applying a diff operator
CEED_QFUNCTION(f_apply_diff)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
BuildContext *bc = (BuildContext *)ctx;
// in[0], out[0] have shape [dim, nc=1, Q]
const CeedScalar *ug = in[0], *qd = in[1];
CeedScalar *vg = out[0];
switch (bc->dim)
{
case 1:
for (CeedInt i = 0; i < Q; i++)
{
vg[i] = ug[i] * qd[i];
}
break;
case 2:
for (CeedInt i = 0; i < Q; i++)
{
const CeedScalar ug0 = ug[i + Q * 0];
const CeedScalar ug1 = ug[i + Q * 1];
vg[i + Q * 0] = qd[i + Q * 0] * ug0 + qd[i + Q * 1] * ug1;
vg[i + Q * 1] = qd[i + Q * 1] * ug0 + qd[i + Q * 2] * ug1;
}
break;
case 3:
for (CeedInt i = 0; i < Q; i++)
{
const CeedScalar ug0 = ug[i + Q * 0];
const CeedScalar ug1 = ug[i + Q * 1];
const CeedScalar ug2 = ug[i + Q * 2];
vg[i + Q * 0] = qd[i + Q * 0] * ug0 + qd[i + Q * 1] * ug1 + qd[i + Q * 2] * ug2;
vg[i + Q * 1] = qd[i + Q * 1] * ug0 + qd[i + Q * 3] * ug1 + qd[i + Q * 4] * ug2;
vg[i + Q * 2] = qd[i + Q * 2] * ug0 + qd[i + Q * 4] * ug1 + qd[i + Q * 5] * ug2;
}
break;
}
return 0;
}
+31
View File
@@ -0,0 +1,31 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_DIFF_HPP
#define MFEM_LIBCEED_DIFF_HPP
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../fespace.hpp"
namespace mfem
{
/// Initialize a Diffusion Integrator using libCEED
void CeedPADiffusionAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &ir, CeedData& ceedData);
}
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_DIFF_HPP
+148
View File
@@ -0,0 +1,148 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "mass.hpp"
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../mesh/mesh.hpp"
#include "../../fem/gridfunc.hpp"
#include "ceed.hpp"
#include "mass.h"
namespace mfem
{
void CeedPAMassAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &irm, CeedData& ceedData)
{
Ceed ceed(internal::ceed);
mfem::Mesh *mesh = fes.GetMesh();
const int ir_order = irm.GetOrder();
const mfem::IntegrationRule &ir =
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir_order);
CeedInt nqpts, nelem = mesh->GetNE();
InitCeedTensorBasisAndRestriction(fes, ir, ceed, &ceedData.basis,
&ceedData.restr);
mesh->EnsureNodes();
const mfem::FiniteElementSpace *mesh_fes = mesh->GetNodalFESpace();
MFEM_VERIFY(mesh_fes, "the Mesh has no nodal FE space");
InitCeedTensorBasisAndRestriction(*mesh_fes, ir, ceed, &ceedData.mesh_basis,
&ceedData.mesh_restr);
CeedBasisGetNumQuadraturePoints(ceedData.basis, &nqpts);
CeedElemRestrictionCreateIdentity(ceed, nelem, nqpts,
nqpts*nelem, 1, &ceedData.restr_i);
CeedElemRestrictionCreateIdentity(ceed, nelem, nqpts,
nqpts*nelem, 1, &ceedData.mesh_restr_i);
CeedVectorCreate(ceed, mesh->GetNodes()->Size(), &ceedData.node_coords);
CeedVectorSetArray(ceedData.node_coords, CEED_MEM_HOST, CEED_USE_POINTER,
mesh->GetNodes()->GetData());
CeedVectorCreate(ceed, nelem*nqpts, &ceedData.rho);
// Context data to be passed to the 'f_build_mass' Q-function.
ceedData.build_ctx.dim = mesh->Dimension();
ceedData.build_ctx.space_dim = mesh->SpaceDimension();
std::string mass_qf_file = GetCeedPath() + "/mass.h";
std::string mass_qf;
// Create the Q-function that builds the mass operator (i.e. computes its
// quadrature data) and set its context data.
switch (ceedData.coeff_type)
{
case CeedCoeff::Const:
mass_qf = mass_qf_file + ":f_build_mass_const";
CeedQFunctionCreateInterior(ceed, 1, f_build_mass_const,
mass_qf.c_str(),
&ceedData.build_qfunc);
ceedData.build_ctx.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
break;
case CeedCoeff::Grid:
mass_qf = mass_qf_file + ":f_build_mass_grid";
CeedQFunctionCreateInterior(ceed, 1, f_build_mass_grid,
mass_qf.c_str(),
&ceedData.build_qfunc);
CeedQFunctionAddInput(ceedData.build_qfunc, "coeff", 1, CEED_EVAL_INTERP);
break;
default:
MFEM_ABORT("This coeff_type is not handled");
}
CeedQFunctionAddInput(ceedData.build_qfunc, "dx",
mesh->SpaceDimension()*mesh->SpaceDimension(),
CEED_EVAL_GRAD);
CeedQFunctionAddInput(ceedData.build_qfunc, "weights", 1, CEED_EVAL_WEIGHT);
CeedQFunctionAddOutput(ceedData.build_qfunc, "rho", 1, CEED_EVAL_NONE);
CeedQFunctionSetContext(ceedData.build_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the operator that builds the quadrature data for the mass operator.
CeedOperatorCreate(ceed, ceedData.build_qfunc, NULL, NULL,
&ceedData.build_oper);
CeedTransposeMode lmode = CEED_NOTRANSPOSE;
if (mesh_fes->GetOrdering()==Ordering::byVDIM)
{
lmode = CEED_TRANSPOSE;
}
if (ceedData.coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* ceedCoeff = (CeedGridCoeff*)ceedData.coeff;
InitCeedTensorBasisAndRestriction(*ceedCoeff->coeff->FESpace(), ir, ceed,
&ceedCoeff->basis,
&ceedCoeff->restr);
CeedVectorCreate(ceed, ceedCoeff->coeff->FESpace()->GetNDofs(),
&ceedCoeff->coeffVector);
CeedVectorSetArray(ceedCoeff->coeffVector, CEED_MEM_HOST, CEED_USE_POINTER,
ceedCoeff->coeff->GetData());
CeedOperatorSetField(ceedData.build_oper, "coeff", ceedCoeff->restr,
CEED_NOTRANSPOSE,
ceedCoeff->basis, ceedCoeff->coeffVector);
}
CeedOperatorSetField(ceedData.build_oper, "dx", ceedData.mesh_restr, lmode,
ceedData.mesh_basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.build_oper, "weights", ceedData.mesh_restr_i,
CEED_NOTRANSPOSE,
ceedData.mesh_basis, CEED_VECTOR_NONE);
CeedOperatorSetField(ceedData.build_oper, "rho", ceedData.restr_i,
CEED_NOTRANSPOSE,
CEED_BASIS_COLLOCATED, CEED_VECTOR_ACTIVE);
// Compute the quadrature data for the mass operator.
CeedOperatorApply(ceedData.build_oper, ceedData.node_coords, ceedData.rho,
CEED_REQUEST_IMMEDIATE);
// Create the Q-function that defines the action of the mass operator.
mass_qf = mass_qf_file + ":f_apply_mass";
CeedQFunctionCreateInterior(ceed, 1, f_apply_mass,
mass_qf.c_str(), &ceedData.apply_qfunc);
CeedQFunctionAddInput(ceedData.apply_qfunc, "u", 1, CEED_EVAL_INTERP);
CeedQFunctionAddInput(ceedData.apply_qfunc, "rho", 1, CEED_EVAL_NONE);
CeedQFunctionAddOutput(ceedData.apply_qfunc, "v", 1, CEED_EVAL_INTERP);
// Create the mass operator.
CeedOperatorCreate(ceed, ceedData.apply_qfunc, NULL, NULL, &ceedData.oper);
CeedOperatorSetField(ceedData.oper, "u", ceedData.restr, CEED_NOTRANSPOSE,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedOperatorSetField(ceedData.oper, "rho", ceedData.restr_i, CEED_NOTRANSPOSE,
CEED_BASIS_COLLOCATED, ceedData.rho);
CeedOperatorSetField(ceedData.oper, "v", ceedData.restr, CEED_NOTRANSPOSE,
ceedData.basis, CEED_VECTOR_ACTIVE);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.u);
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
} // namespace mfem
#endif // MFEM_USE_CEED
+107
View File
@@ -0,0 +1,107 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
/// libCEED Q-function for building quadrature data for a mass operator with a constant coefficient
CEED_QFUNCTION(f_build_mass_const)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
BuildContext *bc = (BuildContext *)ctx;
const CeedScalar coeff = bc->coeff;
const CeedScalar *J = in[0], *qw = in[1];
CeedScalar *rho = out[0];
switch (bc->dim + 10*bc->space_dim)
{
case 11:
for (CeedInt i=0; i<Q; i++)
{
rho[i] = coeff * J[i] * qw[i];
}
break;
case 22:
for (CeedInt i=0; i<Q; i++)
{
// 0 2
// 1 3
rho[i] = coeff * (J[i+Q*0]*J[i+Q*3] - J[i+Q*1]*J[i+Q*2]) * qw[i];
}
break;
case 33:
for (CeedInt i=0; i<Q; i++)
{
// 0 3 6
// 1 4 7
// 2 5 8
rho[i] = (J[i+Q*0]*(J[i+Q*4]*J[i+Q*8] - J[i+Q*5]*J[i+Q*7]) -
J[i+Q*1]*(J[i+Q*3]*J[i+Q*8] - J[i+Q*5]*J[i+Q*6]) +
J[i+Q*2]*(J[i+Q*3]*J[i+Q*7] - J[i+Q*4]*J[i+Q*6])) * coeff * qw[i];
}
break;
}
return 0;
}
/// libCEED Q-function for building quadrature data for a mass operator with a grid function coefficient
CEED_QFUNCTION(f_build_mass_grid)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
// in[0] is Jacobians with shape [dim, nc=dim, Q]
// in[1] is quadrature weights, size (Q)
BuildContext *bc = (BuildContext *)ctx;
const CeedScalar *c = in[0], *J = in[1], *qw = in[2];
CeedScalar *rho = out[0];
switch (bc->dim + 10*bc->space_dim)
{
case 11:
for (CeedInt i=0; i<Q; i++)
{
rho[i] = c[i] * J[i] * qw[i];
}
break;
case 22:
for (CeedInt i=0; i<Q; i++)
{
// 0 2
// 1 3
rho[i] = c[i] * (J[i+Q*0]*J[i+Q*3] - J[i+Q*1]*J[i+Q*2]) * qw[i];
}
break;
case 33:
for (CeedInt i=0; i<Q; i++)
{
// 0 3 6
// 1 4 7
// 2 5 8
rho[i] = (J[i+Q*0]*(J[i+Q*4]*J[i+Q*8] - J[i+Q*5]*J[i+Q*7]) -
J[i+Q*1]*(J[i+Q*3]*J[i+Q*8] - J[i+Q*5]*J[i+Q*6]) +
J[i+Q*2]*(J[i+Q*3]*J[i+Q*7] - J[i+Q*4]*J[i+Q*6])) * c[i] * qw[i];
}
break;
}
return 0;
}
/// libCEED Q-function for applying a mass operator
CEED_QFUNCTION(f_apply_mass)(void *ctx, CeedInt Q,
const CeedScalar *const *in, CeedScalar *const *out)
{
const CeedScalar *u = in[0], *w = in[1];
CeedScalar *v = out[0];
for (CeedInt i=0; i<Q; i++)
{
v[i] = w[i] * u[i];
}
return 0;
}
+31
View File
@@ -0,0 +1,31 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef MFEM_LIBCEED_MASS_HPP
#define MFEM_LIBCEED_MASS_HPP
#include "ceed.hpp"
#ifdef MFEM_USE_CEED
#include "../fespace.hpp"
namespace mfem
{
/// Initialize a Mass Integrator using libCEED
void CeedPAMassAssemble(const FiniteElementSpace &fes,
const mfem::IntegrationRule &ir, CeedData& ceedData);
}
#endif // MFEM_USE_CEED
#endif // MFEM_LIBCEED_MASS_HPP
+8
View File
@@ -81,6 +81,14 @@ public:
Update(FiniteElementSpace *, Vector &, int). */
LinearForm() { fes = NULL; extern_lfs = 0; UseDevice(true); }
/// Construct a LinearForm using previously allocated array @a data.
/** The LinearForm does not assume ownership of @a data which is assumed to
be of size at least `f->GetVSize()`. Similar to the Vector constructor
for externally allocated array, the pointer @a data can be NULL. The data
array can be replaced later using the method SetData(). */
LinearForm(FiniteElementSpace *f, double *data) : Vector(data, f->GetVSize())
{ fes = f; extern_lfs = 0; }
/// Copy assignment. Only the data of the base class Vector is copied.
/** It is assumed that this object and @a rhs use FiniteElementSpace%s that
have the same size.
+45 -2
View File
@@ -14,6 +14,25 @@
namespace mfem
{
void NonlinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
{
if (ext)
{
MFEM_ABORT("the assembly level has already been set!");
}
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::NONE:
// This is the default behavior.
break;
case AssemblyLevel::PARTIAL:
ext = new PANonlinearFormExtension(this);
break;
default:
mfem_error("Unknown assembly level for this form.");
}
}
void NonlinearForm::SetEssentialBC(const Array<int> &bdr_attr_is_ess,
Vector *rhs)
{
@@ -109,13 +128,24 @@ const Vector &NonlinearForm::Prolongate(const Vector &x) const
void NonlinearForm::Mult(const Vector &x, Vector &y) const
{
const Vector &px = Prolongate(x);
if (P) { aux2.SetSize(P->Height()); }
// If we are in parallel, ParNonLinearForm::Mult uses the aux2 vector.
// In serial, place the result directly in y.
Vector &py = P ? aux2 : y;
if (ext)
{
ext->Mult(px, py);
return;
}
Array<int> vdofs;
Vector el_x, el_y;
const FiniteElement *fe;
ElementTransformation *T;
Mesh *mesh = fes->GetMesh();
const Vector &px = Prolongate(x);
Vector &py = P ? aux2.SetSize(P->Height()), aux2 : y;
py = 0.0;
@@ -232,6 +262,11 @@ void NonlinearForm::Mult(const Vector &x, Vector &y) const
Operator &NonlinearForm::GetGradient(const Vector &x) const
{
if (ext)
{
MFEM_ABORT("Not yet implemented!");
}
const int skip_zeros = 0;
Array<int> vdofs;
Vector el_x;
@@ -375,6 +410,8 @@ Operator &NonlinearForm::GetGradient(const Vector &x) const
void NonlinearForm::Update()
{
if (ext) { MFEM_ABORT("Not yet implemented!"); }
if (sequence == fes->GetSequence()) { return; }
height = width = fes->GetTrueVSize();
@@ -387,6 +424,11 @@ void NonlinearForm::Update()
cP = dynamic_cast<const SparseMatrix*>(P);
}
void NonlinearForm::Setup()
{
if (ext) { return ext->AssemblePA(); }
}
NonlinearForm::~NonlinearForm()
{
delete cGrad;
@@ -394,6 +436,7 @@ NonlinearForm::~NonlinearForm()
for (int i = 0; i < dnfi.Size(); i++) { delete dnfi[i]; }
for (int i = 0; i < fnfi.Size(); i++) { delete fnfi[i]; }
for (int i = 0; i < bfnfi.Size(); i++) { delete bfnfi[i]; }
delete ext;
}
+21 -1
View File
@@ -14,6 +14,8 @@
#include "../config/config.hpp"
#include "nonlininteg.hpp"
#include "nonlinearform_ext.hpp"
#include "bilinearform.hpp"
#include "gridfunc.hpp"
namespace mfem
@@ -22,6 +24,13 @@ namespace mfem
class NonlinearForm : public Operator
{
protected:
/// The assembly level.
AssemblyLevel assembly;
/** Extension for supporting Partial Assembly (PA) or
Matrix Free assembly (MF). */
NonlinearFormExtension *ext;
/// FE space on which the form lives.
FiniteElementSpace *fes; // not owned
@@ -59,11 +68,16 @@ public:
/** As an Operator, the NonlinearForm has input and output size equal to the
number of true degrees of freedom, i.e. f->GetTrueVSize(). */
NonlinearForm(FiniteElementSpace *f)
: Operator(f->GetTrueVSize()), fes(f), Grad(NULL), cGrad(NULL),
: Operator(f->GetTrueVSize()), assembly(AssemblyLevel::NONE),
ext(NULL), fes(f), Grad(NULL), cGrad(NULL),
sequence(f->GetSequence()), P(f->GetProlongationMatrix()),
cP(dynamic_cast<const SparseMatrix*>(P))
{ }
/// Set the desired assembly level. The default is AssemblyLevel::NONE.
/** This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level);
FiniteElementSpace *FESpace() { return fes; }
const FiniteElementSpace *FESpace() const { return fes; }
@@ -71,6 +85,9 @@ public:
void AddDomainIntegrator(NonlinearFormIntegrator *nlfi)
{ dnfi.Append(nlfi); }
/// Access all integrators added with AddDomainIntegrator().
Array<NonlinearFormIntegrator*> *GetDNFI() { return &dnfi; }
/// Adds new Interior Face Integrator.
void AddInteriorFaceIntegrator(NonlinearFormIntegrator *nlfi)
{ fnfi.Append(nlfi); }
@@ -144,6 +161,9 @@ public:
set again. */
virtual void Update();
/// Setup the NonlinearForm
virtual void Setup();
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const { return P; }
/// Get the finite element space restriction matrix
+74
View File
@@ -0,0 +1,74 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
// Implementations of classes FABilinearFormExtension, EABilinearFormExtension,
// PABilinearFormExtension and MFBilinearFormExtension.
#include "nonlinearform.hpp"
namespace mfem
{
NonlinearFormExtension::NonlinearFormExtension(NonlinearForm *form)
: Operator(form->FESpace()->GetTrueVSize()), n(form)
{
// empty
}
PANonlinearFormExtension::PANonlinearFormExtension(NonlinearForm *form):
NonlinearFormExtension(form), fes(*form->FESpace())
{
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
elem_restrict_lex = fes.GetElementRestriction(ordering);
if (elem_restrict_lex)
{
localX.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.UseDevice(true); // ensure 'localY = 0.0' is done on device
}
}
void PANonlinearFormExtension::AssemblePA()
{
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int Ni = integrators.Size();
for (int i = 0; i < Ni; ++i)
{
integrators[i]->AssemblePA(*n->FESpace());
}
}
void PANonlinearFormExtension::Mult(const Vector &x, Vector &y) const
{
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
{
elem_restrict_lex->Mult(x, localX);
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(localX, localY);
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
}
}
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#ifndef NONLINEARFORM_EXT_HPP
#define NONLINEARFORM_EXT_HPP
#include "../config/config.hpp"
#include "fespace.hpp"
namespace mfem
{
class NonlinearForm;
class NonlinearFormExtension : public Operator
{
protected:
NonlinearForm *n; ///< Not owned
public:
NonlinearFormExtension(NonlinearForm *form);
virtual void AssemblePA() = 0;
};
/// Data and methods for partially-assembled nonlinear forms
class PANonlinearFormExtension : public NonlinearFormExtension
{
protected:
const FiniteElementSpace &fes; // Not owned
mutable Vector localX, localY;
const Operator *elem_restrict_lex; // Not owned
public:
PANonlinearFormExtension(NonlinearForm*);
void AssemblePA();
void Mult(const Vector &x, Vector &y) const;
};
}
#endif // NONLINEARFORM_EXT_HPP
+140
View File
@@ -10,10 +10,30 @@
// Software Foundation) version 2.1 dated February 1999.
#include "fem.hpp"
#include "../general/forall.hpp"
namespace mfem
{
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
{
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace &,
const FiniteElementSpace &)
{
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
{
mfem_error ("NonlinearFormIntegrator::AddMultPA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssembleElementVector(
const FiniteElement &el, ElementTransformation &Tr,
const Vector &elfun, Vector &elvect)
@@ -673,4 +693,124 @@ void IncompressibleNeoHookeanIntegrator::AssembleElementGrad(
}
const IntegrationRule&
VectorConvectionNLFIntegrator::GetRule(const FiniteElement &fe,
ElementTransformation &T)
{
const int order = 2 * fe.GetOrder() + T.OrderGrad(&fe);
return IntRules.Get(fe.GetGeomType(), order);
}
void VectorConvectionNLFIntegrator::AssembleElementVector(
const FiniteElement &el,
ElementTransformation &T,
const Vector &elfun,
Vector &elvect)
{
const int nd = el.GetDof();
const int dim = el.GetDim();
shape.SetSize(nd);
dshape.SetSize(nd, dim);
elvect.SetSize(nd * dim);
gradEF.SetSize(dim);
EF.UseExternalData(elfun.GetData(), nd, dim);
ELV.UseExternalData(elvect.GetData(), nd, dim);
Vector vec1(dim), vec2(dim);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, T);
ELV = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
T.SetIntPoint(&ip);
el.CalcShape(ip, shape);
el.CalcPhysDShape(T, dshape);
double w = ip.weight * T.Weight();
if (Q) { w *= Q->Eval(T, ip); }
MultAtB(EF, dshape, gradEF);
EF.MultTranspose(shape, vec1);
gradEF.Mult(vec1, vec2);
vec2 *= w;
AddMultVWt(shape, vec2, ELV);
}
}
void VectorConvectionNLFIntegrator::AssembleElementGrad(
const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat)
{
int nd = el.GetDof();
int dim = el.GetDim();
shape.SetSize(nd);
dshape.SetSize(nd, dim);
dshapex.SetSize(nd, dim);
elmat.SetSize(nd * dim);
elmat_comp.SetSize(nd);
gradEF.SetSize(dim);
EF.UseExternalData(elfun.GetData(), nd, dim);
double w;
Vector vec1(dim), vec2(dim), vec3(nd);
const IntegrationRule *ir = IntRule;
if (ir == nullptr)
{
int order = 2 * el.GetOrder() + trans.OrderGrad(&el);
ir = &IntRules.Get(el.GetGeomType(), order);
}
elmat = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
trans.SetIntPoint(&ip);
el.CalcShape(ip, shape);
el.CalcDShape(ip, dshape);
Mult(dshape, trans.InverseJacobian(), dshapex);
w = ip.weight;
if (Q)
{
w *= Q->Eval(trans, ip);
}
MultAtB(EF, dshapex, gradEF);
EF.MultTranspose(shape, vec1);
trans.AdjugateJacobian().Mult(vec1, vec2);
vec2 *= w;
dshape.Mult(vec2, vec3);
MultVWt(shape, vec3, elmat_comp);
for (int i = 0; i < dim; i++)
{
elmat.AddMatrix(elmat_comp, i * nd, i * nd);
}
MultVVt(shape, elmat_comp);
w = ip.weight * trans.Weight();
if (Q)
{
w *= Q->Eval(trans, ip);
}
for (int i = 0; i < dim; i++)
{
for (int j = 0; j < dim; j++)
{
elmat.AddMatrix(w * gradEF(i, j), elmat_comp, i * nd, j * nd);
}
}
}
}
}
+57
View File
@@ -15,6 +15,7 @@
#include "../config/config.hpp"
#include "fe.hpp"
#include "coefficient.hpp"
#include "fespace.hpp"
namespace mfem
{
@@ -68,6 +69,26 @@ public:
ElementTransformation &Tr,
const Vector &elfun);
/// Method defining partial assembly.
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA(). */
virtual void AssemblePA(const FiniteElementSpace &fes);
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA().
Used with BilinearFormIntegrators that have different spaces. */
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
/// Method for partially assembled action.
/** Perform the action of integrator on the input @a x and add the result to
the output @a y. Both @a x and @a y are E-vectors, i.e. they represent
the element-wise discontinuous version of the FE space.
This method can be called only after the method AssemblePA() has been
called. */
virtual void AddMultPA(const Vector &x, Vector &y) const;
virtual ~NonlinearFormIntegrator() { }
};
@@ -285,6 +306,42 @@ public:
const Array2D<DenseMatrix *> &elmats);
};
class VectorConvectionNLFIntegrator : public NonlinearFormIntegrator
{
private:
Coefficient *Q{};
DenseMatrix dshape, dshapex, EF, gradEF, ELV, elmat_comp;
Vector shape;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq;
public:
VectorConvectionNLFIntegrator(Coefficient &q): Q(&q) { }
VectorConvectionNLFIntegrator() = default;
static const IntegrationRule &GetRule(const FiniteElement &fe,
ElementTransformation &T);
virtual void AssembleElementVector(const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
Vector &elvect);
virtual void AssembleElementGrad(const FiniteElement &el,
ElementTransformation &trans,
const Vector &elfun,
DenseMatrix &elmat);
using NonlinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AddMultPA(const Vector &x, Vector &y) const;
};
}
#endif
+816
View File
@@ -0,0 +1,816 @@
// Copyright (c) 2019, Lawrence Livermore National Security, LLC. Produced at
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
// reserved. See file COPYRIGHT for details.
//
// This file is part of the MFEM library. For more information and source code
// availability see http://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the GNU Lesser General Public License (as published by the Free
// Software Foundation) version 2.1 dated February 1999.
#include "../general/forall.hpp"
#include "nonlininteg.hpp"
using namespace std;
namespace mfem
{
void VectorConvectionNLFIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetFE(0);
ElementTransformation &T = *mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, T);
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
pa_data.SetSize(ne * nq * dim * dim, Device::GetMemoryType());
double COEFF = 1.0;
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient *>(Q);
MFEM_VERIFY(cQ != NULL, "only ConstantCoefficient is supported!");
COEFF = cQ->constant;
}
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
if (dim == 1)
{
MFEM_ABORT("dim==1 not supported!");
}
if (dim == 2)
{
auto J = Reshape(geom->J.Read(), NQ, 2, 2, NE);
auto G = Reshape(pa_data.Write(), NQ, 2, 2, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q, 0, 0, e);
const double J12 = J(q, 0, 1, e);
const double J21 = J(q, 1, 0, e);
const double J22 = J(q, 1, 1, e);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = W[q] * COEFF * J22; // 1,1
G(q, 0, 1, e) = W[q] * COEFF * -J12; // 1,2
G(q, 1, 0, e) = W[q] * COEFF * -J21; // 2,1
G(q, 1, 1, e) = W[q] * COEFF * J11; // 2,2
}
});
}
if (dim == 3)
{
auto J = Reshape(geom->J.Read(), NQ, 3, 3, NE);
auto G = Reshape(pa_data.Write(), NQ, 3, 3, NE);
MFEM_FORALL(e, NE,
{
for (int q = 0; q < NQ; ++q)
{
const double J11 = J(q, 0, 0, e);
const double J21 = J(q, 1, 0, e);
const double J31 = J(q, 2, 0, e);
const double J12 = J(q, 0, 1, e);
const double J22 = J(q, 1, 1, e);
const double J32 = J(q, 2, 1, e);
const double J13 = J(q, 0, 2, e);
const double J23 = J(q, 1, 2, e);
const double J33 = J(q, 2, 2, e);
const double cw = W[q] * COEFF;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// Store wq * Q * adj(J)
G(q, 0, 0, e) = cw * A11; // 1,1
G(q, 0, 1, e) = cw * A12; // 1,2
G(q, 0, 2, e) = cw * A13; // 1,3
G(q, 1, 0, e) = cw * A21; // 2,1
G(q, 1, 1, e) = cw * A22; // 2,2
G(q, 1, 2, e) = cw * A23; // 2,3
G(q, 2, 0, e) = cw * A31; // 3,1
G(q, 2, 1, e) = cw * A32; // 3,2
G(q, 2, 2, e) = cw * A33; // 3,3
}
});
}
}
// PA Convection NL 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply2D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D, 2, 2, NE);
auto x = Reshape(x_.Read(), D1D, D1D, 2, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NE);
MFEM_FORALL(e, NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double data[max_Q1D][max_Q1D][2];
double grad0[max_Q1D][max_Q1D][2];
double grad1[max_Q1D][max_Q1D][2];
double Z[max_Q1D][max_Q1D][2];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] = 0.0;
data[qy][qx][1] = 0.0;
grad0[qy][qx][0] = 0.0;
grad0[qy][qx][1] = 0.0;
grad1[qy][qx][0] = 0.0;
grad1[qy][qx][1] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double dataX[max_Q1D][2];
double gradX0[max_Q1D][2];
double gradX1[max_Q1D][2];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s0 = x(dx, dy, 0, e);
const double s1 = x(dx, dy, 1, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx, dx);
const double Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy, dy);
const double Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
data[qy][qx][0] += dataX[qx][0] * By;
data[qy][qx][1] += dataX[qx][1] * By;
grad0[qy][qx][0] += gradX0[qx][0] * By;
grad0[qy][qx][1] += gradX0[qx][1] * Gy;
grad1[qy][qx][0] += gradX1[qx][0] * By;
grad1[qy][qx][1] += gradX1[qx][1] * Gy;
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + qy * Q1D;
const double u1 = data[qy][qx][0];
const double u2 = data[qy][qx][1];
const double grad00 = grad0[qy][qx][0];
const double grad01 = grad0[qy][qx][1];
const double grad10 = grad1[qy][qx][0];
const double grad11 = grad1[qy][qx][1];
const double Dxu1 = grad00 * Q(q, 0, 0, e) + grad01 * Q(q, 1, 0, e);
const double Dyu1 = grad00 * Q(q, 0, 1, e) + grad01 * Q(q, 1, 1, e);
const double Dxu2 = grad10 * Q(q, 0, 0, e) + grad11 * Q(q, 1, 0, e);
const double Dyu2 = grad10 * Q(q, 0, 1, e) + grad11 * Q(q, 1, 1, e);
Z[qy][qx][0] = u1 * Dxu1 + u2 * Dyu1;
Z[qy][qx][1] = u1 * Dxu2 + u2 * Dyu2;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double Y[max_D1D][2];
for (int dx = 0; dx < D1D; ++dx)
{
Y[dx][0] = 0.0;
Y[dx][1] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double Btx = Bt(dx, qx);
Y[dx][0] += Btx * Z[qy][qx][0];
Y[dx][1] += Btx * Z[qy][qx][1];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Bty = Bt(dy, qy);
y(dx, dy, 0, e) += Bty * Y[dx][0];
y(dx, dy, 1, e) += Bty * Y[dx][1];
}
}
}
});
}
// PA Convection NL 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void PAConvectionNLApply3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &bt,
const Vector &q_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= MAX_D1D, "");
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Bt = Reshape(bt.Read(), D1D, Q1D);
auto Q = Reshape(q_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL(e, NE,
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
double data[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad0[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad1[max_Q1D][max_Q1D][max_Q1D][VDIM];
double grad2[max_Q1D][max_Q1D][max_Q1D][VDIM];
double Z[max_Q1D][max_Q1D][max_Q1D][VDIM];
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] = 0.0;
data[qz][qy][qx][1] = 0.0;
data[qz][qy][qx][2] = 0.0;
grad0[qz][qy][qx][0] = 0.0;
grad0[qz][qy][qx][1] = 0.0;
grad0[qz][qy][qx][2] = 0.0;
grad1[qz][qy][qx][0] = 0.0;
grad1[qz][qy][qx][1] = 0.0;
grad1[qz][qy][qx][2] = 0.0;
grad2[qz][qy][qx][0] = 0.0;
grad2[qz][qy][qx][1] = 0.0;
grad2[qz][qy][qx][2] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
double dataXY[max_Q1D][max_Q1D][VDIM];
double gradXY0[max_Q1D][max_Q1D][VDIM];
double gradXY1[max_Q1D][max_Q1D][VDIM];
double gradXY2[max_Q1D][max_Q1D][VDIM];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] = 0.0;
dataXY[qy][qx][1] = 0.0;
dataXY[qy][qx][2] = 0.0;
gradXY0[qy][qx][0] = 0.0;
gradXY0[qy][qx][1] = 0.0;
gradXY0[qy][qx][2] = 0.0;
gradXY1[qy][qx][0] = 0.0;
gradXY1[qy][qx][1] = 0.0;
gradXY1[qy][qx][2] = 0.0;
gradXY2[qy][qx][0] = 0.0;
gradXY2[qy][qx][1] = 0.0;
gradXY2[qy][qx][2] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
double dataX[max_Q1D][VDIM];
double gradX0[max_Q1D][VDIM];
double gradX1[max_Q1D][VDIM];
double gradX2[max_Q1D][VDIM];
for (int qx = 0; qx < Q1D; ++qx)
{
dataX[qx][0] = 0.0;
dataX[qx][1] = 0.0;
dataX[qx][2] = 0.0;
gradX0[qx][0] = 0.0;
gradX0[qx][1] = 0.0;
gradX0[qx][2] = 0.0;
gradX1[qx][0] = 0.0;
gradX1[qx][1] = 0.0;
gradX1[qx][2] = 0.0;
gradX2[qx][0] = 0.0;
gradX2[qx][1] = 0.0;
gradX2[qx][2] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const double s0 = x(dx, dy, dz, 0, e);
const double s1 = x(dx, dy, dz, 1, e);
const double s2 = x(dx, dy, dz, 2, e);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx, dx);
const double Gx = G(qx, dx);
dataX[qx][0] += s0 * Bx;
dataX[qx][1] += s1 * Bx;
dataX[qx][2] += s2 * Bx;
gradX0[qx][0] += s0 * Gx;
gradX0[qx][1] += s0 * Bx;
gradX0[qx][2] += s0 * Bx;
gradX1[qx][0] += s1 * Gx;
gradX1[qx][1] += s1 * Bx;
gradX1[qx][2] += s1 * Bx;
gradX2[qx][0] += s2 * Gx;
gradX2[qx][1] += s2 * Bx;
gradX2[qx][2] += s2 * Bx;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy, dy);
const double Gy = G(qy, dy);
for (int qx = 0; qx < Q1D; ++qx)
{
dataXY[qy][qx][0] += dataX[qx][0] * By;
dataXY[qy][qx][1] += dataX[qx][1] * By;
dataXY[qy][qx][2] += dataX[qx][2] * By;
gradXY0[qy][qx][0] += gradX0[qx][0] * By;
gradXY0[qy][qx][1] += gradX0[qx][1] * Gy;
gradXY0[qy][qx][2] += gradX0[qx][2] * By;
gradXY1[qy][qx][0] += gradX1[qx][0] * By;
gradXY1[qy][qx][1] += gradX1[qx][1] * Gy;
gradXY1[qy][qx][2] += gradX1[qx][2] * By;
gradXY2[qy][qx][0] += gradX2[qx][0] * By;
gradXY2[qy][qx][1] += gradX2[qx][1] * Gy;
gradXY2[qy][qx][2] += gradX2[qx][2] * By;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const double Bz = B(qz, dz);
const double Gz = G(qz, dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
data[qz][qy][qx][0] += dataXY[qy][qx][0] * Bz;
data[qz][qy][qx][1] += dataXY[qy][qx][1] * Bz;
data[qz][qy][qx][2] += dataXY[qy][qx][2] * Bz;
grad0[qz][qy][qx][0] += gradXY0[qy][qx][0] * Bz;
grad0[qz][qy][qx][1] += gradXY0[qy][qx][1] * Bz;
grad0[qz][qy][qx][2] += gradXY0[qy][qx][2] * Gz;
grad1[qz][qy][qx][0] += gradXY1[qy][qx][0] * Bz;
grad1[qz][qy][qx][1] += gradXY1[qy][qx][1] * Bz;
grad1[qz][qy][qx][2] += gradXY1[qy][qx][2] * Gz;
grad2[qz][qy][qx][0] += gradXY2[qy][qx][0] * Bz;
grad2[qz][qy][qx][1] += gradXY2[qy][qx][1] * Bz;
grad2[qz][qy][qx][2] += gradXY2[qy][qx][2] * Gz;
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
const int q = qx + Q1D * (qy + qz * Q1D);
const double u1 = data[qz][qy][qx][0];
const double u2 = data[qz][qy][qx][1];
const double u3 = data[qz][qy][qx][2];
const double grad00 = grad0[qz][qy][qx][0];
const double grad01 = grad0[qz][qy][qx][1];
const double grad02 = grad0[qz][qy][qx][2];
const double grad10 = grad1[qz][qy][qx][0];
const double grad11 = grad1[qz][qy][qx][1];
const double grad12 = grad1[qz][qy][qx][2];
const double grad20 = grad2[qz][qy][qx][0];
const double grad21 = grad2[qz][qy][qx][1];
const double grad22 = grad2[qz][qy][qx][2];
const double Dxu1 = grad00 * Q(q, 0, 0, e)
+ grad01 * Q(q, 1, 0, e)
+ grad02 * Q(q, 2, 0, e);
const double Dyu1 = grad00 * Q(q, 0, 1, e)
+ grad01 * Q(q, 1, 1, e)
+ grad02 * Q(q, 2, 1, e);
const double Dzu1 = grad00 * Q(q, 0, 2, e)
+ grad01 * Q(q, 1, 2, e)
+ grad02 * Q(q, 2, 2, e);
const double Dxu2 = grad10 * Q(q, 0, 0, e)
+ grad11 * Q(q, 1, 0, e)
+ grad12 * Q(q, 2, 0, e);
const double Dyu2 = grad10 * Q(q, 0, 1, e)
+ grad11 * Q(q, 1, 1, e)
+ grad12 * Q(q, 2, 1, e);
const double Dzu2 = grad10 * Q(q, 0, 2, e)
+ grad11 * Q(q, 1, 2, e)
+ grad12 * Q(q, 2, 2, e);
const double Dxu3 = grad20 * Q(q, 0, 0, e)
+ grad21 * Q(q, 1, 0, e)
+ grad22 * Q(q, 2, 0, e);
const double Dyu3 = grad20 * Q(q, 0, 1, e)
+ grad21 * Q(q, 1, 1, e)
+ grad22 * Q(q, 2, 1, e);
const double Dzu3 = grad20 * Q(q, 0, 2, e)
+ grad21 * Q(q, 1, 2, e)
+ grad22 * Q(q, 2, 2, e);
Z[qz][qy][qx][0] = u1 * Dxu1 + u2 * Dyu1 + u3 * Dzu1;
Z[qz][qy][qx][1] = u1 * Dxu2 + u2 * Dyu2 + u3 * Dzu2;
Z[qz][qy][qx][2] = u1 * Dxu3 + u2 * Dyu3 + u3 * Dzu3;
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
double opXY[max_D1D][max_D1D][VDIM];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
opXY[dy][dx][0] = 0.0;
opXY[dy][dx][1] = 0.0;
opXY[dy][dx][2] = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
double opX[max_D1D][VDIM];
for (int dx = 0; dx < D1D; ++dx)
{
opX[dx][0] = 0.0;
opX[dx][1] = 0.0;
opX[dx][2] = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double Btx = Bt(dx, qx);
opX[dx][0] += Btx * Z[qz][qy][qx][0];
opX[dx][1] += Btx * Z[qz][qy][qx][1];
opX[dx][2] += Btx * Z[qz][qy][qx][2];
}
}
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Bty = Bt(dy, qy);
opXY[dy][dx][0] += Bty * opX[dx][0];
opXY[dy][dx][1] += Bty * opX[dx][1];
opXY[dy][dx][2] += Bty * opX[dx][2];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
const double Btz = Bt(dz, qz);
y(dx, dy, dz, 0, e) += Btz * opXY[dy][dx][0];
y(dx, dy, dz, 1, e) += Btz * opXY[dy][dx][1];
y(dx, dy, dz, 2, e) += Btz * opXY[dy][dx][2];
}
}
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX_D1D =0, int T_MAX_Q1D =0>
static void SmemPAConvectionNLApply3D(const int NE,
const Array<double> &b_,
const Array<double> &g_,
const Vector &d_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto D = Reshape(d_.Read(), Q1D * Q1D * Q1D, VDIM, VDIM, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX_Q1D;
MFEM_SHARED double BG[2][MQ1 * MD1];
double(*B)[MD1] = (double(*)[MD1])(BG + 0);
double(*G)[MD1] = (double(*)[MD1])(BG + 1);
double(*Bt)[MQ1] = (double(*)[MQ1])(BG + 0);
MFEM_SHARED double U[2][MQ1][MQ1][MQ1];
MFEM_SHARED double sm0[3][MQ1 * MQ1 * MQ1];
MFEM_SHARED double sm1[3][MQ1 * MQ1 * MQ1];
double(*DDQ0)[MD1][MQ1] = (double(*)[MD1][MQ1])(sm0 + 0);
double(*DDQ1)[MD1][MQ1] = (double(*)[MD1][MQ1])(sm0 + 1);
double(*X)[MD1][MD1] = (double(*)[MD1][MD1])(sm0 + 2);
double(*DQQ0)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 0);
double(*DQQ1)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 1);
double(*DQQ2)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm1 + 2);
double(*QQQ0)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 0);
double(*QQQ1)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 1);
double(*QQQ2)[MQ1][MQ1] = (double(*)[MQ1][MQ1])(sm0 + 2);
double(*QQD0)[MQ1][MD1] = (double(*)[MQ1][MD1])(sm1 + 0);
double(*QDD0)[MD1][MD1] = (double(*)[MD1][MD1])(sm0 + 0);
MFEM_SHARED double Z[MQ1][MQ1][MQ1];
for (int cy = 0; cy < VDIM; ++cy)
{
if (tidz == 0)
{
MFEM_FOREACH_THREAD(q, x, Q1D)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
B[q][d] = b(q, d);
G[q][d] = g(q, d);
}
}
}
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D) { Z[qz][qy][qx] = 0.0; }
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
X[dz][dy][dx] = x(dx, dy, dz, cy, e);
U[0][dz][dy][dx] = x(dx, dy, dz, c, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double z = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double coord = X[dz][dy][dx];
const double value = U[0][dz][dy][dx];
u += coord * B[qx][dx];
v += coord * G[qx][dx];
z += value * B[qx][dx];
}
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
U[1][dz][dy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
double z = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
z += U[1][dz][dy][qx] * B[qy][dy];
}
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
U[0][dz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
double z = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
z += U[0][dz][qy][qx] * B[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
U[1][qz][qy][qx] = z;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
{
const int q = qx + (qy + qz * Q1D) * Q1D;
const double z = U[1][qz][qy][qx];
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
const double d = gX * D(q, 0, c, e) + gY * D(q, 1, c, e)
+ gZ * D(q, 2, c, e);
Z[qz][qy][qx] += z * d;
}
}
}
MFEM_SYNC_THREAD;
} // for each conv component
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d, y, D1D)
{
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[d][q] = b(q, d); }
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(qy, y, Q1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += Z[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz, z, Q1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz, z, D1D)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
{
double u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u += QDD0[qz][dy][dx] * Bt[dz][qz];
}
Y(dx, dy, dz, cy, e) += u;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
void VectorConvectionNLFIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const int NE = ne;
const int D1D = maps->ndof;
const int Q1D = maps->nqpt;
const Vector &Q = pa_data;
const Array<double> &B = maps->B;
const Array<double> &G = maps->G;
const Array<double> &Bt = maps->Bt;
if (dim == 2)
{
return PAConvectionNLApply2D(NE, B, G, Bt, Q, x, y, D1D, Q1D);
}
if (dim == 3)
{
constexpr int T_MAX_D1D = 8;
constexpr int T_MAX_Q1D = 8;
MFEM_VERIFY(D1D <= T_MAX_D1D && Q1D <= T_MAX_Q1D, "Not yet implemented!");
return SmemPAConvectionNLApply3D<0, 0, T_MAX_D1D, T_MAX_Q1D>
(NE, B, G, Q, x, y, D1D, Q1D);
}
MFEM_ABORT("Not yet implemented!");
}
} // namespace mfem
+72 -68
View File
@@ -98,8 +98,6 @@ void ParFiniteElementSpace::ParInit(ParMesh *pm)
gcomm = NULL;
gfdofs = NULL;
P = NULL;
Pconf = NULL;
R = NULL;
@@ -151,7 +149,6 @@ void ParFiniteElementSpace::Construct()
ConstructTrueDofs();
ngedofs = ngfdofs = 0;
gfdofs = NULL;
// calculate number of ghost DOFs
ngvdofs = pncmesh->GetNGhostVertices()
@@ -165,22 +162,8 @@ void ParFiniteElementSpace::Construct()
if (pmesh->Dimension() > 2)
{
if (fdofs != NULL) // have mixed faces
{
gfdofs = new int[pncmesh->GetNGhostFaces()+1];
gfdofs[0] = 0;
for (int i = 0; i < pncmesh->GetNGhostFaces(); i++)
{
int ghost = pncmesh->GetNFaces() + i;
ngfdofs += fec->DofForGeometry(pncmesh->GetFaceGeometry(ghost));
gfdofs[i+1] = ngfdofs;
}
}
else
{
ngfdofs = pncmesh->GetNGhostFaces()
* fec->DofForGeometry(pncmesh->GetFaceGeometry(0));
}
int stride = fec->DofForGeometry(Geometry::SQUARE);
ngfdofs = pncmesh->GetNGhostFaces() * stride;
}
// total number of ghost DOFs. Ghost DOFs start at index 'ndofs', i.e.,
@@ -763,6 +746,7 @@ void ParFiniteElementSpace::GetEssentialTrueDofs(const Array<int>
GetEssentialVDofs(bdr_attr_is_ess, ess_dofs, component);
GetRestrictionMatrix()->BooleanMult(ess_dofs, true_ess_dofs);
#ifdef MFEM_DEBUG
// Verify that in boolean arithmetic: P^T ess_dofs = R ess_dofs.
Array<int> true_ess_dofs2(true_ess_dofs.Size());
@@ -778,6 +762,7 @@ void ParFiniteElementSpace::GetEssentialTrueDofs(const Array<int>
}
MFEM_VERIFY(counter == 0, "internal MFEM error: counter = " << counter);
#endif
MarkerToList(true_ess_dofs, ess_tdof_list);
}
@@ -785,7 +770,7 @@ int ParFiniteElementSpace::GetLocalTDofNumber(int ldof) const
{
if (Nonconforming())
{
Dof_TrueDof_Matrix(); // inline method
Dof_TrueDof_Matrix(); // make sure P has been built
return ldof_ltdof[ldof]; // NOTE: contains -1 for slaves/DOFs we don't own
}
@@ -876,7 +861,13 @@ const Operator *ParFiniteElementSpace::GetProlongationMatrix() const
{
if (Conforming())
{
if (!Pconf)
if (Pconf) { return Pconf; }
if (NRanks == 1)
{
Pconf = new IdentityOperator(GetTrueVSize());
}
else
{
if (!Device::Allows(Backend::DEVICE_MASK))
{
@@ -884,10 +875,7 @@ const Operator *ParFiniteElementSpace::GetProlongationMatrix() const
}
else
{
if (NRanks > 1)
{
Pconf = new DeviceConformingProlongationOperator(*this);
}
Pconf = new DeviceConformingProlongationOperator(*this);
}
}
return Pconf;
@@ -1348,8 +1336,9 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
int nv = fec->DofForGeometry(Geometry::POINT);
int ne = fec->DofForGeometry(Geometry::SEGMENT);
int nf = fec->DofForGeometry((nfv == 3) ?
Geometry::TRIANGLE : Geometry::SQUARE);
int nf_tri = fec->DofForGeometry(Geometry::TRIANGLE);
int nf_quad = fec->DofForGeometry(Geometry::SQUARE);
int nf = (nfv == 3) ? nf_tri : nf_quad;
dofs.SetSize(nfv*(nv + ne) + nf);
@@ -1378,8 +1367,7 @@ void ParFiniteElementSpace::GetGhostFaceDofs(const MeshId &face_id,
}
const int ghost_face_index = face_id.index - pncmesh->GetNFaces();
int first = ndofs + ngvdofs + ngedofs;
first += gfdofs ? gfdofs[ghost_face_index] : nf*ghost_face_index;
int first = ndofs + ngvdofs + ngedofs + nf_quad*ghost_face_index;
for (int j = 0; j < nf; j++)
{
@@ -1422,7 +1410,11 @@ void ParFiniteElementSpace::GetBareDofs(int entity, int index,
break;
default:
ned = fec->DofForGeometry(pncmesh->GetFaceGeometry(index));
Geometry::Type geom = pncmesh->GetFaceGeometry(index);
MFEM_ASSERT(geom == Geometry::SQUARE ||
geom == Geometry::TRIANGLE, "");
ned = fec->DofForGeometry(geom);
ghost = pncmesh->GetNFaces();
if (index < ghost) // regular face
@@ -1432,8 +1424,8 @@ void ParFiniteElementSpace::GetBareDofs(int entity, int index,
else // ghost face
{
index -= ghost;
first = ndofs + ngvdofs + ngedofs +
(gfdofs ? gfdofs[index] : index*ned);
int stride = fec->DofForGeometry(Geometry::SQUARE);
first = ndofs + ngvdofs + ngedofs + index*stride;
}
break;
}
@@ -1480,8 +1472,8 @@ int ParFiniteElementSpace::PackDof(int entity, int index, int edof) const
else // ghost face
{
index -= ghost;
return ndofs + ngvdofs + ngedofs +
(gfdofs ? gfdofs[index] : index*ned) + edof;
int stride = fec->DofForGeometry(Geometry::SQUARE);
return ndofs + ngvdofs + ngedofs + index*stride + edof;
}
}
}
@@ -1553,16 +1545,8 @@ void ParFiniteElementSpace::UnpackDof(int dof,
dof -= ngedofs;
if (dof < ngfdofs) // ghost face
{
if (gfdofs) // have mixed faces
{
index = bisect(gfdofs+1, pncmesh->GetNGhostFaces(), dof);
edof = dof - gfdofs[index];
}
else // uniform faces
{
int nf = fec->DofForGeometry(pncmesh->GetFaceGeometry(0));
index = pncmesh->GetNFaces() + dof / nf, edof = dof % nf;
}
int stride = fec->DofForGeometry(Geometry::SQUARE);
index = pncmesh->GetNFaces() + dof / stride, edof = dof % stride;
entity = 2;
return;
}
@@ -2011,7 +1995,7 @@ int ParFiniteElementSpace
const NCMesh::Slave &sf = list.slaves[si];
if (pncmesh->IsGhost(entity, sf.index)) { continue; }
GetEntityDofs(entity, sf.index, slave_dofs);
GetEntityDofs(entity, sf.index, slave_dofs, mf.Geom());
if (!slave_dofs.Size()) { continue; }
sf.OrientedPointMatrix(T.GetPointMat());
@@ -2495,10 +2479,11 @@ ParFiniteElementSpace::RebalanceMatrix(int old_ndofs,
// create the offdiagonal part of the matrix
HYPRE_Int* i_offd = make_i_array(vsize);
for (int i = 0; i < new_elements.Size(); i++)
for (int i = 0, pos = 0; i < new_elements.Size(); i++)
{
GetElementDofs(new_elements[i], dofs);
const long* old_dofs = &old_remote_dofs[i * dofs.Size() * vdim];
const long* old_dofs = &old_remote_dofs[pos];
pos += dofs.Size() * vdim;
for (int vd = 0; vd < vdim; vd++)
{
@@ -2553,24 +2538,35 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
int nrk = HYPRE_AssumedPartitionCheck() ? 2 : NRanks;
MFEM_VERIFY(Nonconforming(), "Not implemented for conforming meshes.");
MFEM_VERIFY(pmesh->GetNumGeometries(pmesh->Dimension()) == 1,
"Not implemented for mixed meshes.");
MFEM_VERIFY(old_dof_offsets[nrk], "Missing previous (finer) space.");
#if 0 // check no longer seems to work with NC tet refinement
MFEM_VERIFY(dof_offsets[nrk] <= old_dof_offsets[nrk],
"Previous space is not finer.");
#endif
// Note to the reader: please make sure you first read
// FiniteElementSpace::RefinementMatrix, then
// FiniteElementSpace::DerefinementMatrix, and only then this function.
// You have been warned! :-)
Mesh::GeometryList elem_geoms(*mesh);
Array<int> dofs, old_dofs, old_vdofs;
Vector row;
ParNCMesh* pncmesh = pmesh->pncmesh;
Geometry::Type geom = pncmesh->GetElementGeometry(0); // TODO mixed meshes
int ldof = fec->FiniteElementForGeometry(geom)->GetDof();
int ldof[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; i++)
{
ldof[i] = 0;
}
for (int i = 0; i < elem_geoms.Size(); i++)
{
Geometry::Type geom = elem_geoms[i];
ldof[geom] = fec->FiniteElementForGeometry(geom)->GetDof();
}
const CoarseFineTransformations &dtrans = pncmesh->GetDerefinementTransforms();
const Array<int> &old_ranks = pncmesh->GetDerefineOldRanks();
@@ -2607,10 +2603,13 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
}
else if (coarse_rank == MyRank && fine_rank != MyRank)
{
DerefDofMessage &msg = messages[k];
msg.dofs.resize(ldof*vdim);
MFEM_ASSERT(emb.parent >= 0, "");
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
MPI_Irecv(&msg.dofs[0], ldof*vdim, HYPRE_MPI_INT,
DerefDofMessage &msg = messages[k];
msg.dofs.resize(ldof[geom]*vdim);
MPI_Irecv(&msg.dofs[0], ldof[geom]*vdim, HYPRE_MPI_INT,
fine_rank, 291, MyComm, &msg.request);
}
// TODO: coalesce Isends/Irecvs to the same rank. Typically, on uniform
@@ -2618,14 +2617,18 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
// from MyRank+1
}
DenseTensor localR;
GetLocalDerefinementMatrices(geom, localR);
DenseTensor localR[Geometry::NumGeom];
for (int i = 0; i < elem_geoms.Size(); i++)
{
GetLocalDerefinementMatrices(elem_geoms[i], localR[elem_geoms[i]]);
}
// create the diagonal part of the derefinement matrix
SparseMatrix *diag = new SparseMatrix(ndofs*vdim, old_ndofs*vdim);
Array<char> mark(diag->Height());
mark = 0;
for (int k = 0; k < dtrans.embeddings.Size(); k++)
{
const Embedding &emb = dtrans.embeddings[k];
@@ -2636,7 +2639,8 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (coarse_rank == MyRank && fine_rank == MyRank)
{
DenseMatrix &lR = localR(emb.matrix);
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
old_elem_dof->GetRow(k, old_dofs);
@@ -2648,7 +2652,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
for (int i = 0; i < lR.Height(); i++)
{
if (lR(i, 0) == infinity()) { continue; }
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -2666,8 +2670,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
diag->Finalize();
// wait for all sends/receives to complete
for (std::map<int, DerefDofMessage>::iterator
it = messages.begin(); it != messages.end(); ++it)
for (auto it = messages.begin(); it != messages.end(); ++it)
{
MPI_Wait(&it->second.request, MPI_STATUS_IGNORE);
}
@@ -2686,7 +2689,8 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (coarse_rank == MyRank && fine_rank != MyRank)
{
DenseMatrix &lR = localR(emb.matrix);
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
DenseMatrix &lR = localR[geom](emb.matrix);
elem_dof->GetRow(emb.parent, dofs);
@@ -2695,11 +2699,12 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
for (int vd = 0; vd < vdim; vd++)
{
HYPRE_Int* remote_dofs = &msg.dofs[vd*ldof];
MFEM_ASSERT(ldof[geom], "");
HYPRE_Int* remote_dofs = &msg.dofs[vd*ldof[geom]];
for (int i = 0; i < lR.Height(); i++)
{
if (lR(i, 0) == infinity()) { continue; }
if (!std::isfinite(lR(i, 0))) { continue; }
int r = DofToVDof(dofs[i], vd);
int m = (r >= 0) ? r : (-1 - r);
@@ -2707,7 +2712,8 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
if (!mark[m])
{
lR.GetRow(i, row);
for (int j = 0; j < ldof; j++)
MFEM_ASSERT(ldof[geom] == row.Size(), "");
for (int j = 0; j < ldof[geom]; j++)
{
if (row[j] == 0.0) { continue; } // NOTE: lR thresholded
int &lcol = col_map[remote_dofs[j]];
@@ -2737,7 +2743,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
// sure cmap is determined and sorted before the offd matrix is created
{
int width = offd->Width();
Array<Pair<int, int> > reorder(width);
Array<Pair<HYPRE_Int, int> > reorder(width);
for (int i = 0; i < width; i++)
{
reorder[i].one = cmap[i];
@@ -2793,8 +2799,6 @@ void ParFiniteElementSpace::Destroy()
delete Pconf; Pconf = NULL;
delete R; R = NULL;
delete [] gfdofs; gfdofs = NULL;
delete gcomm; gcomm = NULL;
num_face_nbr_dofs = -1;
+1 -2
View File
@@ -46,7 +46,6 @@ private:
/// Number of vertex/edge/face/total ghost DOFs (nonconforming case).
int ngvdofs, ngedofs, ngfdofs, ngdofs;
int* gfdofs;
/// The group of each local dof.
Array<int> ldof_group;
@@ -72,7 +71,7 @@ private:
/// The matrix P (interpolation from true dof to dof). Owned.
mutable HypreParMatrix *P;
/// Optimized action-only prolongation operator for conforming meshes. Owned.
mutable class ConformingProlongationOperator *Pconf;
mutable Operator *Pconf;
/// The (block-diagonal) matrix R (restriction of dof to true dof). Owned.
mutable SparseMatrix *R;
+34 -46
View File
@@ -404,33 +404,27 @@ void ParGridFunction::ProjectBdrCoefficient(
{
Array<int> values_counter;
AccumulateAndCountBdrValues(coeff, vcoeff, attr, values_counter);
if (pfes->Conforming())
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
{
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
(*this)(i) = values(i)/values_counter[i];
}
(*this)(i) = values(i)/values_counter[i];
}
}
else
{
// TODO: is this the same as the conforming case (after the merge of
// cut-mesh-groups-dev)?
ComputeMeans(ARITHMETIC, values_counter);
}
#ifdef MFEM_DEBUG
Array<int> ess_vdofs_marker;
pfes->GetEssentialVDofs(attr, ess_vdofs_marker);
@@ -448,33 +442,27 @@ void ParGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
{
Array<int> values_counter;
AccumulateAndCountBdrTangentValues(vcoeff, bdr_attr, values_counter);
if (pfes->Conforming())
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
{
Vector values(Size());
for (int i = 0; i < values.Size(); i++)
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
values(i) = values_counter[i] ? (*this)(i) : 0.0;
}
// Count the values globally.
GroupCommunicator &gcomm = pfes->GroupComm();
gcomm.Reduce<int>(values_counter, GroupCommunicator::Sum);
// Accumulate the values globally.
gcomm.Reduce<double>(values, GroupCommunicator::Sum);
// Only the values in the master are guaranteed to be correct!
for (int i = 0; i < values.Size(); i++)
{
if (values_counter[i])
{
(*this)(i) = values(i)/values_counter[i];
}
(*this)(i) = values(i)/values_counter[i];
}
}
else
{
// TODO: is this the same as the conforming case (after the merge of
// cut-mesh-groups-dev)?
ComputeMeans(ARITHMETIC, values_counter);
}
#ifdef MFEM_DEBUG
Array<int> ess_vdofs_marker;
pfes->GetEssentialVDofs(bdr_attr, ess_vdofs_marker);
+9
View File
@@ -45,6 +45,15 @@ public:
/** The pointer @a pf is not owned by the newly constructed object. */
ParLinearForm(ParFiniteElementSpace *pf) : LinearForm(pf) { pfes = pf; }
/// Construct a ParLinearForm using previously allocated array @a data.
/** The ParLinearForm does not assume ownership of @a data which is assumed
to be of size at least `pf->GetVSize()`. Similar to the LinearForm and
Vector constructors for externally allocated array, the pointer @a data
can be NULL. The data array can be replaced later using the method
SetData(). */
ParLinearForm(ParFiniteElementSpace *pf, double *data) :
LinearForm(pf, data), pfes(pf) { }
/** @brief Create a ParLinearForm on the ParFiniteElementSpace @a *pf, using
the same integrators as the ParLinearForm @a *plf.
+2
View File
@@ -17,6 +17,8 @@
#ifdef MFEM_USE_SIDRE
#include "datacollection.hpp"
// Ignore warnings from the axom/sidre header (GCC + Clang versions)
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
# pragma GCC diagnostic push
# if defined(__clang__)
+2
View File
@@ -29,7 +29,9 @@ void AdvectorCG::SetInitialField(const Vector &init_nodes,
void AdvectorCG::ComputeAtNewPosition(const Vector &new_nodes,
Vector &new_field)
{
#if defined(MFEM_DEBUG) || defined(MFEM_USE_MPI)
int myid = 0;
#endif
Mesh *m = mesh;
#ifdef MFEM_USE_MPI
+95 -9
View File
@@ -12,6 +12,9 @@
#include "forall.hpp"
#include "cuda.hpp"
#include "occa.hpp"
#ifdef MFEM_USE_CEED
#include <ceed.h>
#endif
#include <string>
#include <map>
@@ -29,20 +32,26 @@ namespace internal
occa::device occaDevice;
#endif
#ifdef MFEM_USE_CEED
Ceed ceed = NULL;
#endif
// Backends listed by priority, high to low:
static const Backend::Id backend_list[Backend::NUM_BACKENDS] =
{
Backend::OCCA_CUDA, Backend::RAJA_CUDA, Backend::CUDA,
Backend::CEED_CUDA, Backend::OCCA_CUDA, Backend::RAJA_CUDA, Backend::CUDA,
Backend::HIP,
Backend::OCCA_OMP, Backend::RAJA_OMP, Backend::OMP,
Backend::OCCA_CPU, Backend::RAJA_CPU, Backend::CPU
Backend::CEED_CPU, Backend::OCCA_CPU, Backend::RAJA_CPU, Backend::CPU
};
// Backend names listed by priority, high to low:
static const char *backend_name[Backend::NUM_BACKENDS] =
{
"occa-cuda", "raja-cuda", "cuda", "hip", "occa-omp", "raja-omp", "omp",
"occa-cpu", "raja-cpu", "cpu"
"ceed-cuda", "occa-cuda", "raja-cuda", "cuda",
"hip",
"occa-omp", "raja-omp", "omp",
"ceed-cpu", "occa-cpu", "raja-cpu", "cpu"
};
} // namespace mfem::internal
@@ -54,7 +63,14 @@ Device Device::device_singleton;
Device::~Device()
{
if (destroy_mm) { mm.Destroy(); }
if (destroy_mm)
{
free(ceed_option);
#ifdef MFEM_USE_CEED
CeedDestroy(&internal::ceed);
#endif
mm.Destroy();
}
}
void Device::Configure(const std::string &device, const int dev)
@@ -64,15 +80,29 @@ void Device::Configure(const std::string &device, const int dev)
{
bmap[internal::backend_name[i]] = internal::backend_list[i];
}
std::string::size_type beg = 0, end;
std::string::size_type beg = 0, end, option;
while (1)
{
end = device.find(',', beg);
end = (end != std::string::npos) ? end : device.size();
const std::string bname = device.substr(beg, end - beg);
std::map<std::string, Backend::Id>::iterator it = bmap.find(bname);
MFEM_VERIFY(it != bmap.end(), "invalid backend name: '" << bname << '\'');
Get().MarkBackend(it->second);
option = bname.find(':');
if (option==std::string::npos) // No option
{
const std::string backend = bname;
std::map<std::string, Backend::Id>::iterator it = bmap.find(backend);
MFEM_VERIFY(it != bmap.end(), "invalid backend name: '" << backend << '\'');
Get().MarkBackend(it->second);
}
else
{
const std::string backend = bname.substr(0, option);
const std::string boption = bname.substr(option+1);
Get().ceed_option = strdup(boption.c_str());
std::map<std::string, Backend::Id>::iterator it = bmap.find(backend);
MFEM_VERIFY(it != bmap.end(), "invalid backend name: '" << backend << '\'');
Get().MarkBackend(it->second);
}
if (end == device.size()) { break; }
beg = end + 1;
}
@@ -82,6 +112,10 @@ void Device::Configure(const std::string &device, const int dev)
{
Get().MarkBackend(Backend::CUDA);
}
if (Allows(Backend::CEED_CUDA))
{
Get().MarkBackend(Backend::CUDA);
}
// Perform setup.
Get().Setup(dev);
@@ -110,6 +144,14 @@ void Device::Print(std::ostream &out)
}
}
out << '\n';
#ifdef MFEM_USE_CEED
if (Allows(Backend::CEED_MASK))
{
const char *ceed_backend;
CeedGetResource(internal::ceed, &ceed_backend);
out << "libCEED backend: " << ceed_backend << '\n';
}
#endif
}
void Device::UpdateMemoryTypeAndClass()
@@ -223,6 +265,21 @@ static void OccaDeviceSetup(const int dev)
#endif
}
static void CeedDeviceSetup(const char* ceed_spec)
{
#ifdef MFEM_USE_CEED
CeedInit(ceed_spec, &internal::ceed);
const char *ceed_backend;
CeedGetResource(internal::ceed, &ceed_backend);
if (strcmp(ceed_spec, ceed_backend) && strcmp(ceed_spec, "/cpu/self"))
{
mfem::out << std::endl << "WARNING!!!\n"
"libCEED is not using the requested backend!!!\n"
"WARNING!!!\n" << std::endl;
}
#endif
}
void Device::Setup(const int device)
{
MFEM_VERIFY(ngpu == -1, "the mfem::Device is already configured!");
@@ -245,12 +302,41 @@ void Device::Setup(const int device)
MFEM_VERIFY(!Allows(Backend::OMP|Backend::RAJA_OMP),
"the OpenMP and RAJA OpenMP backends require MFEM built with"
" MFEM_USE_OPENMP=YES");
#endif
#ifndef MFEM_USE_CEED
MFEM_VERIFY(!Allows(Backend::CEED_MASK),
"the CEED backends require MFEM built with MFEM_USE_CEED=YES");
#else
MFEM_VERIFY(!Allows(Backend::CEED_CPU) || !Allows(Backend::CEED_CUDA),
"Only one CEED backend can be enabled at a time!");
#endif
if (Allows(Backend::CUDA)) { CudaDeviceSetup(dev, ngpu); }
if (Allows(Backend::HIP)) { HipDeviceSetup(dev, ngpu); }
if (Allows(Backend::RAJA_CUDA)) { RajaDeviceSetup(dev, ngpu); }
// The check for MFEM_USE_OCCA is in the function OccaDeviceSetup().
if (Allows(Backend::OCCA_MASK)) { OccaDeviceSetup(dev); }
if (Allows(Backend::CEED_CPU))
{
if (!ceed_option)
{
CeedDeviceSetup("/cpu/self");
}
else
{
CeedDeviceSetup(ceed_option);
}
}
if (Allows(Backend::CEED_CUDA))
{
if (!ceed_option)
{
CeedDeviceSetup("/gpu/cuda/gen");
}
else
{
CeedDeviceSetup(ceed_option);
}
}
}
} // mfem
+24 -8
View File
@@ -54,7 +54,14 @@ struct Backend
OCCA_OMP = 1 << 8,
/** @brief [device] OCCA CUDA backend. Enabled when MFEM_USE_OCCA = YES
and MFEM_USE_CUDA = YES. */
OCCA_CUDA = 1 << 9
OCCA_CUDA = 1 << 9,
/** @brief [host] CEED CPU backend. GPU backends can still be used, but
with expensive memory transfers. Enabled when MFEM_USE_CEED = YES. */
CEED_CPU = 1 << 10,
/** @brief [device] CEED CUDA backend working in colaboration with the
CUDA backend. Enabled when MFEM_USE_CEED = YES and
MFEM_USE_CUDA = YES. */
CEED_CUDA = 1 << 11
};
/** @brief Additional useful constants. For example, the *_MASK constants can
@@ -62,16 +69,18 @@ struct Backend
enum
{
/// Number of backends: from (1 << 0) to (1 << (NUM_BACKENDS-1)).
NUM_BACKENDS = 10,
NUM_BACKENDS = 12,
/// Biwise-OR of all CPU backends
CPU_MASK = CPU | RAJA_CPU | OCCA_CPU,
CPU_MASK = CPU | RAJA_CPU | OCCA_CPU | CEED_CPU,
/// Biwise-OR of all CUDA backends
CUDA_MASK = CUDA | RAJA_CUDA | OCCA_CUDA,
CUDA_MASK = CUDA | RAJA_CUDA | OCCA_CUDA | CEED_CUDA,
/// Biwise-OR of all HIP backends
HIP_MASK = HIP,
/// Biwise-OR of all OpenMP backends
OMP_MASK = OMP | RAJA_OMP | OCCA_OMP,
/// Bitwise-OR of all CEED backends
CEED_MASK = CEED_CPU | CEED_CUDA,
/// Biwise-OR of all device backends
DEVICE_MASK = CUDA_MASK | HIP_MASK,
@@ -116,6 +125,7 @@ private:
MemoryType mem_type; ///< Current Device MemoryType
MemoryClass mem_class; ///< Current Device MemoryClass
char *ceed_option = NULL;
Device(Device const&);
void operator=(Device const&);
static Device& Get() { return device_singleton; }
@@ -181,13 +191,19 @@ public:
Backend::Id enumeration constant with '_' replaced by '-', e.g. the
string name of 'RAJA_CPU' is 'raja-cpu'.
* The 'cpu' backend is always enabled with lowest priority.
* The current backend priority from highest to lowest is: 'occa-cuda',
'raja-cuda', 'cuda', 'hip', 'occa-omp', 'raja-omp', 'omp', 'occa-cpu',
'raja-cpu', 'cpu'.
* The current backend priority from highest to lowest is: 'ceed-cuda',
'occa-cuda', 'raja-cuda', 'cuda', 'hip', 'occa-omp', 'raja-omp', 'omp',
'ceed-cpu', 'occa-cpu', 'raja-cpu', 'cpu'.
* Multiple backends can be configured at the same time.
* Only one 'occa-*' backend can be configured at a time.
* The backend 'occa-cuda' enables the 'cuda' backend unless 'raja-cuda'
is already enabled. */
is already enabled.
* The backend 'ceed-cpu' delegates to a libCEED CPU backend the setup and
evaluation of the operator.
* The backend 'ceed-cuda' delegates to a libCEED CUDA backend the setup
and evaluation of the operator and enables the 'cuda' backend to avoid
transfer between host and device.
*/
void Configure(const std::string &device, const int dev = 0);
/// Print the configuration of the MFEM virtual device object.
+3 -3
View File
@@ -90,9 +90,9 @@ std::string MakeParFilename(const std::string &prefix, const int myid,
Functions for getting and setting the MPI communicator used by the library
as the "global" communicator.
Currently, the MFEM "global" communicator is used only by the function
mfem_error(), invoked when an error is detected - the "global" communicator
is used as a parameter to MPI_Abort() to terminate all "global" tasks. */
This "global" communicator is used for example in the function mfem_error(),
which is invoked when an error is detected - the "global" communicator is
used as a parameter to MPI_Abort() to terminate all "global" tasks. */
///@{
/// Get MFEM's "global" MPI communicator.
+4 -1
View File
@@ -117,7 +117,10 @@ public:
required));
}
/** Parse the command-line options. Note that this function expects all the
options provided through the command line to have a corresponding
AddOption. In particular, this function cannot be used for partial
parsing. */
void Parse();
bool Good() const { return (error_type == 0); }
bool Help() const { return (error_type == 1); }
+12
View File
@@ -45,6 +45,8 @@ list(APPEND HDRS
tmatrix.hpp
ttensor.hpp
vector.hpp
tadvector.hpp
taddensemat.hpp
)
if (MFEM_USE_MPI)
@@ -64,6 +66,11 @@ if (MFEM_USE_MPI)
endif()
endif()
if (MFEM_USE_GINKGO)
list(APPEND SRCS ginkgo.cpp)
list(APPEND HDRS ginkgo.hpp)
endif()
if (MFEM_USE_SUNDIALS)
list(APPEND SRCS sundials.cpp)
list(APPEND HDRS sundials.hpp)
@@ -81,6 +88,11 @@ if (MFEM_USE_STRUMPACK)
list(APPEND HDRS strumpack.hpp)
endif()
if (MFEM_USE_HIOP)
list(APPEND SRCS hiop.cpp)
list(APPEND HDRS hiop.hpp)
endif()
convert_filenames_to_full_paths(SRCS)
convert_filenames_to_full_paths(HDRS)
+433 -9
View File
@@ -10,6 +10,28 @@
// Software Foundation) version 2.1 dated February 1999.
#include "complex_operator.hpp"
#include <set>
#include <map>
// Define macro wrappers for hypre_TAlloc, hypre_CTAlloc and hypre_TFree:
// mfem_hypre_TAlloc, mfem_hypre_CTAlloc, and mfem_hypre_TFree, respectively.
// Note: the same macros are defined in hypre.cpp and hypre_parser.cpp.
#if MFEM_HYPRE_VERSION < 21400
#define mfem_hypre_TAlloc(type, size) hypre_TAlloc(type, size)
#define mfem_hypre_CTAlloc(type, size) hypre_CTAlloc(type, size)
#define mfem_hypre_TFree(ptr) hypre_TFree(ptr)
#else // MFEM_HYPRE_VERSION >= 21400
// See the notes about hypre 2.14.0 in hypre.cpp
#define mfem_hypre_TAlloc(type, size) \
hypre_TAlloc(type, size, HYPRE_MEMORY_HOST)
#define mfem_hypre_CTAlloc(type, size) \
hypre_CTAlloc(type, size, HYPRE_MEMORY_HOST)
#define mfem_hypre_TFree(ptr) hypre_TFree(ptr, HYPRE_MEMORY_HOST)
#endif // #if MFEM_HYPRE_VERSION < 21400
namespace mfem
{
@@ -17,16 +39,17 @@ namespace mfem
ComplexOperator::ComplexOperator(Operator * Op_Real, Operator * Op_Imag,
bool ownReal, bool ownImag,
Convention convention)
: Operator(2*Op_Real->Height(), 2*Op_Real->Width())
: Operator(2*((Op_Real)?Op_Real->Height():Op_Imag->Height()),
2*((Op_Real)?Op_Real->Width():Op_Imag->Width()))
, Op_Real_(Op_Real)
, Op_Imag_(Op_Imag)
, ownReal_(ownReal)
, ownImag_(ownImag)
, convention_(convention)
, x_r_(NULL, Op_Real->Width())
, x_i_(NULL, Op_Real->Width())
, y_r_(NULL, Op_Real->Height())
, y_i_(NULL, Op_Real->Height())
, x_r_(NULL, width / 2)
, x_i_(NULL, width / 2)
, y_r_(NULL, height / 2)
, y_i_(NULL, height / 2)
, u_(NULL)
, v_(NULL)
{}
@@ -39,14 +62,38 @@ ComplexOperator::~ComplexOperator()
delete v_;
}
Operator & ComplexOperator::real()
{
MFEM_ASSERT(Op_Real_, "ComplexOperator has no real part!");
return *Op_Real_;
}
Operator & ComplexOperator::imag()
{
MFEM_ASSERT(Op_Imag_, "ComplexOperator has no imaginary part!");
return *Op_Imag_;
}
const Operator & ComplexOperator::real() const
{
MFEM_ASSERT(Op_Real_, "ComplexOperator has no real part!");
return *Op_Real_;
}
const Operator & ComplexOperator::imag() const
{
MFEM_ASSERT(Op_Imag_, "ComplexOperator has no imaginary part!");
return *Op_Imag_;
}
void ComplexOperator::Mult(const Vector &x, Vector &y) const
{
double * x_data = x.GetData();
x_r_.SetData(x_data);
x_i_.SetData(&x_data[Op_Real_->Width()]);
x_i_.SetData(&x_data[width / 2]);
y_r_.SetData(&y[0]);
y_i_.SetData(&y[Op_Real_->Height()]);
y_i_.SetData(&y[height / 2]);
this->Mult(x_r_, x_i_, y_r_, y_i_);
}
@@ -83,10 +130,10 @@ void ComplexOperator::MultTranspose(const Vector &x, Vector &y) const
{
double * x_data = x.GetData();
y_r_.SetData(x_data);
y_i_.SetData(&x_data[Op_Real_->Height()]);
y_i_.SetData(&x_data[height / 2]);
x_r_.SetData(&y[0]);
x_i_.SetData(&y[Op_Real_->Width()]);
x_i_.SetData(&y[width / 2]);
this->MultTranspose(y_r_, y_i_, x_r_, x_i_);
}
@@ -120,6 +167,30 @@ void ComplexOperator::MultTranspose(const Vector &x_r, const Vector &x_i,
}
SparseMatrix & ComplexSparseMatrix::real()
{
MFEM_ASSERT(Op_Real_, "ComplexSparseMatrix has no real part!");
return dynamic_cast<SparseMatrix &>(*Op_Real_);
}
SparseMatrix & ComplexSparseMatrix::imag()
{
MFEM_ASSERT(Op_Imag_, "ComplexSparseMatrix has no imaginary part!");
return dynamic_cast<SparseMatrix &>(*Op_Imag_);
}
const SparseMatrix & ComplexSparseMatrix::real() const
{
MFEM_ASSERT(Op_Real_, "ComplexSparseMatrix has no real part!");
return dynamic_cast<const SparseMatrix &>(*Op_Real_);
}
const SparseMatrix & ComplexSparseMatrix::imag() const
{
MFEM_ASSERT(Op_Imag_, "ComplexSparseMatrix has no imaginary part!");
return dynamic_cast<const SparseMatrix &>(*Op_Imag_);
}
SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
{
SparseMatrix * A_r = dynamic_cast<SparseMatrix*>(Op_Real_);
@@ -184,4 +255,357 @@ SparseMatrix * ComplexSparseMatrix::GetSystemMatrix() const
return new SparseMatrix(I, J, D, this->Height(), this->Width());
}
#ifdef MFEM_USE_MPI
ComplexHypreParMatrix::ComplexHypreParMatrix(HypreParMatrix * A_Real,
HypreParMatrix * A_Imag,
bool ownReal, bool ownImag,
Convention convention)
: ComplexOperator(A_Real, A_Imag, ownReal, ownImag, convention)
{
comm_ = (A_Real) ? A_Real->GetComm() :
((A_Imag) ? A_Imag->GetComm() : MPI_COMM_WORLD);
MPI_Comm_rank(comm_, &myid_);
MPI_Comm_size(comm_, &nranks_);
}
HypreParMatrix & ComplexHypreParMatrix::real()
{
MFEM_ASSERT(Op_Real_, "ComplexHypreParMatrix has no real part!");
return dynamic_cast<HypreParMatrix &>(*Op_Real_);
}
HypreParMatrix & ComplexHypreParMatrix::imag()
{
MFEM_ASSERT(Op_Imag_, "ComplexHypreParMatrix has no imaginary part!");
return dynamic_cast<HypreParMatrix &>(*Op_Imag_);
}
const HypreParMatrix & ComplexHypreParMatrix::real() const
{
MFEM_ASSERT(Op_Real_, "ComplexHypreParMatrix has no real part!");
return dynamic_cast<const HypreParMatrix &>(*Op_Real_);
}
const HypreParMatrix & ComplexHypreParMatrix::imag() const
{
MFEM_ASSERT(Op_Imag_, "ComplexHypreParMatrix has no imaginary part!");
return dynamic_cast<const HypreParMatrix &>(*Op_Imag_);
}
HypreParMatrix * ComplexHypreParMatrix::GetSystemMatrix() const
{
HypreParMatrix * A_r = dynamic_cast<HypreParMatrix*>(Op_Real_);
HypreParMatrix * A_i = dynamic_cast<HypreParMatrix*>(Op_Imag_);
if ( A_r == NULL && A_i == NULL ) { return NULL; }
HYPRE_Int global_num_rows_r = (A_r) ? A_r->GetGlobalNumRows() : 0;
HYPRE_Int global_num_rows_i = (A_i) ? A_i->GetGlobalNumRows() : 0;
HYPRE_Int global_num_rows = std::max(global_num_rows_r, global_num_rows_i);
HYPRE_Int global_num_cols_r = (A_r) ? A_r->GetGlobalNumCols() : 0;
HYPRE_Int global_num_cols_i = (A_i) ? A_i->GetGlobalNumCols() : 0;
HYPRE_Int global_num_cols = std::max(global_num_cols_r, global_num_cols_i);
int row_starts_size = (HYPRE_AssumedPartitionCheck()) ? 2 : nranks_ + 1;
HYPRE_Int * row_starts = mfem_hypre_CTAlloc(HYPRE_Int, row_starts_size);
HYPRE_Int * col_starts = mfem_hypre_CTAlloc(HYPRE_Int, row_starts_size);
const HYPRE_Int * row_starts_z = (A_r) ? A_r->RowPart() :
((A_i) ? A_i->RowPart() : NULL);
const HYPRE_Int * col_starts_z = (A_r) ? A_r->ColPart() :
((A_i) ? A_i->ColPart() : NULL);
for (int i = 0; i < row_starts_size; i++)
{
row_starts[i] = 2 * row_starts_z[i];
col_starts[i] = 2 * col_starts_z[i];
}
SparseMatrix diag_r, diag_i, offd_r, offd_i;
HYPRE_Int * cmap_r, * cmap_i;
int nrows_r = 0, nrows_i = 0, ncols_r = 0, ncols_i = 0;
int ncols_offd_r = 0, ncols_offd_i = 0;
if (A_r)
{
A_r->GetDiag(diag_r);
A_r->GetOffd(offd_r, cmap_r);
nrows_r = diag_r.Height();
ncols_r = diag_r.Width();
ncols_offd_r = offd_r.Width();
}
if (A_i)
{
A_i->GetDiag(diag_i);
A_i->GetOffd(offd_i, cmap_i);
nrows_i = diag_i.Height();
ncols_i = diag_i.Width();
ncols_offd_i = offd_i.Width();
}
int nrows = std::max(nrows_r, nrows_i);
int ncols = std::max(ncols_r, ncols_i);
// Determine the unique set of off-diagonal columns global indices
std::set<int> cset;
for (int i=0; i<ncols_offd_r; i++)
{
cset.insert(cmap_r[i]);
}
for (int i=0; i<ncols_offd_i; i++)
{
cset.insert(cmap_i[i]);
}
int num_cols_offd = (int)cset.size();
// Extract pointers to the various CSR arrays of the diagonal blocks
const int * diag_r_I = (A_r) ? diag_r.GetI() : NULL;
const int * diag_i_I = (A_i) ? diag_i.GetI() : NULL;
const int * diag_r_J = (A_r) ? diag_r.GetJ() : NULL;
const int * diag_i_J = (A_i) ? diag_i.GetJ() : NULL;
const double * diag_r_D = (A_r) ? diag_r.GetData() : NULL;
const double * diag_i_D = (A_i) ? diag_i.GetData() : NULL;
int diag_r_nnz = (diag_r_I) ? diag_r_I[nrows] : 0;
int diag_i_nnz = (diag_i_I) ? diag_i_I[nrows] : 0;
int diag_nnz = 2 * (diag_r_nnz + diag_i_nnz);
// Extract pointers to the various CSR arrays of the off-diagonal blocks
const int * offd_r_I = (A_r) ? offd_r.GetI() : NULL;
const int * offd_i_I = (A_i) ? offd_i.GetI() : NULL;
const int * offd_r_J = (A_r) ? offd_r.GetJ() : NULL;
const int * offd_i_J = (A_i) ? offd_i.GetJ() : NULL;
const double * offd_r_D = (A_r) ? offd_r.GetData() : NULL;
const double * offd_i_D = (A_i) ? offd_i.GetData() : NULL;
int offd_r_nnz = (offd_r_I) ? offd_r_I[nrows] : 0;
int offd_i_nnz = (offd_i_I) ? offd_i_I[nrows] : 0;
int offd_nnz = 2 * (offd_r_nnz + offd_i_nnz);
// Allocate CSR arrays for the combined matrix
HYPRE_Int * diag_I = mfem_hypre_CTAlloc(HYPRE_Int, 2 * nrows + 1);
HYPRE_Int * diag_J = mfem_hypre_CTAlloc(HYPRE_Int, diag_nnz);
double * diag_D = mfem_hypre_CTAlloc(double, diag_nnz);
HYPRE_Int * offd_I = mfem_hypre_CTAlloc(HYPRE_Int, 2 * nrows + 1);
HYPRE_Int * offd_J = mfem_hypre_CTAlloc(HYPRE_Int, offd_nnz);
double * offd_D = mfem_hypre_CTAlloc(double, offd_nnz);
HYPRE_Int * cmap = mfem_hypre_CTAlloc(HYPRE_Int, 2 * num_cols_offd);
// Fill the CSR arrays for the diagonal portion of the matrix
const double factor = (convention_ == HERMITIAN) ? 1.0 : -1.0;
diag_I[0] = 0;
diag_I[nrows] = diag_r_nnz + diag_i_nnz;
for (int i=0; i<nrows; i++)
{
diag_I[i + 1] = ((diag_r_I)?diag_r_I[i+1]:0) +
((diag_i_I)?diag_i_I[i+1]:0);
diag_I[i + nrows + 1] = diag_I[i+1] + diag_r_nnz + diag_i_nnz;
if (diag_r_I)
{
for (int j=0; j<diag_r_I[i+1] - diag_r_I[i]; j++)
{
diag_J[diag_I[i] + j] = diag_r_J[diag_r_I[i] + j];
diag_D[diag_I[i] + j] = diag_r_D[diag_r_I[i] + j];
diag_J[diag_I[i+nrows] + j] =
diag_r_J[diag_r_I[i] + j] + ncols;
diag_D[diag_I[i+nrows] + j] =
factor * diag_r_D[diag_r_I[i] + j];
}
}
if (diag_i_I)
{
const int off_r = (diag_r_I)?(diag_r_I[i+1] - diag_r_I[i]):0;
for (int j=0; j<diag_i_I[i+1] - diag_i_I[i]; j++)
{
diag_J[diag_I[i] + off_r + j] = diag_i_J[diag_i_I[i] + j] + ncols;
diag_D[diag_I[i] + off_r + j] = -diag_i_D[diag_i_I[i] + j];
diag_J[diag_I[i+nrows] + off_r + j] = diag_i_J[diag_i_I[i] + j];
diag_D[diag_I[i+nrows] + off_r + j] =
factor * diag_i_D[diag_i_I[i] + j];
}
}
}
// Determine the mappings describing the layout of off-diagonal columns
int num_recv_procs = 0;
HYPRE_Int * offd_col_start_stop = NULL;
this->getColStartStop(A_r, A_i, num_recv_procs, offd_col_start_stop);
std::set<int>::iterator sit;
std::map<int,int> cmapa, cmapb, cinvmap;
for (sit=cset.begin(); sit!=cset.end(); sit++)
{
int col_orig = *sit;
int col_2x2 = -1;
int col_size = 0;
for (int i=0; i<num_recv_procs; i++)
{
if (offd_col_start_stop[2*i] <= col_orig &&
col_orig < offd_col_start_stop[2*i+1])
{
col_2x2 = offd_col_start_stop[2*i] + col_orig;
col_size = offd_col_start_stop[2*i+1] - offd_col_start_stop[2*i];
break;
}
}
cmapa[*sit] = col_2x2;
cmapb[*sit] = col_2x2 + col_size;
cinvmap[col_2x2] = -1;
cinvmap[col_2x2 + col_size] = -1;
}
delete [] offd_col_start_stop;
std::map<int, int>::iterator mit;
int i = 0;
for (mit=cinvmap.begin(); mit!=cinvmap.end(); mit++, i++)
{
mit->second = i;
cmap[i] = mit->first;
}
// Fill the CSR arrays for the off-diagonal portion of the matrix
offd_I[0] = 0;
offd_I[nrows] = offd_r_nnz + offd_i_nnz;
for (int i=0; i<nrows; i++)
{
offd_I[i + 1] = ((offd_r_I)?offd_r_I[i+1]:0) +
((offd_i_I)?offd_i_I[i+1]:0);
offd_I[i + nrows + 1] = offd_I[i+1] + offd_r_nnz + offd_i_nnz;
if (offd_r_I)
{
const int off_i = (offd_i_I)?(offd_i_I[i+1] - offd_i_I[i]):0;
for (int j=0; j<offd_r_I[i+1] - offd_r_I[i]; j++)
{
offd_J[offd_I[i] + j] =
cinvmap[cmapa[cmap_r[offd_r_J[offd_r_I[i] + j]]]];
offd_D[offd_I[i] + j] = offd_r_D[offd_r_I[i] + j];
offd_J[offd_I[i+nrows] + off_i + j] =
cinvmap[cmapb[cmap_r[offd_r_J[offd_r_I[i] + j]]]];
offd_D[offd_I[i+nrows] + off_i + j] =
factor * offd_r_D[offd_r_I[i] + j];
}
}
if (offd_i_I)
{
const int off_r = (offd_r_I)?(offd_r_I[i+1] - offd_r_I[i]):0;
for (int j=0; j<offd_i_I[i+1] - offd_i_I[i]; j++)
{
offd_J[offd_I[i] + off_r + j] =
cinvmap[cmapb[cmap_i[offd_i_J[offd_i_I[i] + j]]]];
offd_D[offd_I[i] + off_r + j] = -offd_i_D[offd_i_I[i] + j];
offd_J[offd_I[i+nrows] + j] =
cinvmap[cmapa[cmap_i[offd_i_J[offd_i_I[i] + j]]]];
offd_D[offd_I[i+nrows] + j] = factor * offd_i_D[offd_i_I[i] + j];
}
}
}
// Construct the combined matrix
HypreParMatrix * A = new HypreParMatrix(comm_,
2 * global_num_rows,
2 * global_num_cols,
row_starts, col_starts,
diag_I, diag_J, diag_D,
offd_I, offd_J, offd_D,
2 * num_cols_offd, cmap);
// Give the new matrix ownership of its internal arrays
A->SetOwnerFlags(-1,-1,-1);
hypre_CSRMatrixSetDataOwner(((hypre_ParCSRMatrix*)(*A))->diag,1);
hypre_CSRMatrixSetDataOwner(((hypre_ParCSRMatrix*)(*A))->offd,1);
hypre_ParCSRMatrixSetRowStartsOwner((hypre_ParCSRMatrix*)(*A),1);
hypre_ParCSRMatrixSetColStartsOwner((hypre_ParCSRMatrix*)(*A),1);
return A;
}
void
ComplexHypreParMatrix::getColStartStop(const HypreParMatrix * A_r,
const HypreParMatrix * A_i,
int & num_recv_procs,
HYPRE_Int *& offd_col_start_stop) const
{
hypre_ParCSRCommPkg * comm_pkg_r =
(A_r) ? hypre_ParCSRMatrixCommPkg((hypre_ParCSRMatrix*)(*A_r)) : NULL;
hypre_ParCSRCommPkg * comm_pkg_i =
(A_i) ? hypre_ParCSRMatrixCommPkg((hypre_ParCSRMatrix*)(*A_i)) : NULL;
std::set<HYPRE_Int> send_procs, recv_procs;
if ( comm_pkg_r )
{
for (HYPRE_Int i=0; i<comm_pkg_r->num_sends; i++)
{
send_procs.insert(comm_pkg_r->send_procs[i]);
}
for (HYPRE_Int i=0; i<comm_pkg_r->num_recvs; i++)
{
recv_procs.insert(comm_pkg_r->recv_procs[i]);
}
}
if ( comm_pkg_i )
{
for (HYPRE_Int i=0; i<comm_pkg_i->num_sends; i++)
{
send_procs.insert(comm_pkg_i->send_procs[i]);
}
for (HYPRE_Int i=0; i<comm_pkg_i->num_recvs; i++)
{
recv_procs.insert(comm_pkg_i->recv_procs[i]);
}
}
num_recv_procs = (int)recv_procs.size();
HYPRE_Int loc_start_stop[2];
offd_col_start_stop = new HYPRE_Int[2 * num_recv_procs];
const HYPRE_Int * row_part = (A_r) ? A_r->RowPart() :
((A_i) ? A_i->RowPart() : NULL);
int row_part_ind = (HYPRE_AssumedPartitionCheck()) ? 0 : myid_;
loc_start_stop[0] = row_part[row_part_ind];
loc_start_stop[1] = row_part[row_part_ind+1];
MPI_Request * req = new MPI_Request[send_procs.size()+recv_procs.size()];
MPI_Status * stat = new MPI_Status[send_procs.size()+recv_procs.size()];
int send_count = 0;
int recv_count = 0;
int tag = 0;
std::set<HYPRE_Int>::iterator sit;
for (sit=send_procs.begin(); sit!=send_procs.end(); sit++)
{
MPI_Isend(loc_start_stop, 2, HYPRE_MPI_INT,
*sit, tag, comm_, &req[send_count]);
send_count++;
}
for (sit=recv_procs.begin(); sit!=recv_procs.end(); sit++)
{
MPI_Irecv(&offd_col_start_stop[2*recv_count], 2, HYPRE_MPI_INT,
*sit, tag, comm_, &req[send_count+recv_count]);
recv_count++;
}
MPI_Waitall(send_count+recv_count, req, stat);
delete [] req;
delete [] stat;
}
#endif // MFEM_USE_MPI
}
+108 -21
View File
@@ -14,6 +14,9 @@
#include "operator.hpp"
#include "sparsemat.hpp"
#ifdef MFEM_USE_MPI
#include "hypre.hpp"
#endif
namespace mfem
{
@@ -27,7 +30,8 @@ namespace mfem
ComplexOperator allows one to choose a convention upon construction, which
facilitates symmetry.
Matrix-vector products are then computed as:
If we let (y_r + i y_i) = (Op_r + i Op_i)(x_r + i x_i) then Matrix-vector
products are computed as:
1. When Convention::HERMITIAN is used (default)
/ y_r \ / Op_r -Op_i \ / x_r \
@@ -38,24 +42,25 @@ namespace mfem
/ y_r \ / Op_r -Op_i \ / x_r \
| | = | | | |
\-y_i / \-Op_i -Op_r / \ x_i /
In other words, Matrix-vector products with Convention::BLOCK_SYMMETRIC
compute the complex conjugate of Op*x.
Either convention can be used with a given complex operator,
however, each of them is best suited for certain classes of
problems. For example:
Either convention can be used with a given complex operator, however, each
of them may be best suited for different classes of problems. For example:
1. Convention::HERMITIAN, is well suited for Hermitian operators,
i.e. operators where the real part is symmetric and the imaginary part of
the operator is anti-symmetric, hence the name. In such cases the resulting
2 x 2 operator will be symmetric.
1. Convention::HERMITIAN, is well suited for Hermitian operators, i.e.,
operators where the real part is symmetric and the imaginary part of the
operator is anti-symmetric, hence the name. In such cases the resulting 2
x 2 operator will be symmetric.
2. Convention::BLOCK_SYMMETRIC, is well suited for operators where both the
real and imaginary parts are symmetric. In this case the resulting 2 x 2
operator will again be symmetric. Such operators are common when studying
damped oscillations, for example.
real and imaginary parts are symmetric. In this case the resulting 2 x 2
operator will also be symmetric. Such operators are common when studying
damped oscillations, for example.
Note: this class cannot be used to represent a general nonlinear complex
operator.
*/
*/
class ComplexOperator : public Operator
{
public:
@@ -67,14 +72,12 @@ public:
/** @brief Constructs complex operator object
Note that either @p Op_Real or @p Op_Imag can be NULL,
thus eliminating their action (see documentation of the
class for more details).
Note that either @p Op_Real or @p Op_Imag can be NULL, thus eliminating
their action (see documentation of the class for more details).
In case ownership of the passed operator is transferred
to this class through @p ownReal and @p ownImag,
the operators will be explicitly destroyed at the end
of the life of this object.
In case ownership of the passed operator is transferred to this class
through @p ownReal and @p ownImag, the operators will be explicitly
destroyed at the end of the life of this object.
*/
ComplexOperator(Operator * Op_Real, Operator * Op_Imag,
bool ownReal, bool ownImag,
@@ -82,9 +85,30 @@ public:
virtual ~ComplexOperator();
/** @brief Check for existence of real or imaginary part of the operator
These methods do not check that the operators are non-zero but only that
the operators have been set.
*/
bool hasRealPart() const { return Op_Real_ != NULL; }
bool hasImagPart() const { return Op_Imag_ != NULL; }
/** @brief Real or imaginary part accessor methods
The following accessor methods should only be called if the requested
part of the opertor is known to exist. This can be checked with
hasRealPart() or hasImagPart().
*/
virtual Operator & real();
virtual Operator & imag();
virtual const Operator & real() const;
virtual const Operator & imag() const;
virtual void Mult(const Vector &x, Vector &y) const;
virtual void MultTranspose(const Vector &x, Vector &y) const;
virtual Type GetType() const { return Complex_Operator; }
protected:
// Let this be hidden from the public interface since the implementation
// depends on internal members
@@ -116,7 +140,7 @@ protected:
require access to the CSR matrix data such as SuperLU, STRUMPACK, or similar
sparse linear solvers.
See ComplexOperator documentation in operator.hpp for more information.
See ComplexOperator documentation above for more information.
*/
class ComplexSparseMatrix : public ComplexOperator
{
@@ -127,9 +151,72 @@ public:
: ComplexOperator(A_Real, A_Imag, ownReal, ownImag, convention)
{}
virtual SparseMatrix & real();
virtual SparseMatrix & imag();
virtual const SparseMatrix & real() const;
virtual const SparseMatrix & imag() const;
/** Combine the blocks making up this complex operator into a single
SparseMatrix. The resulting matrix can be passed to solvers which require
access to the matrix entries themselves, such as sparse direct solvers,
rather than simply the action of the opertor. Note that this combined
operator requires roughly twice the memory of the block structured
operator. */
SparseMatrix * GetSystemMatrix() const;
virtual Type GetType() const { return MFEM_ComplexSparseMat; }
};
#ifdef MFEM_USE_MPI
/** @brief Specialization of the ComplexOperator built from a pair of
HypreParMatrices.
The purpose of this specialization is to construct a single HypreParMatrix
object which is equivalent to the 2x2 block system that the ComplexOperator
mimics. The resulting HypreParMatrix can then be passed along to solvers
which require access to the CSR matrix data such as SuperLU, STRUMPACK, or
similar sparse linear solvers.
See ComplexOperator documentation above for more information.
*/
class ComplexHypreParMatrix : public ComplexOperator
{
public:
ComplexHypreParMatrix(HypreParMatrix * A_Real, HypreParMatrix * A_Imag,
bool ownReal, bool ownImag,
Convention convention = HERMITIAN);
virtual HypreParMatrix & real();
virtual HypreParMatrix & imag();
virtual const HypreParMatrix & real() const;
virtual const HypreParMatrix & imag() const;
/** Combine the blocks making up this complex operator into a single
HypreParMatrix. The resulting matrix can be passed to solvers which
require access to the matrix entries themselves, such as sparse direct
solvers or Hypre preconditioners, rather than simply the action of the
opertor. Note that this combined operator requires roughly twice the
memory of the block structured operator. */
HypreParMatrix * GetSystemMatrix() const;
virtual Type GetType() const { return Complex_Hypre_ParCSR; }
private:
void getColStartStop(const HypreParMatrix * A_r,
const HypreParMatrix * A_i,
int & num_recv_procs,
HYPRE_Int *& offd_col_start_stop) const;
MPI_Comm comm_;
int myid_;
int nranks_;
};
#endif // MFEM_USE_MPI
}
#endif
#endif // MFEM_COMPLEX_OPERATOR
+30 -1
View File
@@ -716,7 +716,12 @@ public:
/// Sets the tensor elements equal to constant c
DenseTensor &operator=(double c);
DenseMatrix &operator()(int k) { Mk.data = GetData(k); return Mk; }
DenseMatrix &operator()(int k)
{
MFEM_ASSERT_INDEX_IN_RANGE(k, 0, SizeK());
Mk.data = GetData(k);
return Mk;
}
const DenseMatrix &operator()(int k) const
{ return const_cast<DenseTensor&>(*this)(k); }
@@ -758,6 +763,30 @@ public:
long MemoryUsage() const { return nk*Mk.MemoryUsage(); }
/// Shortcut for mfem::Read( GetMemory(), TotalSize(), on_dev).
const double *Read(bool on_dev = true) const
{ return mfem::Read(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
/// Shortcut for mfem::Read(GetMemory(), TotalSize(), false).
const double *HostRead() const
{ return mfem::Read(tdata, Mk.Height()*Mk.Width()*nk, false); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), on_dev).
double *Write(bool on_dev = true)
{ return mfem::Write(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), false).
double *HostWrite()
{ return mfem::Write(tdata, Mk.Height()*Mk.Width()*nk, false); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), on_dev).
double *ReadWrite(bool on_dev = true)
{ return mfem::ReadWrite(tdata, Mk.Height()*Mk.Width()*nk, on_dev); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), false).
double *HostReadWrite()
{ return mfem::ReadWrite(tdata, Mk.Height()*Mk.Width()*nk, false); }
~DenseTensor() { tdata.Delete(); }
};

Some files were not shown because too many files have changed in this diff Show More