Compare commits

..
Author SHA1 Message Date
Hennes Hajduk 6497f4d829 minor 2020-11-05 10:02:01 +01:00
Hennes Hajduk 3b533595b8 save-animation.glvs 2020-11-05 09:36:18 +01:00
HennesHajduk d21491f91e burgers something 2020-11-03 08:28:35 +01:00
HennesHajduk 0a1f3a0b0a minor, new script 2020-10-30 16:07:07 +01:00
Hennes Hajduk 438a9968ce benchmark data and glvis scripts 2020-10-29 19:47:07 +01:00
Hennes Hajduk b1a343f514 updated scripts, changed initial condition for stationary burgers 2020-10-29 16:04:40 +01:00
HennesHajduk a0a29e3303 overestimated burgers wave speed. Implemented lumped L2 projection. minor. 2020-10-28 18:24:34 +01:00
HennesHajduk a921f56177 minor 2020-10-27 11:57:31 +01:00
HennesHajduk 2a6edf7d43 style 2020-10-26 11:38:39 +01:00
HennesHajduk d6fa604631 started on fixing advection 2020-10-26 11:38:06 +01:00
HennesHajduk 3add636a18 removed some absolute paths 2020-10-25 11:45:44 +01:00
HennesHajduk 21000a319e removed MoST from euler 2020-10-21 13:41:06 +02:00
HennesHajduk 95703e3a6d new test cases. 2020-10-21 13:32:47 +02:00
HennesHajduk 39dfcac5b3 config/burgers.sh 2020-10-21 13:27:51 +02:00
HennesHajduk 45b618fab2 changed mesh back, enabled warning 2020-10-08 21:47:08 +02:00
HennesHajduk 22838a13af minor 2020-10-08 16:33:23 +02:00
HennesHajduk ee58c6828a changed mesh 2020-10-08 11:35:19 +02:00
HennesHajduk b50b0c0ac1 . 2020-10-06 21:55:35 +02:00
HennesHajduk 6c40560766 disabled warning 2020-10-06 21:54:07 +02:00
Hennes Hajduk cadc4516ef merge 2020-10-06 16:51:36 +02:00
Hennes Hajduk bb584f04f8 minor 2020-10-06 16:51:01 +02:00
HennesHajduk cec500bb11 most gimmick for swe 2020-10-02 17:03:14 +02:00
HennesHajduk a88917276a changed default quadrature rule (gridfunc) + config, BL equation 2020-10-02 11:24:04 +02:00
HennesHajduk 2227958c30 update 2020-08-13 17:06:45 +02:00
HennesHajduk 8c0d8c3708 config, meshes, hll, wip 2020-08-07 19:03:18 +02:00
HennesHajduk d9c7e28444 wip 2020-08-03 18:27:10 +02:00
HennesHajduk 5dd725f5f0 wip. done with apps directory 2020-08-03 17:23:17 +02:00
HennesHajduk 9405bc04b3 rescaled some tests for euler, swe, improved grids, corrected double-mach boundary type, wip 2020-07-30 19:16:02 +02:00
HennesHajduk 6ddaa46fc2 updated apps subdir. 2020-07-30 12:32:47 +02:00
HennesHajduk 4400dc08ae wip, some cleanup, wave speed with abs of height/density now. 2020-07-29 18:25:07 +02:00
HennesHajduk e90095d854 wip, todo swe euler cleanup 2020-07-27 18:00:02 +02:00
HennesHajduk c1154025e3 wip, BL, some changes in design of HyperbolicSystem's routines. 2020-07-27 14:07:18 +02:00
HennesHajduk fcfe3c8353 update. most TODOs in apps. Missing: advection with general velocity. 2020-07-27 10:37:19 +02:00
HennesHajduk 143da8287a derived quantities, save instead SaveAsOne, output dir + minor 2020-07-06 17:52:18 +02:00
HennesHajduk 23d0f0b637 with git pull 2020-07-05 12:15:36 +02:00
HennesHajduk 08af14793b merge master 2020-07-05 12:13:31 +02:00
HennesHajduk 6b6b871999 merge 2020-07-05 12:05:23 +02:00
HennesHajduk 02d9c69c0d BL equation + update of scripts and configs 2020-07-05 12:00:51 +02:00
Hennes Hajduk efd7aa8686 minor 2020-06-30 16:48:09 +02:00
Hennes Hajduk 4abec105c3 change in config 2020-06-23 18:05:39 +02:00
HennesHajduk 723f891e37 bugfix, change in config 2020-06-23 10:04:01 +02:00
HennesHajduk 889a7598f4 minor 2020-06-20 10:26:23 +02:00
Hennes Hajduk 6a76844b4f sequential bound version that seems to work and converge. 2020-06-18 19:29:58 +02:00
Hennes Hajduk d8550d7309 ex15 2020-06-17 10:27:25 +02:00
HennesHajduk a86534bba6 . 2020-06-12 18:37:57 +02:00
Hennes Hajduk a01a4ace7a data 2020-06-12 18:33:18 +02:00
HennesHajduk 070cdb3c6a wip on bounds, reincluded all bar states for height 2020-06-10 18:10:44 +02:00
Hennes Hajduk 13818643c1 gresho, not quite working yet. some changes in mcl. 2020-06-09 19:05:50 +02:00
HennesHajduk 61f60d2f88 denomminator free implementation again. WIP on chosing the right bounds. 2020-06-09 15:17:14 +02:00
HennesHajduk 3c1cf15c04 minor 2020-06-09 15:04:46 +02:00
HennesHajduk 6624e2250c removed results from github 2020-06-03 16:04:39 +02:00
HennesHajduk eff0a9c79f no actual changes, just comments, scripts, gitignore 2020-05-29 11:06:27 +02:00
HennesHajduk b814cf96d5 Bugfix bounds - extended bounds by closest nbrs for first unknown. Double Mach works. 2020-05-22 17:50:51 +02:00
HennesHajduk 5ff5022af3 updated bounds as I believe it makes sense. TODO find differnce between serial and parallel 2020-05-22 13:39:41 +02:00
HennesHajduk 3623bd2994 bugfix bounds for scalar problems 2020-05-20 15:52:22 +02:00
HennesHajduk cccbb11057 WIP. Sequential limiting for volume and flux terms. Bounds work fine for problems with shocks. 2020-05-20 14:40:38 +02:00
HennesHajduk 0de495cc3c GMS for Euler - just in case. 2020-05-15 18:35:25 +02:00
HennesHajduk c95db258bd added min max with zero for face terms, abort check in 1D. Included bar states in bounds for volume terms. Using bar states instead of w's + minor. 2020-05-13 11:04:05 +02:00
HennesHajduk 2aa2759192 closest nbrs for cubes 2020-05-08 12:04:36 +02:00
HennesHajduk 1fa4323498 minor 2020-05-07 18:28:03 +02:00
HennesHajduk fde74a329b restructured fe_evol methods slightly 2020-05-07 17:28:25 +02:00
HennesHajduk 319d4e7f05 minor 2020-05-07 16:58:01 +02:00
HennesHajduk ecb9d77a29 Restructured and optimized MCL Evolution. 2020-05-07 16:48:25 +02:00
HennesHajduk c7d5ee2654 Restructering of apps, idea of rescaling t in config, new swe dam break test case. 2020-05-07 16:39:38 +02:00
HennesHajduk a0e0c15913 included bound classes 2020-05-06 18:30:24 +02:00
HennesHajduk 4fda37f739 New bounds classes - functionality that was previously in dofs. WIP. Bounds are computed correctly in serial and parallel. 2020-05-06 18:01:15 +02:00
HennesHajduk 7b5e078731 wip 2020-05-05 18:38:48 +02:00
HennesHajduk 7511d65aa9 minor 2020-05-05 17:40:47 +02:00
HennesHajduk 18adf880d7 updated (p)dofs. 2020-05-05 17:28:02 +02:00
HennesHajduk 8df51cbc3f Fixed DG flux term limiting 2020-05-05 13:27:48 +02:00
HennesHajduk a8634a7fa0 MCL almost done. Works for all considered element types. Currently using low order scheme for DG fluxes, other TODO: Bound computation for systems 2020-05-04 18:32:52 +02:00
HennesHajduk 0541c74f08 bugfix - proper bounds 2020-05-04 17:29:18 +02:00
HennesHajduk 95f77d6c9a new burgers test, new functionality in dofs (Q-spaces seem to work, bug for triangles). 2020-05-04 16:09:52 +02:00
HennesHajduk f469d54afa MCL limiting for scalars works now. 2020-04-28 17:16:37 +02:00
HennesHajduk b4951bd02d TriangleDofMap + minor 2020-04-28 17:14:07 +02:00
HennesHajduk f5137a5eed Improved MCL low order scheme 2020-04-27 11:40:14 +02:00
HennesHajduk 9ad28ce2c8 dof2LocNbr in MCL, new euler test cases or modified them, low order method used in MCL 2020-04-27 10:09:31 +02:00
HennesHajduk 92aa5d65e2 Changed AntiDiff to DenseTensor. Moved LORMassMat and added option for it. 2020-04-24 16:08:05 +02:00
HennesHajduk d4c7b609b1 updated burgers.sh, subcell-distribution works, bound computation for scalars. 2020-04-24 13:47:54 +02:00
HennesHajduk 3c53e9a9c5 Implemented ElFlux differently. 2020-04-23 17:28:21 +02:00
HennesHajduk fb9726d4f8 all anti-diffusive fluxes are now working. 2020-04-23 16:30:34 +02:00
HennesHajduk 84bdebeb8c Changed advection. For now only constant velocity fields are supported. New test case in swe. Some TODOs left. 2020-04-23 16:09:25 +02:00
HennesHajduk 8448b32bb1 Changed order of parameters and orientation of BdrDofs for triangles. 2020-04-23 15:56:41 +02:00
HennesHajduk 5bf4712efd wip, updated test.sh 2020-04-23 15:28:10 +02:00
HennesHajduk 1d571c04a8 split integration weight from BdrTerms, WIP on ADF. HO volume fluxes ready. 2020-04-17 11:50:19 +02:00
HennesHajduk 4aec552eeb merge master 2020-04-14 13:33:55 +02:00
HennesHajduk a9c9808b49 wip on ADFs: galerkin and low order method can be reccovered properly. TODO all the DG flux terms and ADFs gi 2020-04-10 12:10:32 +02:00
HennesHajduk 0aff563810 revert to not overriding routines in fe_evol classes. 2020-04-10 10:17:57 +02:00
HennesHajduk a3ce5422d9 wip on anti-diffusive fluxes 2020-04-10 09:58:15 +02:00
HennesHajduk 2f3010f2f0 wip on mcl, optimized problem dependent fluxes, mass error computed in terms of first unknown. 2020-04-09 14:42:58 +02:00
HennesHajduk b4dd6b5997 Low order IDP for MCL works fine. Advection with non-const velocity doesn't work for now. Reorganized fe_evol stuff. New scripts. WIP 2020-04-03 19:02:47 +02:00
HennesHajduk a739bcd2f8 . 2020-04-02 18:29:30 +02:00
HennesHajduk ffbe7710b9 style. 2020-04-02 18:27:32 +02:00
HennesHajduk 22a9fa0892 wip on MCL low order works mostly. change in valuerange for glvis. 2020-04-02 18:26:27 +02:00
HennesHajduk 45ca17f09c Changes in advection 2020-04-02 17:47:35 +02:00
HennesHajduk 3014937c08 NodalQuadRule option for face integrals 2020-04-02 17:42:41 +02:00
HennesHajduk c29c742a29 restructured and extendede (p)dofs by subcellcross 2020-04-02 17:41:34 +02:00
HennesHajduk 2e0d1d89bc wip, low order method works fine in some cases, still some problems. removed some meshes, new config scripts started. 2020-04-02 17:19:38 +02:00
Hennes Hajduk c35ad8f18a wip 2020-03-30 11:35:59 +02:00
HennesHajduk 2d289a1a0a wip, switch for nodal eval in advection. todo: fix rusanov dij for general problems. 2020-03-29 15:04:23 +02:00
HennesHajduk 53634954d6 sub2ind for triangles. other than that MCL-LO method seems to work for simple 1d advection. WIP 2020-03-27 18:11:30 +01:00
HennesHajduk b880107442 wip on mcl 2020-03-25 19:09:55 +01:00
HennesHajduk a25fbaa530 Renaming, restructuring, MCL introduced. 2020-03-25 13:34:29 +01:00
HennesHajduk 17c653afae fe_evol directory 2020-03-25 11:43:20 +01:00
HennesHajduk 3567ff10af removed variable scheme 2020-03-25 10:49:34 +01:00
HennesHajduk f2b1c68fd3 header 2020-03-24 18:42:55 +01:00
HennesHajduk a996e039a0 mass error check for sum of variables. 2020-03-24 18:13:11 +01:00
HennesHajduk c2f300a13b fixed lumpedMassMatrix 2020-03-24 18:09:07 +01:00
HennesHajduk a63c2adeac Mostly done with reorganizing. Some TODOs left and moving stuff to derived classes from FE_Evolution constructor is necessary 2020-03-24 17:56:10 +01:00
HennesHajduk c14d057b33 Initial working restructured code 2020-03-24 16:52:48 +01:00
HennesHajduk 268dabe5b3 reorganized evolution schemes to separete classes. 2020-03-24 15:43:30 +01:00
HennesHajduk 89adf31fa9 Only restructuring. 2020-03-23 10:17:18 +01:00
Hennes Hajduk 1818159e6e WIP 2020-03-20 18:21:35 +01:00
Hennes Hajduk 39c4b2c6eb minor, wip, 2 new euler test cases (don't work yet) 2020-03-20 18:02:42 +01:00
Hennes Hajduk 26f28aef94 Reorganized directory 2020-03-20 16:59:34 +01:00
Hennes Hajduk 14ff697e6d Restructured some hyperbolic systems, new grids, minor, wip. 2020-03-20 16:48:23 +01:00
Hennes Hajduk 0024f03b36 bdr ids as supposed to again 2020-03-20 10:37:32 +01:00
Hennes Hajduk 7cf631f2ae minor fix plus stlye. 2020-03-20 10:35:37 +01:00
Hennes Hajduk bf6660abe7 valuerange 2020-03-20 10:27:55 +01:00
HennesHajduk aab7cbd0ba preliminary [0,1] scaling. TODO custom for example and problem. 2020-03-19 20:15:18 +01:00
Hennes Hajduk 5e04c9e759 wip 2020-03-19 17:55:12 +01:00
Hennes Hajduk 78deadd0b2 using bdr attributes in DofInfo class 2020-03-19 09:50:04 +01:00
Hennes Hajduk c88d99a660 wip, b.c 2020-03-18 19:33:10 +01:00
Hennes Hajduk ef5e133a06 Merge branch 'master' into hypsys-dev 2020-03-18 17:07:42 +01:00
Hennes Hajduk 97e2e04ef1 b.c. wip 2020-03-18 16:59:16 +01:00
Hennes Hajduk 1c87d3aec9 WIP on bdr cond. new Euler test case 2020-03-17 18:43:14 +01:00
Hennes Hajduk 5944bc1e98 Reorganized (P)FE_Evolution, plus minor. 2020-03-16 11:48:02 +01:00
Hennes Hajduk a92b985363 some todos, minor, apps 2020-03-12 18:36:47 +01:00
Hennes Hajduk a71d9f4aa2 removed inflow from HyperbolicSystem class. 2020-03-12 18:06:55 +01:00
Hennes Hajduk 438f5cd0f1 WIP, glvis window title, minor output bug fixed 2020-03-12 17:34:59 +01:00
Hennes Hajduk ccee9d48ba Bugfix 2020-03-12 12:07:40 +01:00
Hennes Hajduk f7d2aad9dd makefile 2020-03-11 16:15:09 +01:00
Hennes Hajduk 420af6dacd valgrind mpi 2020-03-06 18:13:06 +01:00
Hennes Hajduk 0e05495bad wip /parallel valgrind debug 2020-03-06 10:58:17 +01:00
Hennes Hajduk d70fe9904e Correct bdr cond projections. 2020-03-06 10:21:14 +01:00
Hennes Hajduk 2591ff1f61 valgrind check mpi - wip 2020-03-05 11:26:16 +01:00
Hennes Hajduk 77407972c0 minor 2020-03-05 10:47:24 +01:00
Hennes Hajduk 8e3dda40eb fixed destructor in HyperbolicSystem. 2020-03-04 11:05:43 +01:00
Hennes Hajduk fcc0bcbd4c euler 2020-03-03 11:45:11 +01:00
HennesHajduk c60edb07f4 wip 2020-02-21 15:20:26 +01:00
HennesHajduk 6cdf650c44 renamed hypsys.hpp + minor 2020-02-21 12:35:22 +01:00
HennesHajduk 85651b4737 bugfix 2020-02-21 12:05:29 +01:00
HennesHajduk 055d9f1052 minor 2020-02-21 11:54:02 +01:00
HennesHajduk 7b92942e96 minor, bugfix 2020-02-19 18:06:44 +01:00
HennesHajduk 679739b1e3 KPP problem 2020-02-18 18:00:06 +01:00
HennesHajduk 7abd2a94e8 WriteErrors as Function of hypsys, minor, changes required in template 2020-02-18 17:30:49 +01:00
HennesHajduk 1f119edde1 style 2020-02-18 17:11:55 +01:00
HennesHajduk ad86262437 Burgers implemneted and working. Inflow is now time-dependent and member of fe_evol, rather than of hyp. Some TODOs remain at this stage. 2020-02-18 17:09:19 +01:00
Hennes Hajduk 07497cae07 minor, valgrind issue in advection. 2020-02-14 17:25:14 +01:00
Hennes Hajduk c617b3afd4 advection and swe work in serial and parallel. 2020-02-14 14:46:11 +01:00
Hennes Hajduk bc27a94112 WIP: only problem is now solving systems in parallel due to wrong NbrDof indexing. 2020-02-12 17:51:37 +01:00
Hennes Hajduk 460492012e WIP, changed swe test case 2020-02-12 17:46:48 +01:00
Hennes Hajduk bc258e1d94 Manuel's H1 codes for monolithic convex limiting 2020-02-12 17:42:34 +01:00
Hennes Hajduk e0b0472bf2 style 2020-02-10 18:11:05 +01:00
Hennes Hajduk 71855ba554 SWE and advection now using same fe_evolution. 2020-02-10 18:10:17 +01:00
Hennes Hajduk 229fe92b41 wip 2020-02-10 16:03:24 +01:00
Hennes Hajduk 2b8cf51e09 wip, swe and advection work (in serial) 2020-02-10 15:20:18 +01:00
Hennes Hajduk bfffa918f7 wip, parallel works again for advection. 2020-02-07 16:20:37 +01:00
Hennes Hajduk a40bd2f790 Merge branch 'master' into hypsys-dev
updating my branch.
2020-02-07 15:49:25 +01:00
Hennes Hajduk b2382119da merging systems with advection - wip 2020-02-06 17:35:41 +01:00
Hennes Hajduk 7995844fb9 wip 2020-02-06 17:21:00 +01:00
Hennes Hajduk 5225e2dea3 wip 2020-02-06 09:07:51 +01:00
Hennes Hajduk 0ba34e52e3 wip 2020-02-05 09:18:12 +01:00
Hennes Hajduk baaddcf782 tic, toc, astyle 2020-02-04 09:18:21 +01:00
HennesHajduk ab2251197f minor 2020-01-31 17:05:17 +01:00
HennesHajduk 3f9dc61322 make style 2020-01-31 16:36:32 +01:00
Hennes Hajduk 421f43f6ec makefile, Lax-Friedrichs-type flux plus minor. 2020-01-31 16:20:14 +01:00
Hennes Hajduk 101cb10b18 advection in serial and parallel. 2020-01-30 19:51:49 +01:00
Hennes Hajduk 265aa483ca new serial/parallel structure 2020-01-28 18:01:21 +01:00
Hennes Hajduk d685787c50 Merge branch 'master' into hypsys-dev
occasional merge.
2020-01-27 17:16:53 +01:00
Hennes Hajduk 0741c02a38 minor 2020-01-27 17:10:03 +01:00
Hennes Hajduk eb3c078e94 wip merge with parallel 2020-01-21 14:36:09 +01:00
Hennes Hajduk 50c56fdc83 merge with parallel 2020-01-21 13:50:02 +01:00
Hennes Hajduk 35331da9c9 Minor. 2020-01-16 18:48:56 +01:00
Hennes Hajduk 3fc885ed73 wip, minor fixes in serial. 2020-01-16 15:04:57 +01:00
Hennes Hajduk d09d927b86 started work on parallel. 2020-01-14 18:00:32 +01:00
Hennes Hajduk 32b67c327c serial code works for advection. 2020-01-14 17:02:56 +01:00
Hennes Hajduk 39827e802f infrastructure for grid convergence studies. 2020-01-14 10:41:42 +01:00
Hennes Hajduk cac649144f WIP 2020-01-13 17:00:02 +01:00
Hennes Hajduk 7903b11b9b Memory issues 2020-01-13 11:54:42 +01:00
Hennes Hajduk 8d17794793 WIP, advection equation is working. 2020-01-10 17:48:07 +01:00
Hennes Hajduk 175c47d8ff wip 2020-01-07 19:00:22 +01:00
Hennes Hajduk f3bdd37ecf wip 2020-01-06 17:37:53 +01:00
Hennes Hajduk 5e943f7998 added mesh. 2019-12-17 15:44:32 +01:00
HennesHajduk 43555a801d Initial commit for hypsys miniapp. 2019-12-01 20:38:01 +01:00
296 changed files with 34438 additions and 20371 deletions
+8 -10
View File
@@ -15,10 +15,8 @@ install:
- msmpisdk.msi /passive
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
# Install METIS, use a mirror because the original source server is not always
# up. Original url:
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
# Install METIS
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
- cd metis-5.1.0
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
@@ -28,17 +26,17 @@ install:
- cd ..
# Install hypre
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/v2.19.0.tar.gz'
- 7z x v2.19.0.tar.gz -so | 7z x -si -ttar > nul
- cd hypre-2.19.0/src
- cmake -H. -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/V2-10-0b.tar.gz'
- 7z x V2-10-0b.tar.gz -so | 7z x -si -ttar > nul
- cd hypre-2-10-0b
- cmake -H. -Bbuild -DHYPRE_USING_FEI=OFF -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
- cmake --build build
- cmake --build build --target install
- cd ../..
- cd ..
# MFEM
before_build:
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_DIR=%cd%\hypre-2.19.0\src\hypre -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2-10-0b\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2-10-0b\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE
build_script:
+14 -12
View File
@@ -163,6 +163,20 @@ miniapps/electromagnetics/Tesla-AMR*
miniapps/electromagnetics/Maxwell-Parallel*
miniapps/electromagnetics/Joule_*
miniapps/hypsys/build
miniapps/hypsys/errors.txt
miniapps/hypsys/grid*
miniapps/hypsys/hypsys
miniapps/hypsys/initial*
miniapps/hypsys/output
miniapps/hypsys/phypsys
miniapps/hypsys/pressure*
miniapps/hypsys/results
miniapps/hypsys/scripts/gridfunc-scatter
miniapps/hypsys/ultimate*
miniapps/hypsys/velocity*
miniapps/hypsys/various
miniapps/meshing/mobius-strip
miniapps/meshing/klein-bottle
miniapps/meshing/toroid
@@ -244,24 +258,12 @@ miniapps/navier/navier_3dfoc
miniapps/navier/tgv_out*.txt
miniapps/navier/*_output
miniapps/adjoint/cvsRoberts_ASAi_dns
miniapps/adjoint/adjoint_advection_diffusion
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
tests/unit/punit_tests
tests/unit/sedov_tests_*
tests/unit/psedov_tests_*
tests/unit/tmop_tests_*
tests/unit/ptmop_tests_*
tests/unit/cube.mesh
tests/unit/star.mesh
tests/unit/blade.mesh
tests/unit/square01.mesh
tests/unit/toroid-hex.mesh
tests/unit/beam-hex-nurbs.mesh
tests/unit/square-disc-nurbs.mesh
# Test script output
tests/scripts/*.err
+33 -98
View File
@@ -11,20 +11,11 @@
language: cpp
os: linux
dist: bionic
stages:
- checks
- tests
- optional
env:
global:
- HYPRE_ARCHIVE=v2.19.0.tar.gz
HYPRE_URL=https://github.com/hypre-space/hypre/archive/$HYPRE_ARCHIVE
HYPRE_TOP_DIR=hypre-2.19.0
jobs:
include:
@@ -37,7 +28,6 @@ jobs:
- stage: checks
os: linux
dist: xenial
name: "code-style"
addons:
apt:
@@ -56,6 +46,9 @@ jobs:
packages:
- doxygen
- graphviz
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
@@ -70,24 +63,13 @@ jobs:
- mpich
- libmpich-dev
env: MPI=YES
before_script:
script:
- cd ${TRAVIS_BUILD_DIR}
- mpicxx -v
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
- make all -j3
- make test-noclean
script:
- cd tests/scripts
- ./runtest gitignore
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a Lib ..; rm -rf * ; mv ../libmetis.a ../Lib .;
rm -f Lib/*.{c,o}
# ========================
# Optional Checks/Tests
@@ -96,7 +78,6 @@ jobs:
- stage: optional
name: "branch-history"
if: branch != next
# need full git history for the binary/big files check
git:
depth: false
@@ -125,8 +106,6 @@ jobs:
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
cache:
ccache: true
- os: linux
compiler: gcc
@@ -135,8 +114,6 @@ jobs:
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
cache:
ccache: true
- os: linux
compiler: gcc
@@ -160,9 +137,9 @@ jobs:
MFEM_TEST_TARGET=check
NPROCS=2
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -191,9 +168,9 @@ jobs:
MFEM_TEST_TARGET=test
NPROCS=2
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -216,16 +193,16 @@ jobs:
- cd ${TRAVIS_BUILD_DIR}/build
- cmake ..
-DMFEM_USE_MPI=ON
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../$HYPRE_TOP_DIR/src/hypre
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../hypre-2.10.0b/src/hypre
-DMFEM_MPI_NP=$NPROCS
- make -j3 mfem examples
- cd ${TRAVIS_BUILD_DIR}/build/tests/unit
- make -j3
- ctest --output-on-failure
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
@@ -241,43 +218,27 @@ jobs:
# - parallel
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial + Debug"
addons:
homebrew:
packages:
- ccache
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
cache:
ccache: true
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial"
addons:
homebrew:
packages:
- ccache
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
cache:
ccache: true
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel + Debug"
addons:
homebrew:
packages:
- ccache
env: DEBUG=YES
MPI=YES
CODECOV=NO
@@ -285,9 +246,9 @@ jobs:
NPROCS=4
TMPDIR=/tmp
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
@@ -296,13 +257,9 @@ jobs:
rm -f Lib/*.{c,o}
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel"
addons:
homebrew:
packages:
- ccache
env: DEBUG=NO
MPI=YES
CODECOV=YES
@@ -310,9 +267,9 @@ jobs:
NPROCS=4
TMPDIR=/tmp
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
- $TRAVIS_BUILD_DIR/../metis-4.0
- $HOME/local-cached
before_cache:
@@ -326,19 +283,14 @@ before_install:
# brew install open-mpi;
# fi
# Disable ccache while building dependencies that are cached:
- echo "before \$PATH = $PATH";
export PATH=${PATH//\/usr\/lib\/ccache:/};
echo "after \$PATH = $PATH"
# On Mac OS X, build and cache OpenMPI 2.1.6:
# On Mac OS X, build and cache OpenMPI 2.1.1:
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
mkdir -p $HOME/builds && cd $HOME/builds &&
wget https://download.open-mpi.org/release/open-mpi/v2.1/openmpi-2.1.6.tar.bz2 &&
tar jxf openmpi-2.1.6.tar.bz2 &&
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
tar jxf openmpi-2.1.1.tar.bz2 &&
mkdir openmpi-build && cd openmpi-build &&
../openmpi-2.1.6/configure --prefix=$HOME/local-cached &&
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
make -j3 all && make install;
fi;
PATH=$HOME/local-cached/bin:$PATH;
@@ -383,28 +335,26 @@ install:
# hypre
- if [ $MPI == "YES" ]; then
if [ ! -e $HYPRE_TOP_DIR/src/hypre/lib/libHYPRE.a ]; then
wget $HYPRE_URL;
rm -rf $HYPRE_TOP_DIR;
tar xvzf $HYPRE_ARCHIVE;
cd $HYPRE_TOP_DIR/src;
./configure --disable-fortran CC=mpicc CXX=mpic++;
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
rm -rf hypre-2.10.0b;
tar xvzf hypre-2.10.0b.tar.gz;
cd hypre-2.10.0b/src;
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
make -j3;
cd ../..;
else
echo "Reusing cached $HYPRE_TOP_DIR/";
echo "Reusing cached hypre-2.10.0b/";
fi;
ln -s $HYPRE_TOP_DIR hypre;
ln -s hypre-2.10.0b hypre;
else
echo "Serial build, not using hypre";
fi
# METIS, use a mirror because the original source server is not always up.
# Original url:
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz
# METIS
- if [ $MPI == "YES" ]; then
if [ ! -e metis-4.0/libmetis.a ]; then
wget https://mfem.github.io/tpls/metis-4.0.3.tar.gz;
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
tar xvzf metis-4.0.3.tar.gz;
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
rm -rf metis-4.0;
@@ -414,18 +364,6 @@ install:
fi;
fi
# Re-enable ccache on linux; enable ccache on mac os:
- if [ $TRAVIS_OS_NAME == "linux" ]; then
export PATH="/usr/lib/ccache:$PATH";
else
if [ $TRAVIS_OS_NAME == "osx" ]; then
export PATH="/usr/local/opt/ccache/libexec:$PATH";
fi;
fi
- printf "which \$CC = "; which $CC;
printf "which \$CXX = "; which $CXX
script:
# Compiler
- if [ $MPI == "YES" ]; then
@@ -446,9 +384,6 @@ script:
if [ "$CODECOV" == "YES" ]; then
CPPFLAGS="--coverage -g";
fi;
if [ "$TRAVIS_OS_NAME" != "linux" ] || [ "$DEBUG" == "YES" ]; then
CPPFLAGS+=" -pedantic -Wall -Werror";
fi
# Configure the library
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG $MAKE_CXX_FLAG
+5 -38
View File
@@ -33,9 +33,6 @@ Meshing improvements
for example the periodic-annulus-sector and periodic-torus-sector files in
the data directory.
- Added complete action of the TMOP Integrator to account for the spatial
derivatives of discrete and analytic targets.
Performance improvements
------------------------
- Added support for explicit vectorization in the high-performance templated
@@ -44,26 +41,12 @@ Performance improvements
- x86 (SSE/AVX/AVX2/AVX512),
- Power8 & Power9 (VSX),
- BG/Q (QPX).
These are disabled by default, and can be enabled with MFEM_USE_SIMD=YES.
These are now enabled by default, and can be disabled with MFEM_USE_SIMD=NO.
See the new file linalg/simd.hpp and the new directory linalg/simd.
Improved GPU capabilities
-------------------------
- Added support for Chebyshev accelerated polynomial smoother on GPU.
- The TMOP mesh optimization algorithms were extended to GPU:
- QualityMetric #1, #2 and #7 are available in 2D, #302, #303 and #321 in 3D
- Both AnalyticAdaptTC and DiscreteAdaptTC TargetConstructor are available
- Kernels for normalization and limiting have been added
- The AdvectorCG now also support AssemblyLevel::PARTIAL
- Optimized AMD/HIP kernel support.
- Added a Full Assembly mode compatible with Device kernel execution. This
assembly level builds on top of the current Element Assembly kernels to
compute a global sparse matrix. All integrators supported by element assembly
are also supported by full assembly. See the '-fa' option in Example 9.
- Added support for BlockOperator on GPU. See the updated Example 5.
Discretization improvements
---------------------------
@@ -88,7 +71,7 @@ Discretization improvements
and, in the continuous field case, arbitrary mesh edges and faces.
- Added new coefficient and vector coefficient classes for QuadratureFunctions.
Additionally, new LinearForm integrators were also added which make use of
Additionaly, new LinearForm integrators were also added which make use of
these new QuadratureFunction coefficient classes.
- Added support face integrals on the boundaries of NURBS meshes.
@@ -105,10 +88,6 @@ Linear and nonlinear solvers
and solution during the solving process of an IterativeSolver after every
iteration.
- Added support for the CVODES package in SUNDIALS which provides ODE
solvers with sensitivity analysis capabilities. See the CVODESSolver
class and the new adjoint miniapps below.
- Block arrays of parallel matrices can now be merged into a single parallel
matrix with the function HypreParMatrixFromBlocks. This could be useful for
solving block systems with parallel direct solvers such as STRUMPACK.
@@ -136,14 +115,6 @@ New and updated examples and miniapps
equations of incompressible fluid dynamics. See the miniapps/navier directory
for more details.
- Added a new miniapps/adjoint directory with two miniapps demonstrating how to
solve adjoint problems in MFEM using the CVODES package in SUNDIALS. Both of
these miniapps require the MFEM_USE_SUNDIALS configuration option.
* The cvsRoberts_ASAi_dns miniapp solves a backward adjoint problem for a
system of ODEs, evaluating both forward and adjoint quadratures in serial.
* The adjoint_advection_diffusion miniapp solves a backward adjoint problem
for an advection diffusion PDE, evaluating adjoint quadratures in parallel.
- Ported Example 11p to SLEPc, to demonstrate solving the Laplace eigenvalue
equation with the shift-and-invert spectral transformation method.
@@ -154,13 +125,11 @@ New and updated examples and miniapps
- Added a new meshing miniapp, Minimal Surface, which solves Plateau's problem:
the Dirichlet problem for the minimal surface equation.
- Added partial assembly support to Example 4/4p and Example 5/5p, with diagonal
- Added partial assembly support to examples 4/4p and 5/5p, with diagonal
preconditioning.
- Added full assembly support in Example 9/9p.
- Added a new test problem in Example 24/24p, demonstrating a mixed bilinear
form for H1, H(curl), H(div) and L_2, with partial assembly support.
- Added a new test problem in example 24/24p, demonstrating a mixed bilinear
form for H(div) and L_2, with partial assembly support.
- Added weak Dirichlet boundary conditions (Nitsche) to the NURBS miniapp.
@@ -168,8 +137,6 @@ New and updated examples and miniapps
mesh based on element attributes. Any newly exposed boundary elements are
assigned attribute numbers related to the trimmed element attributes.
- Added device support in Example 5/5p.
Improved testing
----------------
- Added a GitLab pipeline that automates PR testing on supercomputing systems
+2 -2
View File
@@ -211,10 +211,10 @@ endif()
# SUNDIALS
if (MFEM_USE_SUNDIALS)
if (NOT MFEM_USE_MPI)
find_package(SUNDIALS REQUIRED NVector_Serial CVODES ARKODE KINSOL)
find_package(SUNDIALS REQUIRED NVector_Serial CVODE ARKODE KINSOL)
else()
find_package(SUNDIALS REQUIRED
NVector_Serial NVector_Parallel NVector_ParHyp CVODES ARKODE KINSOL)
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
endif()
endif()
-1
View File
@@ -109,7 +109,6 @@ The MFEM source code has the following structure:
├── linalg
├── mesh
├── miniapps
│ ├── adjoint
│ ├── common
│ ├── electromagnetics
│ ├── gslib
-1
View File
@@ -25,6 +25,5 @@ mfem_find_package(SUNDIALS SUNDIALS SUNDIALS_DIR
ADD_COMPONENT NVector_ParHyp
"include" nvector/nvector_parhyp.h "lib" sundials_nvecparhyp
ADD_COMPONENT CVODE "include" cvode/cvode.h "lib" sundials_cvode
ADD_COMPONENT CVODES "include" cvodes/cvodes.h "lib" sundials_cvodes
ADD_COMPONENT ARKODE "include" arkode/arkode.h "lib" sundials_arkode
ADD_COMPONENT KINSOL "include" kinsol/kinsol.h "lib" sundials_kinsol)
+1 -3
View File
@@ -50,7 +50,7 @@ option(MFEM_USE_OCCA "Enable OCCA" OFF)
option(MFEM_USE_RAJA "Enable RAJA" OFF)
option(MFEM_USE_CEED "Enable CEED" OFF)
option(MFEM_USE_UMPIRE "Enable Umpire" OFF)
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" OFF)
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" ON)
option(MFEM_USE_ADIOS2 "Enable ADIOS2" OFF)
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
@@ -88,8 +88,6 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
# and modify cmake variables for hypre for sundials
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-5.0.0/instdir" CACHE PATH
"Path to the SUNDIALS library.")
# The following may be necessary, if SUNDIALS was built with KLU:
+4 -12
View File
@@ -138,8 +138,7 @@ MFEM_USE_RAJA = NO
MFEM_USE_OCCA = NO
MFEM_USE_CEED = NO
MFEM_USE_UMPIRE = NO
MFEM_USE_CAMP = NO
MFEM_USE_SIMD = NO
MFEM_USE_SIMD = YES
MFEM_USE_ADIOS2 = NO
# Compile and link options for zlib.
@@ -190,12 +189,10 @@ OPENMP_LIB =
POSIX_CLOCKS_LIB = -lrt
# SUNDIALS library configuration
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
# and modify cmake variables for hypre for sundials
SUNDIALS_DIR = @MFEM_DIR@/../sundials-5.0.0/instdir
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib64 -L$(SUNDIALS_DIR)/lib64\
-lsundials_arkode -lsundials_cvodes -lsundials_nvecserial -lsundials_kinsol
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
ifeq ($(MFEM_USE_MPI),YES)
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
@@ -342,9 +339,9 @@ GSLIB_DIR = @MFEM_DIR@/../gslib/build
GSLIB_OPT = -I$(GSLIB_DIR)/include
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
# CUDA library configuration
# CUDA library configuration (currently not needed)
CUDA_OPT =
CUDA_LIB = -lcusparse
CUDA_LIB =
# HIP library configuration (currently not needed)
HIP_OPT =
@@ -373,11 +370,6 @@ UMPIRE_DIR = @MFEM_DIR@/../umpire
UMPIRE_OPT = -I$(UMPIRE_DIR)/include
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -lumpire
# CAMP library configuration
CAMP_DIR = @MFEM_DIR@/../camp
CAMP_OPT = -I$(CAMP_DIR)/include
CAMP_LIB = -L$(CAMP_DIR)/lib
# If YES, enable some informational messages
VERBOSE = NO
-1
View File
@@ -770,7 +770,6 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/examples/pumi \
@MFEM_SOURCE_DIR@/examples/hiop \
@MFEM_SOURCE_DIR@/examples/sundials \
@MFEM_SOURCE_DIR@/miniapps/adjoint \
@MFEM_SOURCE_DIR@/miniapps/common \
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
@MFEM_SOURCE_DIR@/miniapps/gslib \
+4 -8
View File
@@ -88,8 +88,8 @@ namespace mfem {
* - <a class="el" href="ex24p_8cpp_source.html">Example 24p</a>: parallel mixed finite element spaces and interpolators
* - <a class="el" href="ex25_8cpp_source.html">Example 25</a>: simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
* - <a class="el" href="ex25p_8cpp_source.html">Example 25p</a>: parallel simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
*
* <H4>SUNDIALS Examples</H4>
* - Variants of Examples
@@ -101,9 +101,6 @@ namespace mfem {
* and
* <a class="el" href="sundials_2ex16p_8cpp_source.html">16p</a>
* demonstrating the use of MFEM's \link sundials.hpp SUNDIALS classes\endlink
* - CVODES adjoint examples:
* <a class="el" href="cvsRoberts__ASAi__dns_8cpp_source.html">serial ODE system</a>,
* <a class="el" href="adjoint__advection__diffusion_8cpp_source.html">parallel advection-diffusion</a>
*
* <H4>PETSc Examples</H4>
* - Variants of Examples
@@ -143,9 +140,7 @@ namespace mfem {
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
* - <a class="el" href="toroid_8cpp_source.html">Toroid</a>: generate simple toroidal meshes
* - <a class="el" href="twist_8cpp_source.html">Twist</a>: generate simple periodic meshes
@@ -162,6 +157,7 @@ namespace mfem {
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
*
+31 -34
View File
@@ -103,8 +103,8 @@ int main(int argc, char *argv[])
// 3. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh mesh(mesh_file, 1, 1);
int dim = mesh.Dimension();
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
@@ -112,10 +112,10 @@ int main(int argc, char *argv[])
// elements.
{
int ref_levels =
(int)floor(log(50000./mesh.GetNE())/log(2.)/dim);
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh.UniformRefinement();
mesh->UniformRefinement();
}
}
@@ -123,70 +123,66 @@ int main(int argc, char *argv[])
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (mesh.GetNodes())
else if (mesh->GetNodes())
{
fec = mesh.GetNodes()->OwnFEC();
delete_fec = false;
fec = mesh->GetNodes()->OwnFEC();
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
FiniteElementSpace fespace(&mesh, fec);
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
cout << "Number of finite element unknowns: "
<< fespace.GetTrueVSize() << endl;
<< fespace->GetTrueVSize() << endl;
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
// In this example, the boundary conditions are defined by marking all
// the boundary attributes from the mesh as essential (Dirichlet) and
// converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (mesh.bdr_attributes.Size())
if (mesh->bdr_attributes.Size())
{
Array<int> ess_bdr(mesh.bdr_attributes.Max());
Array<int> ess_bdr(mesh->bdr_attributes.Max());
ess_bdr = 1;
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
// the basis functions in the finite element fespace.
LinearForm b(&fespace);
LinearForm *b = new LinearForm(fespace);
ConstantCoefficient one(1.0);
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
// 8. Define the solution vector x as a finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
GridFunction x(&fespace);
GridFunction x(fespace);
x = 0.0;
// 9. Set up the bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
BilinearForm a(&fespace);
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a.AddDomainIntegrator(new DiffusionIntegrator(one));
BilinearForm *a = new BilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
// 10. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: eliminating boundary
// conditions, applying conforming constraints for non-conforming AMR,
// static condensation, etc.
if (static_cond) { a.EnableStaticCondensation(); }
a.Assemble();
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
OperatorPtr A;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
cout << "Size of linear system: " << A->Height() << endl;
@@ -207,9 +203,9 @@ int main(int argc, char *argv[])
}
else // Jacobi preconditioning in partial assembly mode
{
if (UsesTensorBasis(fespace))
if (UsesTensorBasis(*fespace))
{
OperatorJacobiSmoother M(a, ess_tdof_list);
OperatorJacobiSmoother M(*a, ess_tdof_list);
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
}
else
@@ -219,13 +215,13 @@ int main(int argc, char *argv[])
}
// 12. Recover the solution as a finite element grid function.
a.RecoverFEMSolution(X, b, x);
a->RecoverFEMSolution(X, *b, x);
// 13. Save the refined mesh and the solution. This output can be viewed later
// using GLVis: "glvis -m refined.mesh -g sol.gf".
ofstream mesh_ofs("refined.mesh");
mesh_ofs.precision(8);
mesh.Print(mesh_ofs);
mesh->Print(mesh_ofs);
ofstream sol_ofs("sol.gf");
sol_ofs.precision(8);
x.Save(sol_ofs);
@@ -237,14 +233,15 @@ int main(int argc, char *argv[])
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << x << flush;
sol_sock << "solution\n" << *mesh << x << flush;
}
// 15. Free the used memory.
if (delete_fec)
{
delete fec;
}
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete mesh;
return 0;
}
+35 -37
View File
@@ -112,8 +112,8 @@ int main(int argc, char *argv[])
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh mesh(mesh_file, 1, 1);
int dim = mesh.Dimension();
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 5. Refine the serial mesh on all processors to increase the resolution. In
// this example we do 'ref_levels' of uniform refinement. We choose
@@ -121,23 +121,23 @@ int main(int argc, char *argv[])
// more than 10,000 elements.
{
int ref_levels =
(int)floor(log(10000./mesh.GetNE())/log(2.)/dim);
(int)floor(log(10000./mesh->GetNE())/log(2.)/dim);
for (int l = 0; l < ref_levels; l++)
{
mesh.UniformRefinement();
mesh->UniformRefinement();
}
}
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh pmesh(MPI_COMM_WORLD, mesh);
mesh.Clear();
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
{
int par_ref_levels = 2;
for (int l = 0; l < par_ref_levels; l++)
{
pmesh.UniformRefinement();
pmesh->UniformRefinement();
}
}
@@ -145,16 +145,13 @@ int main(int argc, char *argv[])
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (pmesh.GetNodes())
else if (pmesh->GetNodes())
{
fec = pmesh.GetNodes()->OwnFEC();
delete_fec = false;
fec = pmesh->GetNodes()->OwnFEC();
if (myid == 0)
{
cout << "Using isoparametric FEs: " << fec->Name() << endl;
@@ -163,10 +160,9 @@ int main(int argc, char *argv[])
else
{
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_Int size = fespace.GlobalTrueVSize();
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
HYPRE_Int size = fespace->GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl;
@@ -177,44 +173,44 @@ int main(int argc, char *argv[])
// by marking all the boundary attributes from the mesh as essential
// (Dirichlet) and converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (pmesh.bdr_attributes.Size())
if (pmesh->bdr_attributes.Size())
{
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
Array<int> ess_bdr(pmesh->bdr_attributes.Max());
ess_bdr = 1;
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 9. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system, which in this case is
// (1,phi_i) where phi_i are the basis functions in fespace.
ParLinearForm b(&fespace);
ParLinearForm *b = new ParLinearForm(fespace);
ConstantCoefficient one(1.0);
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
b->AddDomainIntegrator(new DomainLFIntegrator(one));
b->Assemble();
// 10. Define the solution vector x as a parallel finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
ParGridFunction x(&fespace);
ParGridFunction x(fespace);
x = 0.0;
// 11. Set up the parallel bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
ParBilinearForm a(&fespace);
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a.AddDomainIntegrator(new DiffusionIntegrator(one));
ParBilinearForm *a = new ParBilinearForm(fespace);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
a->AddDomainIntegrator(new DiffusionIntegrator(one));
// 12. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, static condensation, etc.
if (static_cond) { a.EnableStaticCondensation(); }
a.Assemble();
if (static_cond) { a->EnableStaticCondensation(); }
a->Assemble();
OperatorPtr A;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
@@ -222,9 +218,9 @@ int main(int argc, char *argv[])
Solver *prec = NULL;
if (pa)
{
if (UsesTensorBasis(fespace))
if (UsesTensorBasis(*fespace))
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
}
}
else
@@ -242,7 +238,7 @@ int main(int argc, char *argv[])
// 14. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
a.RecoverFEMSolution(X, b, x);
a->RecoverFEMSolution(X, *b, x);
// 15. Save the refined mesh and the solution in parallel. This output can
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
@@ -253,7 +249,7 @@ int main(int argc, char *argv[])
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
pmesh.Print(mesh_ofs);
pmesh->Print(mesh_ofs);
ofstream sol_ofs(sol_name.str().c_str());
sol_ofs.precision(8);
@@ -268,14 +264,16 @@ int main(int argc, char *argv[])
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << pmesh << x << flush;
sol_sock << "solution\n" << *pmesh << x << flush;
}
// 17. Free the used memory.
if (delete_fec)
{
delete fec;
}
delete a;
delete b;
delete fespace;
if (order > 0) { delete fec; }
delete pmesh;
MPI_Finalize();
return 0;
+28 -37
View File
@@ -13,11 +13,6 @@
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
//
// With partial assembly:
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1 -pa
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2 -pa
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0 -pa
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
@@ -81,7 +76,6 @@ int main(int argc, char *argv[])
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
bool pa = false;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -112,8 +106,6 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.Parse();
if (!args.Good())
{
@@ -290,7 +282,6 @@ int main(int argc, char *argv[])
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -327,8 +318,6 @@ int main(int argc, char *argv[])
// -Grad(a Div) - omega^2 b + omega c
//
BilinearForm *pcOp = new BilinearForm(fespace);
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -359,8 +348,19 @@ int main(int argc, char *argv[])
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
cout << "Size of linear system: " << A->Width() << endl << endl;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
{
ComplexSparseMatrix * Asp =
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * Asp->real().Width() << endl << endl;
}
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the appropriate sparse smoother.
@@ -368,8 +368,8 @@ int main(int argc, char *argv[])
Array<int> blockOffsets;
blockOffsets.SetSize(3);
blockOffsets[0] = 0;
blockOffsets[1] = A->Height() / 2;
blockOffsets[2] = A->Height() / 2;
blockOffsets[1] = PCOp.Ptr()->Height();
blockOffsets[2] = PCOp.Ptr()->Height();
blockOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockOffsets);
@@ -377,31 +377,22 @@ int main(int argc, char *argv[])
Operator * pc_r = NULL;
Operator * pc_i = NULL;
if (pa)
double s = 1.0;
switch (prob)
{
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
case 0:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
s = -1.0;
break;
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
default: break; // This should be unreachable
}
else
{
OperatorHandle PCOp;
pcOp->SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
switch (prob)
{
case 0:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
case 1:
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
break;
case 2:
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
break;
default:
break; // This should be unreachable
}
}
double s = (prob != 1) ? 1.0 : -1.0;
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
s:-s);
+28 -39
View File
@@ -13,11 +13,6 @@
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
//
// With partial assembly:
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 1 -p 1 -pa
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 1 -p 2 -pa
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0 -pa
//
// Description: This example code demonstrates the use of MFEM to define and
// solve simple complex-valued linear systems. It implements three
// variants of a damped harmonic oscillator:
@@ -89,7 +84,6 @@ int main(int argc, char *argv[])
bool visualization = 1;
bool herm_conv = true;
bool exact_sol = true;
bool pa = false;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -122,8 +116,6 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.Parse();
if (!args.Good())
{
@@ -323,7 +315,6 @@ int main(int argc, char *argv[])
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -360,7 +351,6 @@ int main(int argc, char *argv[])
// -Grad(a Div) - omega^2 b + omega c
//
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
@@ -392,11 +382,19 @@ int main(int argc, char *argv[])
Vector B, U;
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
u = 0.0;
U = 0.0;
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
if (myid == 0)
{
ComplexHypreParMatrix * Ahyp =
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
cout << "Size of linear system: "
<< 2 * fespace->GlobalTrueVSize() << endl << endl;
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
}
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
@@ -406,8 +404,8 @@ int main(int argc, char *argv[])
Array<int> blockTrueOffsets;
blockTrueOffsets.SetSize(3);
blockTrueOffsets[0] = 0;
blockTrueOffsets[1] = A->Height() / 2;
blockTrueOffsets[2] = A->Height() / 2;
blockTrueOffsets[1] = PCOp.Ptr()->Height();
blockTrueOffsets[2] = PCOp.Ptr()->Height();
blockTrueOffsets.PartialSum();
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
@@ -415,34 +413,25 @@ int main(int argc, char *argv[])
Operator * pc_r = NULL;
Operator * pc_i = NULL;
if (pa)
switch (prob)
{
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
}
else
{
OperatorHandle PCOp;
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
switch (prob)
{
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
case 0:
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
break;
case 1:
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
break;
case 2:
if (dim == 2 )
{
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
break;
case 2:
if (dim == 2 )
{
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
}
}
else
{
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
}
break;
default: break; // This should be unreachable
}
pc_i = new ScaledOperator(pc_r,
(conv == ComplexOperator::HERMITIAN) ?
+7 -86
View File
@@ -7,7 +7,6 @@
// ex24 -m ../data/beam-tet.mesh
// ex24 -m ../data/beam-hex.mesh -o 2 -pa
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 1
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 2
// ex24 -m ../data/escher.mesh
// ex24 -m ../data/escher.mesh -o 2
// ex24 -m ../data/fichera.mesh
@@ -25,13 +24,12 @@
// ex24 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code illustrates usage of mixed finite element
// spaces, with three variants:
// spaces, with two variants:
//
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
// 3) (div v, q) for v in H(div) tested against q in L_2
// 2) (div v, q) for v in H(div) tested against q in L_2
//
// Using different approaches, we project the gradient, curl, or
// Using different approaches, we project the gradient or
// divergence to the appropriate space.
//
// We recommend viewing examples 1, 3, and 5 before viewing this
@@ -47,11 +45,8 @@ using namespace mfem;
double p_exact(const Vector &x);
void gradp_exact(const Vector &, Vector &);
double div_gradp_exact(const Vector &x);
void v_exact(const Vector &x, Vector &v);
void curlv_exact(const Vector &x, Vector &cv);
int dim;
double freq = 1.0, kappa;
int main(int argc, char *argv[])
{
@@ -88,7 +83,6 @@ int main(int argc, char *argv[])
return 1;
}
args.PrintOptions(cout);
kappa = freq * M_PI;
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
@@ -125,15 +119,10 @@ int main(int argc, char *argv[])
trial_fec = new H1_FECollection(order, dim);
test_fec = new ND_FECollection(order, dim);
}
else if (prob == 1)
{
trial_fec = new ND_FECollection(order, dim);
test_fec = new RT_FECollection(order-1, dim);
}
else
{
trial_fec = new RT_FECollection(order-1, dim);
test_fec = new L2_FECollection(order-1, dim);
trial_fec = new RT_FECollection(order - 1, dim);
test_fec = new L2_FECollection(order - 1, dim);
}
FiniteElementSpace trial_fes(mesh, trial_fec);
@@ -147,12 +136,6 @@ int main(int argc, char *argv[])
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
}
else if (prob == 1)
{
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
endl;
}
else
{
cout << "Number of Raviart-Thomas finite element unknowns: "
@@ -167,18 +150,12 @@ int main(int argc, char *argv[])
GridFunction x(&test_fes);
FunctionCoefficient p_coef(p_exact);
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
VectorFunctionCoefficient v_coef(sdim, v_exact);
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
FunctionCoefficient divgradp_coef(div_gradp_exact);
if (prob == 0)
{
gftrial.ProjectCoefficient(p_coef);
}
else if (prob == 1)
{
gftrial.ProjectCoefficient(v_coef);
}
else
{
gftrial.ProjectCoefficient(gradp_coef);
@@ -202,11 +179,6 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
}
else if (prob == 1)
{
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
}
else
{
a.AddDomainIntegrator(new MassIntegrator(one));
@@ -272,10 +244,6 @@ int main(int argc, char *argv[])
{
dlo.AddDomainInterpolator(new GradientInterpolator());
}
else if (prob == 1)
{
dlo.AddDomainInterpolator(new CurlInterpolator());
}
else
{
dlo.AddDomainInterpolator(new DivergenceInterpolator());
@@ -290,10 +258,6 @@ int main(int argc, char *argv[])
{
exact_proj.ProjectCoefficient(gradp_coef);
}
else if (prob == 1)
{
exact_proj.ProjectCoefficient(curlv_coef);
}
else
{
exact_proj.ProjectCoefficient(divgradp_coef);
@@ -312,21 +276,8 @@ int main(int argc, char *argv[])
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
" ||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
else if (prob == 1)
{
double errSol = x.ComputeL2Error(curlv_coef);
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
double errProj = exact_proj.ComputeL2Error(curlv_coef);
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in H(div): "
"|| E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
else
@@ -344,7 +295,7 @@ int main(int argc, char *argv[])
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v "
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
"||_{L_2} = " << errProj << '\n' << endl;
@@ -420,33 +371,3 @@ double div_gradp_exact(const Vector &x)
return 0.0;
}
void v_exact(const Vector &x, Vector &v)
{
if (dim == 3)
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(2));
v(2) = sin(kappa * x(0));
}
else
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(0));
if (x.Size() == 3) { v(2) = 0.0; }
}
}
void curlv_exact(const Vector &x, Vector &cv)
{
if (dim == 3)
{
cv(0) = -kappa * cos(kappa * x(2));
cv(1) = -kappa * cos(kappa * x(0));
cv(2) = -kappa * cos(kappa * x(1));
}
else
{
cv = 0.0;
}
}
+11 -93
View File
@@ -6,8 +6,7 @@
// mpirun -np 4 ex24p -m ../data/square-disc.mesh -o 2
// mpirun -np 4 ex24p -m ../data/beam-tet.mesh
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 1
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 2
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -p 1 -pa
// mpirun -np 4 ex24p -m ../data/escher.mesh
// mpirun -np 4 ex24p -m ../data/escher.mesh -o 2
// mpirun -np 4 ex24p -m ../data/fichera.mesh
@@ -25,13 +24,12 @@
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code illustrates usage of mixed finite element
// spaces, with three variants:
// spaces, with two variants:
//
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
// 3) (div v, q) for v in H(div) tested against q in L_2
// 2) (div v, q) for v in H(div) tested against q in L_2
//
// Using different approaches, we project the gradient, curl, or
// Using different approaches, we project the gradient or
// divergence to the appropriate space.
//
// We recommend viewing examples 1, 3, and 5 before viewing this
@@ -47,11 +45,8 @@ using namespace mfem;
double p_exact(const Vector &x);
void gradp_exact(const Vector &, Vector &);
double div_gradp_exact(const Vector &x);
void v_exact(const Vector &x, Vector &v);
void curlv_exact(const Vector &x, Vector &cv);
int dim;
double freq = 1.0, kappa;
int main(int argc, char *argv[])
{
@@ -101,7 +96,6 @@ int main(int argc, char *argv[])
{
args.PrintOptions(cout);
}
kappa = freq * M_PI;
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
@@ -153,15 +147,10 @@ int main(int argc, char *argv[])
trial_fec = new H1_FECollection(order, dim);
test_fec = new ND_FECollection(order, dim);
}
else if (prob == 1)
{
trial_fec = new ND_FECollection(order, dim);
test_fec = new RT_FECollection(order-1, dim);
}
else
{
trial_fec = new RT_FECollection(order-1, dim);
test_fec = new L2_FECollection(order-1, dim);
trial_fec = new RT_FECollection(order - 1, dim);
test_fec = new L2_FECollection(order - 1, dim);
}
ParFiniteElementSpace trial_fes(pmesh, trial_fec);
@@ -177,12 +166,6 @@ int main(int argc, char *argv[])
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
}
else if (prob == 1)
{
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
endl;
}
else
{
cout << "Number of Raviart-Thomas finite element unknowns: "
@@ -198,18 +181,12 @@ int main(int argc, char *argv[])
ParGridFunction x(&test_fes);
FunctionCoefficient p_coef(p_exact);
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
VectorFunctionCoefficient v_coef(sdim, v_exact);
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
FunctionCoefficient divgradp_coef(div_gradp_exact);
if (prob == 0)
{
gftrial.ProjectCoefficient(p_coef);
}
else if (prob == 1)
{
gftrial.ProjectCoefficient(v_coef);
}
else
{
gftrial.ProjectCoefficient(gradp_coef);
@@ -233,11 +210,6 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
}
else if (prob == 1)
{
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
}
else
{
a.AddDomainIntegrator(new MassIntegrator(one));
@@ -321,10 +293,6 @@ int main(int argc, char *argv[])
{
dlo.AddDomainInterpolator(new GradientInterpolator());
}
else if (prob == 1)
{
dlo.AddDomainInterpolator(new CurlInterpolator());
}
else
{
dlo.AddDomainInterpolator(new DivergenceInterpolator());
@@ -339,10 +307,6 @@ int main(int argc, char *argv[])
{
exact_proj.ProjectCoefficient(gradp_coef);
}
else if (prob == 1)
{
exact_proj.ProjectCoefficient(curlv_coef);
}
else
{
exact_proj.ProjectCoefficient(divgradp_coef);
@@ -360,27 +324,11 @@ int main(int argc, char *argv[])
if (myid == 0)
{
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl)"
": || E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad"
" p ||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
}
else if (prob == 1)
{
double errSol = x.ComputeL2Error(curlv_coef);
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
double errProj = exact_proj.ComputeL2Error(curlv_coef);
if (myid == 0)
{
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in "
"H(div): || E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
"||_{L_2} = " << errProj << '\n' << endl;
}
}
@@ -402,7 +350,7 @@ int main(int argc, char *argv[])
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
" ||_{L_2} = " << errInterp << '\n' << endl;
"||_{L_2} = " << errInterp << '\n' << endl;
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
"||_{L_2} = " << errProj << '\n' << endl;
}
@@ -488,33 +436,3 @@ double div_gradp_exact(const Vector &x)
return 0.0;
}
void v_exact(const Vector &x, Vector &v)
{
if (dim == 3)
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(2));
v(2) = sin(kappa * x(0));
}
else
{
v(0) = sin(kappa * x(1));
v(1) = sin(kappa * x(0));
if (x.Size() == 3) { v(2) = 0.0; }
}
}
void curlv_exact(const Vector &x, Vector &cv)
{
if (dim == 3)
{
cv(0) = -kappa * cos(kappa * x(2));
cv(1) = -kappa * cos(kappa * x(0));
cv(2) = -kappa * cos(kappa * x(1));
}
else
{
cv = 0.0;
}
}
+23 -15
View File
@@ -389,22 +389,27 @@ int main(int argc, char *argv[])
// applying any necessary transformations such as: assembly, eliminating
// boundary conditions, applying conforming constraints for
// non-conforming AMR, etc.
a.Assemble(0);
a.Assemble();
OperatorPtr A;
OperatorHandle Ah;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 13. Solve using a direct or an iterative solver
// 13. Transform to monolithic SparseMatrix
SparseMatrix *A = Ah.As<ComplexSparseMatrix>()->GetSystemMatrix();
cout << "Size of linear system: " << A->Height() << endl;
// 14. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUITESPARSE
{
ComplexUMFPackSolver csolver(*A.As<ComplexSparseMatrix>());
csolver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
csolver.SetPrintLevel(1);
csolver.Mult(B, X);
UMFPackSolver solver(*A);
solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
solver.Mult(B, X);
}
#else
// 13a. Set up the Bilinear form a(.,.) for the preconditioner
// 14a. Set up the Bilinear form a(.,.) for the preconditioner
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) + omega^2 * epsilon (E,F)
@@ -432,10 +437,10 @@ int main(int argc, char *argv[])
prec.Assemble();
OperatorPtr PCOpAh;
OperatorHandle PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 13b. Define and apply a GMRES solver for AU=B with a block diagonal
// 14b. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the Gauss-Seidel sparse smoother.
Array<int> offsets(3);
offsets[0] = 0;
@@ -462,15 +467,17 @@ int main(int argc, char *argv[])
}
#endif
// 14. Recover the solution as a finite element grid function and compute the
// 15. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a.RecoverFEMSolution(X, b, x);
// If exact is known compute the error
if (exact_known)
{
ComplexGridFunction x_gf(fespace);
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
int order_quad = max(2, 2 * order + 1);
const IntegrationRule *irs[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; ++i)
@@ -499,7 +506,7 @@ int main(int argc, char *argv[])
<< sqrt(L2Error_Re*L2Error_Re + L2Error_Im*L2Error_Im) << "\n\n";
}
// 15. Save the refined mesh and the solution. This output can be viewed
// 16. Save the refined mesh and the solution. This output can be viewed
// later using GLVis: "glvis -m mesh -g sol".
{
ofstream mesh_ofs("ex25.mesh");
@@ -514,7 +521,7 @@ int main(int argc, char *argv[])
x.imag().Save(sol_i_ofs);
}
// 16. Send the solution by socket to a GLVis server.
// 17. Send the solution by socket to a GLVis server.
if (visualization)
{
// Define visualization keys for GLVis (see GLVis documentation)
@@ -565,7 +572,8 @@ int main(int argc, char *argv[])
}
}
// 17. Free the used memory.
// 18. Free the used memory.
delete A;
delete pml;
delete fespace;
delete fec;
+16 -7
View File
@@ -419,15 +419,21 @@ int main(int argc, char *argv[])
// constraints for non-conforming AMR, etc.
a.Assemble();
OperatorPtr Ah;
OperatorHandle Ah;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 15. Solve using a direct or an iterative solver
// 15. Transform to monolithic HypreParMatrix
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
if (myid == 0)
{
cout << "Size of linear system: " << A->GetGlobalNumRows() << endl;
}
// 16. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUPERLU
{
// Transform to monolithic HypreParMatrix
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
SuperLURowLocMatrix SA(*A);
SuperLUSolver superlu(MPI_COMM_WORLD);
superlu.SetPrintStatistics(false);
@@ -435,9 +441,9 @@ int main(int argc, char *argv[])
superlu.SetColumnPermutation(superlu::PARMETIS);
superlu.SetOperator(SA);
superlu.Mult(B, X);
delete A;
}
#else
// 16a. Set up the parallel Bilinear form a(.,.) for the preconditioner
//
// In Comp
@@ -466,7 +472,7 @@ int main(int argc, char *argv[])
prec.Assemble();
OperatorPtr PCOpAh;
OperatorHandle PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 16b. Define and apply a parallel GMRES solver for AU=B with a block
@@ -490,7 +496,7 @@ int main(int argc, char *argv[])
gmres.SetMaxIter(2000);
gmres.SetRelTol(1e-5);
gmres.SetAbsTol(0.0);
gmres.SetOperator(*Ah);
gmres.SetOperator(*A);
gmres.SetPreconditioner(BlockAMS);
gmres.Mult(B, X);
}
@@ -503,8 +509,10 @@ int main(int argc, char *argv[])
// If exact is known compute the error
if (exact_known)
{
ParComplexGridFunction x_gf(fespace);
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
int order_quad = max(2, 2 * order + 1);
const IntegrationRule *irs[Geometry::NumGeom];
for (int i = 0; i < Geometry::NumGeom; ++i)
@@ -621,6 +629,7 @@ int main(int argc, char *argv[])
}
// 20. Free the used memory.
delete A;
delete pml;
delete fespace;
delete fec;
+17 -36
View File
@@ -11,12 +11,6 @@
// ex5 -m ../data/escher.mesh
// ex5 -m ../data/fichera.mesh
//
// Device sample runs:
// ex5 -m ../data/star.mesh -pa -d cuda
// ex5 -m ../data/star.mesh -pa -d raja-cuda
// ex5 -m ../data/star.mesh -pa -d raja-omp
// ex5 -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple 2D/3D mixed Darcy problem
// corresponding to the saddle point system
// k*u + grad p = f
@@ -56,7 +50,6 @@ int main(int argc, char *argv[])
const char *mesh_file = "../data/star.mesh";
int order = 1;
bool pa = false;
const char *device_config = "cpu";
bool visualization = 1;
OptionsParser args(argc, argv);
@@ -66,8 +59,6 @@ int main(int argc, char *argv[])
"Finite element order (polynomial degree).");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -79,18 +70,13 @@ int main(int argc, char *argv[])
}
args.PrintOptions(cout);
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Read the mesh from the given mesh file. We can handle triangular,
// 2. Read the mesh from the given mesh file. We can handle triangular,
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
// the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 4. Refine the mesh to increase the resolution. In this example we do
// 3. Refine the mesh to increase the resolution. In this example we do
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
// largest number that gives a final mesh with no more than 10,000
// elements.
@@ -103,7 +89,7 @@ int main(int argc, char *argv[])
}
}
// 5. Define a finite element space on the mesh. Here we use the
// 4. Define a finite element space on the mesh. Here we use the
// Raviart-Thomas finite elements of the specified order.
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
@@ -111,7 +97,7 @@ int main(int argc, char *argv[])
FiniteElementSpace *R_space = new FiniteElementSpace(mesh, hdiv_coll);
FiniteElementSpace *W_space = new FiniteElementSpace(mesh, l2_coll);
// 6. Define the BlockStructure of the problem, i.e. define the array of
// 5. Define the BlockStructure of the problem, i.e. define the array of
// offsets for each variable. The last component of the Array is the sum
// of the dimensions of each block.
Array<int> block_offsets(3); // number of variables + 1
@@ -126,7 +112,7 @@ int main(int argc, char *argv[])
std::cout << "dim(R+W) = " << block_offsets.Last() << "\n";
std::cout << "***********************************************************\n";
// 7. Define the coefficients, analytical solution, and rhs of the PDE.
// 6. Define the coefficients, analytical solution, and rhs of the PDE.
ConstantCoefficient k(1.0);
VectorFunctionCoefficient fcoeff(dim, fFun);
@@ -136,28 +122,25 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
FunctionCoefficient pcoeff(pFun_ex);
// 8. Allocate memory (x, rhs) for the analytical solution and the right hand
// 7. Allocate memory (x, rhs) for the analytical solution and the right hand
// side. Define the GridFunction u,p for the finite element solution and
// linear forms fform and gform for the right hand side. The data
// allocated by x and rhs are passed as a reference to the grid functions
// (u,p) and the linear forms (fform, gform).
MemoryType mt = device.GetMemoryType();
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
BlockVector x(block_offsets), rhs(block_offsets);
LinearForm *fform(new LinearForm);
fform->Update(R_space, rhs.GetBlock(0), 0);
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
fform->Assemble();
fform->SyncAliasMemory(rhs);
LinearForm *gform(new LinearForm);
gform->Update(W_space, rhs.GetBlock(1), 0);
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
gform->Assemble();
gform->SyncAliasMemory(rhs);
// 9. Assemble the finite element matrices for the Darcy operator
// 8. Assemble the finite element matrices for the Darcy operator
//
// D = [ M B^T ]
// [ B 0 ]
@@ -202,7 +185,7 @@ int main(int argc, char *argv[])
darcyOp.SetBlock(1,0, &B);
}
// 10. Construct the operators for preconditioner
// 9. Construct the operators for preconditioner
//
// P = [ diag(M) 0 ]
// [ 0 B diag(M)^-1 B^T ]
@@ -219,11 +202,10 @@ int main(int argc, char *argv[])
if (pa)
{
mVarf->AssembleDiagonal(Md);
auto Md_host = Md.HostRead();
Vector invMd(mVarf->Height());
for (int i=0; i<mVarf->Height(); ++i)
{
invMd(i) = 1.0 / Md_host[i];
invMd(i) = 1.0 / Md(i);
}
Vector BMBt_diag(bVarf->Height());
@@ -264,7 +246,7 @@ int main(int argc, char *argv[])
darcyPrec.SetDiagonalBlock(0, invM);
darcyPrec.SetDiagonalBlock(1, invS);
// 11. Solve the linear system with MINRES.
// 10. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(1000);
double rtol(1.e-6);
@@ -281,7 +263,6 @@ int main(int argc, char *argv[])
solver.SetPrintLevel(1);
x = 0.0;
solver.Mult(rhs, x);
if (device.IsEnabled()) { x.HostRead(); }
chrono.Stop();
if (solver.GetConverged())
@@ -292,7 +273,7 @@ int main(int argc, char *argv[])
<< " iterations. Residual norm is " << solver.GetFinalNorm() << ".\n";
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
// 12. Create the grid functions u and p. Compute the L2 error norms.
// 11. Create the grid functions u and p. Compute the L2 error norms.
GridFunction u, p;
u.MakeRef(R_space, x.GetBlock(0), 0);
p.MakeRef(W_space, x.GetBlock(1), 0);
@@ -312,7 +293,7 @@ int main(int argc, char *argv[])
std::cout << "|| u_h - u_ex || / || u_ex || = " << err_u / norm_u << "\n";
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
// 13. Save the mesh and the solution. This output can be viewed later using
// 12. Save the mesh and the solution. This output can be viewed later using
// GLVis: "glvis -m ex5.mesh -g sol_u.gf" or "glvis -m ex5.mesh -g
// sol_p.gf".
{
@@ -329,13 +310,13 @@ int main(int argc, char *argv[])
p.Save(p_ofs);
}
// 14. Save data in the VisIt format
// 13. Save data in the VisIt format
VisItDataCollection visit_dc("Example5", mesh);
visit_dc.RegisterField("velocity", &u);
visit_dc.RegisterField("pressure", &p);
visit_dc.Save();
// 15. Save data in the ParaView format
// 14. Save data in the ParaView format
ParaViewDataCollection paraview_dc("Example5", mesh);
paraview_dc.SetPrefixPath("ParaView");
paraview_dc.SetLevelsOfDetail(order);
@@ -347,7 +328,7 @@ int main(int argc, char *argv[])
paraview_dc.RegisterField("pressure",&p);
paraview_dc.Save();
// 16. Send the solution by socket to a GLVis server.
// 15. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -360,7 +341,7 @@ int main(int argc, char *argv[])
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
}
// 17. Free the used memory.
// 16. Free the used memory.
delete fform;
delete gform;
delete invM;
+21 -42
View File
@@ -11,12 +11,6 @@
// mpirun -np 4 ex5p -m ../data/escher.mesh
// mpirun -np 4 ex5p -m ../data/fichera.mesh
//
// Device sample runs:
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d cuda
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-cuda
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-omp
// mpirun -np 4 ex5p -m ../data/beam-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple 2D/3D mixed Darcy problem
// corresponding to the saddle point system
// k*u + grad p = f
@@ -66,7 +60,6 @@ int main(int argc, char *argv[])
int order = 1;
bool par_format = false;
bool pa = false;
const char *device_config = "cpu";
bool visualization = 1;
bool adios2 = false;
@@ -82,8 +75,6 @@ int main(int argc, char *argv[])
"Format to use when saving the results for VisIt.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
@@ -105,18 +96,13 @@ int main(int argc, char *argv[])
args.PrintOptions(cout);
}
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
if (myid == 0) { device.Print(); }
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// 3. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh *mesh = new Mesh(mesh_file, 1, 1);
int dim = mesh->Dimension();
// 5. Refine the serial mesh on all processors to increase the resolution. In
// 4. Refine the serial mesh on all processors to increase the resolution. In
// this example we do 'ref_levels' of uniform refinement. We choose
// 'ref_levels' to be the largest number that gives a final mesh with no
// more than 10,000 elements, unless the user specifies it as input.
@@ -132,7 +118,7 @@ int main(int argc, char *argv[])
}
}
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
@@ -145,7 +131,7 @@ int main(int argc, char *argv[])
}
}
// 7. Define a parallel finite element space on the parallel mesh. Here we
// 6. Define a parallel finite element space on the parallel mesh. Here we
// use the Raviart-Thomas finite elements of the specified order.
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
@@ -165,7 +151,7 @@ int main(int argc, char *argv[])
std::cout << "***********************************************************\n";
}
// 8. Define the two BlockStructure of the problem. block_offsets is used
// 7. Define the two BlockStructure of the problem. block_offsets is used
// for Vector based on dof (like ParGridFunction or ParLinearForm),
// block_trueOffstes is used for Vector based on trueDof (HypreParVector
// for the rhs and solution of the linear system). The offsets computed
@@ -182,7 +168,7 @@ int main(int argc, char *argv[])
block_trueOffsets[2] = W_space->TrueVSize();
block_trueOffsets.PartialSum();
// 9. Define the coefficients, analytical solution, and rhs of the PDE.
// 8. Define the coefficients, analytical solution, and rhs of the PDE.
ConstantCoefficient k(1.0);
VectorFunctionCoefficient fcoeff(dim, fFun);
@@ -192,30 +178,25 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
FunctionCoefficient pcoeff(pFun_ex);
// 10. Define the parallel grid function and parallel linear forms, solution
// vector and rhs.
MemoryType mt = device.GetMemoryType();
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
BlockVector trueX(block_trueOffsets, mt), trueRhs(block_trueOffsets, mt);
// 9. Define the parallel grid function and parallel linear forms, solution
// vector and rhs.
BlockVector x(block_offsets), rhs(block_offsets);
BlockVector trueX(block_trueOffsets), trueRhs(block_trueOffsets);
ParLinearForm *fform(new ParLinearForm);
fform->Update(R_space, rhs.GetBlock(0), 0);
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
fform->Assemble();
fform->SyncAliasMemory(rhs);
fform->ParallelAssemble(trueRhs.GetBlock(0));
trueRhs.GetBlock(0).SyncAliasMemory(trueRhs);
ParLinearForm *gform(new ParLinearForm);
gform->Update(W_space, rhs.GetBlock(1), 0);
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
gform->Assemble();
gform->SyncAliasMemory(rhs);
gform->ParallelAssemble(trueRhs.GetBlock(1));
trueRhs.GetBlock(1).SyncAliasMemory(trueRhs);
// 11. Assemble the finite element matrices for the Darcy operator
// 10. Assemble the finite element matrices for the Darcy operator
//
// D = [ M B^T ]
// [ B 0 ]
@@ -268,7 +249,7 @@ int main(int argc, char *argv[])
darcyOp->SetBlock(1,0, B);
}
// 12. Construct the operators for preconditioner
// 11. Construct the operators for preconditioner
//
// P = [ diag(M) 0 ]
// [ 0 B diag(M)^-1 B^T ]
@@ -285,11 +266,10 @@ int main(int argc, char *argv[])
{
Md_PA.SetSize(R_space->GetTrueVSize());
mVarf->AssembleDiagonal(Md_PA);
auto Md_host = Md_PA.HostRead();
Vector invMd(Md_PA.Size());
for (int i=0; i<Md_PA.Size(); ++i)
{
invMd(i) = 1.0 / Md_host[i];
invMd(i) = 1.0 / Md_PA(i);
}
Vector BMBt_diag(W_space->GetTrueVSize());
@@ -322,7 +302,7 @@ int main(int argc, char *argv[])
darcyPr->SetDiagonalBlock(0, invM);
darcyPr->SetDiagonalBlock(1, invS);
// 13. Solve the linear system with MINRES.
// 12. Solve the linear system with MINRES.
// Check the norm of the unpreconditioned residual.
int maxIter(pa ? 1000 : 500);
double rtol(1.e-6);
@@ -339,7 +319,6 @@ int main(int argc, char *argv[])
solver.SetPrintLevel(verbose);
trueX = 0.0;
solver.Mult(trueRhs, trueX);
if (device.IsEnabled()) { trueX.HostRead(); }
chrono.Stop();
if (verbose)
@@ -353,7 +332,7 @@ int main(int argc, char *argv[])
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
}
// 14. Extract the parallel grid function corresponding to the finite element
// 13. Extract the parallel grid function corresponding to the finite element
// approximation X. This is the local solution on each processor. Compute
// L2 error norms.
ParGridFunction *u(new ParGridFunction);
@@ -381,7 +360,7 @@ int main(int argc, char *argv[])
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
}
// 15. Save the refined mesh and the solution in parallel. This output can be
// 14. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol_*".
{
ostringstream mesh_name, u_name, p_name;
@@ -402,7 +381,7 @@ int main(int argc, char *argv[])
p->Save(p_ofs);
}
// 16. Save data in the VisIt format
// 15. Save data in the VisIt format
VisItDataCollection visit_dc("Example5-Parallel", pmesh);
visit_dc.RegisterField("velocity", u);
visit_dc.RegisterField("pressure", p);
@@ -411,7 +390,7 @@ int main(int argc, char *argv[])
DataCollection::PARALLEL_FORMAT);
visit_dc.Save();
// 17. Save data in the ParaView format
// 16. Save data in the ParaView format
ParaViewDataCollection paraview_dc("Example5P", pmesh);
paraview_dc.SetPrefixPath("ParaView");
paraview_dc.SetLevelsOfDetail(order);
@@ -423,7 +402,7 @@ int main(int argc, char *argv[])
paraview_dc.RegisterField("pressure",p);
paraview_dc.Save();
// 18. Optionally output a BP (binary pack) file using ADIOS2. This can be
// 17. Optionally output a BP (binary pack) file using ADIOS2. This can be
// visualized with the ParaView VTX reader.
#ifdef MFEM_USE_ADIOS2
if (adios2)
@@ -443,7 +422,7 @@ int main(int argc, char *argv[])
}
#endif
// 19. Send the solution by socket to a GLVis server.
// 18. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
@@ -463,7 +442,7 @@ int main(int argc, char *argv[])
<< endl;
}
// 20. Free the used memory.
// 19. Free the used memory.
delete fform;
delete gform;
delete u;
+17 -19
View File
@@ -20,11 +20,8 @@
// Device sample runs:
// ex9 -pa
// ex9 -ea
// ex9 -fa
// ex9 -pa -m ../data/periodic-cube.mesh
// ex9 -pa -m ../data/periodic-cube.mesh -d cuda
// ex9 -ea -m ../data/periodic-cube.mesh -d cuda
// ex9 -fa -m ../data/periodic-cube.mesh -d cuda
//
// Description: This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
@@ -147,7 +144,6 @@ int main(int argc, char *argv[])
int order = 3;
bool pa = false;
bool ea = false;
bool fa = false;
const char *device_config = "cpu";
int ode_solver_type = 4;
double t_final = 10.0;
@@ -174,8 +170,6 @@ int main(int argc, char *argv[])
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
"--no-element-assembly", "Enable Element Assembly.");
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
"--no-full-assembly", "Enable Full Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
@@ -260,7 +254,7 @@ int main(int argc, char *argv[])
// 5. Define the discontinuous DG finite element space of the given
// polynomial order on the refined mesh.
DG_FECollection fec(order, dim, BasisType::GaussLobatto);
DG_FECollection fec(order, dim, BasisType::Positive);
FiniteElementSpace fes(&mesh, &fec);
cout << "Number of unknowns: " << fes.GetVSize() << endl;
@@ -284,11 +278,6 @@ int main(int argc, char *argv[])
m.SetAssemblyLevel(AssemblyLevel::ELEMENT);
k.SetAssemblyLevel(AssemblyLevel::ELEMENT);
}
else if (fa)
{
m.SetAssemblyLevel(AssemblyLevel::FULL);
k.SetAssemblyLevel(AssemblyLevel::FULL);
}
m.AddDomainIntegrator(new MassIntegrator);
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k.AddInteriorFaceIntegrator(
@@ -389,6 +378,10 @@ int main(int argc, char *argv[])
// iterations, ti, with a time-step dt).
FE_Evolution adv(m, k, b);
Vector masses(u.Size());
m.SpMat().Mult(u, masses);
double mass = masses.Sum();
double t = 0.0;
adv.SetTime(t);
ode_solver->Init(adv);
@@ -435,6 +428,9 @@ int main(int argc, char *argv[])
u.Save(osol);
}
m.SpMat().Mult(u, masses);
cout << "Mass difference:" << abs(mass - masses.Sum()) << endl;
// 10. Free the used memory.
delete ode_solver;
delete pd;
@@ -448,19 +444,21 @@ int main(int argc, char *argv[])
FE_Evolution::FE_Evolution(BilinearForm &_M, BilinearForm &_K, const Vector &_b)
: TimeDependentOperator(_M.Height()), M(_M), K(_K), b(_b), z(_M.Height())
{
bool pa = M.GetAssemblyLevel() == AssemblyLevel::PARTIAL;
bool ea = M.GetAssemblyLevel() == AssemblyLevel::ELEMENT;
Array<int> ess_tdof_list;
if (M.GetAssemblyLevel() == AssemblyLevel::LEGACYFULL)
{
M_prec = new DSmoother(M.SpMat());
M_solver.SetOperator(M.SpMat());
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
}
else
if (pa || ea)
{
M_prec = new OperatorJacobiSmoother(M, ess_tdof_list);
M_solver.SetOperator(M);
dg_solver = NULL;
}
else
{
M_prec = new DSmoother(M.SpMat());
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
M_solver.SetOperator(M.SpMat());
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
M_solver.SetRelTol(1e-9);
+15 -24
View File
@@ -21,11 +21,8 @@
// Device sample runs:
// mpirun -np 4 ex9p -pa
// mpirun -np 4 ex9p -ea
// mpirun -np 4 ex9p -fa
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh -d cuda
// mpirun -np 4 ex9p -ea -m ../data/periodic-cube.mesh -d cuda
// mpirun -np 4 ex9p -fa -m ../data/periodic-cube.mesh -d cuda
//
// Description: This example code solves the time-dependent advection equation
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
@@ -167,7 +164,6 @@ int main(int argc, char *argv[])
int order = 3;
bool pa = false;
bool ea = false;
bool fa = false;
const char *device_config = "cpu";
int ode_solver_type = 4;
double t_final = 10.0;
@@ -197,8 +193,6 @@ int main(int argc, char *argv[])
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
"--no-element-assembly", "Enable Element Assembly.");
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
"--no-full-assembly", "Enable Full Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
@@ -335,12 +329,6 @@ int main(int argc, char *argv[])
m->SetAssemblyLevel(AssemblyLevel::ELEMENT);
k->SetAssemblyLevel(AssemblyLevel::ELEMENT);
}
else if (fa)
{
m->SetAssemblyLevel(AssemblyLevel::FULL);
k->SetAssemblyLevel(AssemblyLevel::FULL);
}
m->AddDomainIntegrator(new MassIntegrator);
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
k->AddInteriorFaceIntegrator(
@@ -577,21 +565,29 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
M_solver(_M.ParFESpace()->GetComm()),
z(_M.Height())
{
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
{
M.Reset(_M.ParallelAssemble(), true);
K.Reset(_K.ParallelAssemble(), true);
}
else
bool pa = _M.GetAssemblyLevel()==AssemblyLevel::PARTIAL;
bool ea = _M.GetAssemblyLevel()==AssemblyLevel::ELEMENT;
if (pa || ea)
{
M.Reset(&_M, false);
K.Reset(&_K, false);
}
else
{
M.Reset(_M.ParallelAssemble(), true);
K.Reset(_K.ParallelAssemble(), true);
}
M_solver.SetOperator(*M);
Array<int> ess_tdof_list;
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
if (pa || ea)
{
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
dg_solver = NULL;
}
else
{
HypreParMatrix &M_mat = *M.As<HypreParMatrix>();
HypreParMatrix &K_mat = *K.As<HypreParMatrix>();
@@ -600,11 +596,6 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
dg_solver = new DG_Solver(M_mat, K_mat, *_M.FESpace());
}
else
{
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
dg_solver = NULL;
}
M_solver.SetPreconditioner(*M_prec);
M_solver.iterative_mode = false;
-37
View File
@@ -50,43 +50,10 @@ set(SRCS
fespacehierarchy.cpp
nonlininteg_vectorconvection.cpp
quadinterpolator.cpp
quadinterpolator_det.cpp
quadinterpolator_eval_by_nodes.cpp
quadinterpolator_eval_by_vdim.cpp
quadinterpolator_grad_by_nodes.cpp
quadinterpolator_grad_by_vdim.cpp
quadinterpolator_grad_phys_by_nodes.cpp
quadinterpolator_grad_phys_by_vdim.cpp
quadinterpolator_face.cpp
restriction.cpp
staticcond.cpp
tmop.cpp
tmop_pa.cpp
tmop_pa_h2d.cpp
tmop_pa_h2d_c0.cpp
tmop_pa_h2m.cpp
tmop_pa_h2m_c0.cpp
tmop_pa_h2s.cpp
tmop_pa_h2s_c0.cpp
tmop_pa_h3d.cpp
tmop_pa_h3d_c0.cpp
tmop_pa_h3m.cpp
tmop_pa_h3m_c0.cpp
tmop_pa_h3s.cpp
tmop_pa_h3s_c0.cpp
tmop_pa_jp2.cpp
tmop_pa_jp3.cpp
tmop_pa_jt2_tc.cpp
tmop_pa_jt3_datc.cpp
tmop_pa_jt3_tc.cpp
tmop_pa_p2.cpp
tmop_pa_p2_c0.cpp
tmop_pa_p3.cpp
tmop_pa_p3_c0.cpp
tmop_pa_w2.cpp
tmop_pa_w2_c0.cpp
tmop_pa_w3.cpp
tmop_pa_w3_c0.cpp
tmop_tools.cpp
gslib.cpp
transfer.cpp
@@ -116,10 +83,7 @@ set(HDRS
nonlinearform_ext.hpp
nonlininteg.hpp
quadinterpolator.hpp
quadinterpolator_eval.hpp
quadinterpolator_face.hpp
quadinterpolator_grad.hpp
quadinterpolator_grad_phys.hpp
restriction.hpp
fespacehierarchy.hpp
staticcond.hpp
@@ -132,7 +96,6 @@ set(HDRS
tfespace.hpp
tintrules.hpp
tmop.hpp
tmop_pa.hpp
tmop_tools.hpp
gslib.hpp
transfer.hpp
+14 -16
View File
@@ -76,7 +76,7 @@ BilinearForm::BilinearForm(FiniteElementSpace * f)
precompute_sparsity = 0;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::LEGACYFULL;
assembly = AssemblyLevel::FULL;
batch = 1;
ext = NULL;
}
@@ -94,7 +94,7 @@ BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
precompute_sparsity = ps;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::LEGACYFULL;
assembly = AssemblyLevel::FULL;
batch = 1;
ext = NULL;
@@ -121,10 +121,9 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::LEGACYFULL:
break;
case AssemblyLevel::FULL:
ext = new FABilinearFormExtension(this);
// ext = new FABilinearFormExtension(this);
// Use the original BilinearForm implementation for now
break;
case AssemblyLevel::ELEMENT:
ext = new EABilinearFormExtension(this);
@@ -144,7 +143,7 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
void BilinearForm::EnableStaticCondensation()
{
delete static_cond;
if (assembly != AssemblyLevel::LEGACYFULL)
if (assembly != AssemblyLevel::FULL)
{
static_cond = NULL;
MFEM_WARNING("Static condensation not supported for this assembly level");
@@ -169,7 +168,7 @@ void BilinearForm::EnableHybridization(FiniteElementSpace *constr_space,
const Array<int> &ess_tdof_list)
{
delete hybridization;
if (assembly != AssemblyLevel::LEGACYFULL)
if (assembly != AssemblyLevel::FULL)
{
delete constr_integ;
hybridization = NULL;
@@ -224,7 +223,7 @@ MatrixInverse * BilinearForm::Inverse() const
void BilinearForm::Finalize (int skip_zeros)
{
if (assembly == AssemblyLevel::LEGACYFULL)
if (assembly == AssemblyLevel::FULL)
{
if (!static_cond) { mat->Finalize(skip_zeros); }
if (mat_e) { mat_e->Finalize(skip_zeros); }
@@ -640,7 +639,8 @@ void BilinearForm::AssembleDiagonal(Vector &diag) const
}
else
{
mat->GetDiag(diag);
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
"matrix and use SparseMatrix::GetDiag?");
}
}
@@ -1083,7 +1083,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
mat = NULL;
mat_e = NULL;
extern_bfs = 0;
assembly = AssemblyLevel::LEGACYFULL;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
@@ -1108,7 +1108,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
bbfi_marker = mbf->bbfi_marker;
btfbfi_marker = mbf->btfbfi_marker;
assembly = AssemblyLevel::LEGACYFULL;
assembly = AssemblyLevel::FULL;
ext = NULL;
}
@@ -1121,8 +1121,6 @@ void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
assembly = assembly_level;
switch (assembly)
{
case AssemblyLevel::LEGACYFULL:
break;
case AssemblyLevel::FULL:
// ext = new FAMixedBilinearFormExtension(this);
// Use the original BilinearForm implementation for now
@@ -1193,7 +1191,7 @@ void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
MatrixInverse * MixedBilinearForm::Inverse() const
{
if (assembly != AssemblyLevel::LEGACYFULL)
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
return NULL;
@@ -1206,7 +1204,7 @@ MatrixInverse * MixedBilinearForm::Inverse() const
void MixedBilinearForm::Finalize (int skip_zeros)
{
if (assembly == AssemblyLevel::LEGACYFULL)
if (assembly == AssemblyLevel::FULL)
{
mat -> Finalize (skip_zeros);
}
@@ -1483,7 +1481,7 @@ void MixedBilinearForm::AssembleDiagonal_ADAt(const Vector &D,
void MixedBilinearForm::ConformingAssemble()
{
if (assembly != AssemblyLevel::LEGACYFULL)
if (assembly != AssemblyLevel::FULL)
{
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
return;
+3 -6
View File
@@ -29,11 +29,8 @@ namespace mfem
form classes derived from Operator. */
enum class AssemblyLevel
{
/// Legacy fully assembled form, i.e. a global sparse matrix in MFEM, Hypre
/// or PETSC format. This assembly is ALWAYS performed on the host.
LEGACYFULL = 0,
/// Fully assembled form, i.e. a global sparse matrix in MFEM format. This
/// assembly is compatible with device execution.
/// Fully assembled form, i.e. a global sparse matrix in MFEM, Hypre or PETSC
/// format.
FULL,
/// Form assembled at element level, which computes and stores dense element
/// matrices.
@@ -122,7 +119,7 @@ protected:
static_cond = NULL; hybridization = NULL;
precompute_sparsity = 0;
diag_policy = DIAG_KEEP;
assembly = AssemblyLevel::LEGACYFULL;
assembly = AssemblyLevel::FULL;
batch = 1;
ext = NULL;
}
+36 -188
View File
@@ -15,7 +15,6 @@
#include "../general/forall.hpp"
#include "bilinearform.hpp"
#include "libceed/ceed.hpp"
#include "pgridfunc.hpp"
namespace mfem
{
@@ -116,7 +115,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (elem_restrict && !DeviceCanUseCeed())
if (elem_restrict)
{
localY = 0.0;
for (int i = 0; i < iSz; ++i)
@@ -293,8 +292,7 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
// Data and methods for element-assembled bilinear forms
EABilinearFormExtension::EABilinearFormExtension(BilinearForm *form)
: PABilinearFormExtension(form),
factorize_face_terms(form->FESpace()->IsDGSpace())
: PABilinearFormExtension(form)
{
}
@@ -349,17 +347,6 @@ void EABilinearFormExtension::Assemble()
{
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr);
}
if (factorize_face_terms && int_face_restrict_lex)
{
auto restFint = dynamic_cast<const L2FaceRestriction&>(*int_face_restrict_lex);
restFint.AddFaceMatricesToElementMatrices(ea_data_int, ea_data);
}
if (factorize_face_terms && bdr_face_restrict_lex)
{
auto restFbdr = dynamic_cast<const L2FaceRestriction&>(*bdr_face_restrict_lex);
restFbdr.AddFaceMatricesToElementMatrices(ea_data_bdr, ea_data);
}
}
void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
@@ -412,27 +399,24 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
const int NDOFS = faceDofs;
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
if (!factorize_face_terms)
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
res += A_int(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
@@ -459,7 +443,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
// Treatment of boundary faces
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
const int bFISz = bdrFaceIntegrators.Size();
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
if (bdr_face_restrict_lex && bFISz>0)
{
// Apply the Boundary Face Restriction
bdr_face_restrict_lex->Mult(x, faceBdrX);
@@ -538,27 +522,24 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
const int NDOFS = faceDofs;
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
if (!factorize_face_terms)
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
double res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
MFEM_FORALL(glob_j, nf_int*NDOFS,
{
@@ -585,7 +566,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
// Treatment of boundary faces
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
const int bFISz = bdrFaceIntegrators.Size();
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
if (bdr_face_restrict_lex && bFISz>0)
{
// Apply the Boundary Face Restriction
bdr_face_restrict_lex->Mult(x, faceBdrX);
@@ -614,139 +595,6 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
}
}
// Data and methods for fully-assembled bilinear forms
FABilinearFormExtension::FABilinearFormExtension(BilinearForm *form)
: EABilinearFormExtension(form),
mat(form->FESpace()->GetVSize(),form->FESpace()->GetVSize(),0),
face_mat(form->FESpace()->GetVSize(),0,0),
use_face_mat(false)
{
#ifdef MFEM_USE_MPI
if ( ParFiniteElementSpace* pfes =
dynamic_cast<ParFiniteElementSpace*>(form->FESpace()) )
{
if (pfes->IsDGSpace())
{
use_face_mat = true;
pfes->ExchangeFaceNbrData();
face_mat.SetWidth(pfes->GetFaceNbrVSize());
}
}
#endif
}
void FABilinearFormExtension::Assemble()
{
EABilinearFormExtension::Assemble();
FiniteElementSpace &fes = *a->FESpace();
if (fes.IsDGSpace())
{
const L2ElementRestriction *restE =
static_cast<const L2ElementRestriction*>(elem_restrict);
const L2FaceRestriction *restF =
static_cast<const L2FaceRestriction*>(int_face_restrict_lex);
// 1. Fill I
// 1.1 Increment with restE
restE->FillI(mat);
// 1.2 Increment with restF
if (restF) { restF->FillI(mat, face_mat); }
// 1.3 Sum the non-zeros in I
auto h_I = mat.HostReadWriteI();
int cpt = 0;
const int vd = fes.GetVDim();
const int ndofs = ne*elemDofs*vd;
for (int i = 0; i < ndofs; i++)
{
const int nnz = h_I[i];
h_I[i] = cpt;
cpt += nnz;
}
const int nnz = cpt;
h_I[ndofs] = nnz;
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
if (use_face_mat && restF)
{
auto h_I_face = face_mat.HostReadWriteI();
int cpt = 0;
for (int i = 0; i < ndofs; i++)
{
const int nnz = h_I_face[i];
h_I_face[i] = cpt;
cpt += nnz;
}
const int nnz_face = cpt;
h_I_face[ndofs] = nnz_face;
face_mat.GetMemoryJ().New(nnz_face,
face_mat.GetMemoryJ().GetMemoryType());
face_mat.GetMemoryData().New(nnz_face,
face_mat.GetMemoryData().GetMemoryType());
}
// 2. Fill J and Data
// 2.1 Fill J and Data with Elem ea_data
restE->FillJAndData(ea_data, mat);
// 2.2 Fill J and Data with Face ea_data_ext
if (restF) { restF->FillJAndData(ea_data_ext, mat, face_mat); }
// 2.3 Shift indirections in I back to original
auto I = mat.HostReadWriteI();
for (int i = ndofs; i > 0; i--)
{
I[i] = I[i-1];
}
I[0] = 0;
if (use_face_mat && restF)
{
auto I_face = face_mat.HostReadWriteI();
for (int i = ndofs; i > 0; i--)
{
I_face[i] = I_face[i-1];
}
I_face[0] = 0;
}
}
else // continuous Galerkin case
{
const ElementRestriction &rest =
static_cast<const ElementRestriction&>(*elem_restrict);
rest.FillSparseMatrix(ea_data, mat);
}
}
void FABilinearFormExtension::Mult(const Vector &x, Vector &y) const
{
mat.Mult(x, y);
#ifdef MFEM_USE_MPI
if (const ParFiniteElementSpace *pfes =
dynamic_cast<const ParFiniteElementSpace*>(testFes))
{
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
const_cast<Vector&>(x),0);
x_gf.ExchangeFaceNbrData();
Vector &shared_x = x_gf.FaceNbrData();
if (shared_x.Size()) { face_mat.AddMult(shared_x, y); }
}
#endif
}
void FABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
{
mat.MultTranspose(x, y);
#ifdef MFEM_USE_MPI
if (const ParFiniteElementSpace *pfes =
dynamic_cast<const ParFiniteElementSpace*>(testFes))
{
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
const_cast<Vector&>(x),0);
x_gf.ExchangeFaceNbrData();
Vector &shared_x = x_gf.FaceNbrData();
if (shared_x.Size()) { face_mat.AddMultTranspose(shared_x, y); }
}
#endif
}
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
: Operator(form->Height(), form->Width()), a(form)
{
+21 -19
View File
@@ -62,6 +62,27 @@ public:
virtual void Update() = 0;
};
/** @brief Data and methods for fully-assembled bilinear forms.
Not yet implemented! Use the BilinearForm Class instead. */
class FABilinearFormExtension : public BilinearFormExtension
{
public:
FABilinearFormExtension(BilinearForm *form)
: BilinearFormExtension(form) { }
/// TODO
void Assemble() {}
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A) {}
void FormLinearSystem(const Array<int> &ess_tdof_list,
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0) {}
void Mult(const Vector &x, Vector &y) const {}
void MultTranspose(const Vector &x, Vector &y) const {}
void Update() {}
~FABilinearFormExtension() {}
};
/// Data and methods for partially-assembled bilinear forms
class PABilinearFormExtension : public BilinearFormExtension
{
@@ -98,12 +119,10 @@ class EABilinearFormExtension : public PABilinearFormExtension
protected:
int ne;
int elemDofs;
// The element matrices are stored row major
Vector ea_data;
int nf_int, nf_bdr;
int faceDofs;
Vector ea_data_int, ea_data_ext, ea_data_bdr;
bool factorize_face_terms;
public:
EABilinearFormExtension(BilinearForm *form);
@@ -113,23 +132,6 @@ public:
void MultTranspose(const Vector &x, Vector &y) const;
};
/// Data and methods for fully-assembled bilinear forms
class FABilinearFormExtension : public EABilinearFormExtension
{
private:
SparseMatrix mat;
/// face_mat handles parallelism for DG face terms.
SparseMatrix face_mat;
bool use_face_mat;
public:
FABilinearFormExtension(BilinearForm *form);
void Assemble();
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
};
/// Data and methods for matrix-free bilinear forms NOT YET IMPLEMENTED.
class MFBilinearFormExtension : public BilinearFormExtension
{
+2 -32
View File
@@ -1685,22 +1685,6 @@ protected:
{
trial_fe.CalcPhysCurlShape(Trans, shape);
}
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector&, Vector&) const;
private:
// PA extension
Vector pa_data;
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const DofToQuad *mapsOtest; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsCtest; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, dofs1Dtest,quad1D, testType, trialType, coeffDim;
};
/** Class for integrating the bilinear form a(u,v) := (Q u, curl v) in 3D and
@@ -1740,20 +1724,6 @@ protected:
{
test_fe.CalcPhysCurlShape(Trans, shape);
}
using BilinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes);
virtual void AddMultPA(const Vector&, Vector&) const;
private:
// PA extension
Vector pa_data;
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
const GeometricFactors *geom; ///< Not owned
int dim, ne, dofs1D, quad1D, testType, trialType, coeffDim;
};
/** Class for integrating the bilinear form a(u,v) := - (Q u, grad v) in either
@@ -1954,7 +1924,7 @@ public:
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
void SetupPA(const FiniteElementSpace &fes);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
@@ -2030,7 +2000,7 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans);
void SetupPA(const FiniteElementSpace &fes);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Mass integrator (u, v) restricted to the boundary of a domain */
+6 -6
View File
@@ -32,7 +32,7 @@ static void EAConvectionAssemble1D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -54,7 +54,7 @@ static void EAConvectionAssemble1D(const int NE,
{
val += r_Bj[k1] * D(k1, e) * r_Gi[k1];
}
A(i1, j1, e) += val;
A(i1, j1, e) = val;
}
}
});
@@ -76,7 +76,7 @@ static void EAConvectionAssemble2D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -121,7 +121,7 @@ static void EAConvectionAssemble2D(const int NE,
* r_B[k1][j1]* r_B[k2][j2];
}
}
A(i1, i2, j1, j2, e) += val;
A(i1, i2, j1, j2, e) = val;
}
}
}
@@ -145,7 +145,7 @@ static void EAConvectionAssemble3D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 3, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -191,7 +191,7 @@ static void EAConvectionAssemble3D(const int NE,
}
}
}
A(i1, i2, i3, j1, j2, j3, e) += val;
A(i1, i2, i3, j1, j2, j3, e) = val;
}
}
}
+73 -289
View File
@@ -13,10 +13,6 @@
#include "bilininteg.hpp"
#include "gridfunc.hpp"
#include "restriction.hpp"
#include "tmop_pa.hpp"
#include "../linalg/kernels.hpp"
using namespace std;
namespace mfem
@@ -72,53 +68,47 @@ static void PAConvectionSetup3D(const int Q1D,
const double alpha,
Vector &op)
{
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
const bool const_v = vel.Size() == 3;
const auto V = const_v ?
Reshape(vel.Read(), 3,1,1,1,1) :
Reshape(vel.Read(), 3,Q1D,Q1D,Q1D,NE);
auto y = Reshape(op.Write(), Q1D,Q1D,Q1D,3,NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
auto V =
const_v ? Reshape(vel.Read(), 3,1,1) : Reshape(vel.Read(), 3,NQ,NE);
auto y = Reshape(op.Write(), NQ, 3, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
const double J11 = J(qx,qy,qz,0,0,e);
const double J12 = J(qx,qy,qz,0,1,e);
const double J13 = J(qx,qy,qz,0,2,e);
const double J21 = J(qx,qy,qz,1,0,e);
const double J22 = J(qx,qy,qz,1,1,e);
const double J23 = J(qx,qy,qz,1,2,e);
const double J31 = J(qx,qy,qz,2,0,e);
const double J32 = J(qx,qy,qz,2,1,e);
const double J33 = J(qx,qy,qz,2,2,e);
const double w = alpha * W(qx,qy,qz);
const double v0 = const_v ? V(0,0,0,0,0) : V(0,qx,qy,qz,e);
const double v1 = const_v ? V(1,0,0,0,0) : V(1,qx,qy,qz,e);
const double v2 = const_v ? V(2,0,0,0,0) : V(2,qx,qy,qz,e);
const double wx = w * v0;
const double wy = w * v1;
const double wz = w * v2;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// q . J^{-1} = q . adj(J)
y(qx,qy,qz,0,e) = wx * A11 + wy * A12 + wz * A13;
y(qx,qy,qz,1,e) = wx * A21 + wy * A22 + wz * A23;
y(qx,qy,qz,2,e) = wx * A31 + wy * A32 + wz * A33;
}
}
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double w = alpha * W[q];
const double v0 = const_v ? V(0,0,0) : V(0,q,e);
const double v1 = const_v ? V(1,0,0) : V(1,q,e);
const double v2 = const_v ? V(2,0,0) : V(2,q,e);
const double wx = w * v0;
const double wy = w * v1;
const double wz = w * v2;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// q . J^{-1} = q . adj(J)
y(q,0,e) = wx * A11 + wy * A12 + wz * A13;
y(q,1,e) = wx * A21 + wy * A22 + wz * A23;
y(q,2,e) = wx * A31 + wy * A32 + wz * A33;
}
});
}
@@ -194,8 +184,8 @@ void PAConvectionApply2D(const int ne,
Gu[dy][qx] = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double x = u[dy][dx];
Bu[dy][qx] += bx * x;
Gu[dy][qx] += gx * x;
@@ -212,8 +202,8 @@ void PAConvectionApply2D(const int ne,
BGu[qy][qx] = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
const double bx = B(qy,dy);
const double gx = G(qy,dy);
const double bx = B(qy,dy);
const double gx = G(qy,dy);
GBu[qy][qx] += gx * Bu[dy][qx];
BGu[qy][qx] += bx * Gu[dy][qx];
}
@@ -242,7 +232,7 @@ void PAConvectionApply2D(const int ne,
BDGu[dy][qx] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double w = Bt(dy,qy);
const double w = Bt(dy,qy);
BDGu[dy][qx] += w * DGu[qy][qx];
}
}
@@ -254,7 +244,7 @@ void PAConvectionApply2D(const int ne,
double BBDGu = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double w = Bt(dx,qx);
const double w = Bt(dx,qx);
BBDGu += w * BDGu[dy][qx];
}
y(dx,dy,e) += BBDGu;
@@ -320,7 +310,7 @@ void SmemPAConvectionApply2D(const int ne,
{
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double x = u[tidz][dy][dx];
const double x = u[tidz][dy][dx];
Bu[tidz][dy][qx] += bx * x;
Gu[tidz][dy][qx] += gx * x;
}
@@ -337,8 +327,8 @@ void SmemPAConvectionApply2D(const int ne,
BGu[tidz][qy][qx] = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
const double bx = B(qy,dy);
const double gx = G(qy,dy);
const double bx = B(qy,dy);
const double gx = G(qy,dy);
GBu[tidz][qy][qx] += gx * Bu[tidz][dy][qx];
BGu[tidz][qy][qx] += bx * Gu[tidz][dy][qx];
}
@@ -369,7 +359,7 @@ void SmemPAConvectionApply2D(const int ne,
BDGu[tidz][dy][qx] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double w = Bt(dy,qy);
const double w = Bt(dy,qy);
BDGu[tidz][dy][qx] += w * DGu[tidz][qy][qx];
}
}
@@ -382,7 +372,7 @@ void SmemPAConvectionApply2D(const int ne,
double BBDGu = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double w = Bt(dx,qx);
const double w = Bt(dx,qx);
BBDGu += w * BDGu[tidz][dy][qx];
}
y(dx,dy,e) += BBDGu;
@@ -446,8 +436,8 @@ void PAConvectionApply3D(const int ne,
Gu[dz][dy][qx] = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double x = u[dz][dy][dx];
Bu[dz][dy][qx] += bx * x;
Gu[dz][dy][qx] += gx * x;
@@ -469,8 +459,8 @@ void PAConvectionApply3D(const int ne,
BGu[dz][qy][qx] = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
const double bx = B(qy,dy);
const double gx = G(qy,dy);
const double bx = B(qy,dy);
const double gx = G(qy,dy);
BBu[dz][qy][qx] += bx * Bu[dz][dy][qx];
GBu[dz][qy][qx] += gx * Bu[dz][dy][qx];
BGu[dz][qy][qx] += bx * Gu[dz][dy][qx];
@@ -492,8 +482,8 @@ void PAConvectionApply3D(const int ne,
BBGu[qz][qy][qx] = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
const double bx = B(qz,dz);
const double gx = G(qz,dz);
const double bx = B(qz,dz);
const double gx = G(qz,dz);
GBBu[qz][qy][qx] += gx * BBu[dz][qy][qx];
BGBu[qz][qy][qx] += bx * GBu[dz][qy][qx];
BBGu[qz][qy][qx] += bx * BGu[dz][qy][qx];
@@ -531,7 +521,7 @@ void PAConvectionApply3D(const int ne,
BDGu[dz][qy][qx] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const double w = Bt(dz,qz);
const double w = Bt(dz,qz);
BDGu[dz][qy][qx] += w * DGu[qz][qy][qx];
}
}
@@ -547,7 +537,7 @@ void PAConvectionApply3D(const int ne,
BBDGu[dz][dy][qx] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double w = Bt(dy,qy);
const double w = Bt(dy,qy);
BBDGu[dz][dy][qx] += w * BDGu[dz][qy][qx];
}
}
@@ -562,7 +552,7 @@ void PAConvectionApply3D(const int ne,
double BBBDGu = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double w = Bt(dx,qx);
const double w = Bt(dx,qx);
BBBDGu += w * BBDGu[dz][dy][qx];
}
y(dx,dy,dz,e) += BBBDGu;
@@ -635,8 +625,8 @@ void SmemPAConvectionApply3D(const int ne,
double Gu_ = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double bx = B(qx,dx);
const double gx = G(qx,dx);
const double x = u[dz][dy][dx];
Bu_ += bx * x;
Gu_ += gx * x;
@@ -661,8 +651,8 @@ void SmemPAConvectionApply3D(const int ne,
double BGu_ = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
const double bx = B(qy,dy);
const double gx = G(qy,dy);
const double bx = B(qy,dy);
const double gx = G(qy,dy);
BBu_ += bx * Bu[dz][dy][qx];
GBu_ += gx * Bu[dz][dy][qx];
BGu_ += bx * Gu[dz][dy][qx];
@@ -688,8 +678,8 @@ void SmemPAConvectionApply3D(const int ne,
double BBGu_ = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
const double bx = B(qz,dz);
const double gx = G(qz,dz);
const double bx = B(qz,dz);
const double gx = G(qz,dz);
GBBu_ += gx * BBu[dz][qy][qx];
BGBu_ += bx * GBu[dz][qy][qx];
BBGu_ += bx * BGu[dz][qy][qx];
@@ -731,7 +721,7 @@ void SmemPAConvectionApply3D(const int ne,
double BDGu_ = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const double w = Bt(dz,qz);
const double w = Bt(dz,qz);
BDGu_ += w * DGu[qz][qy][qx];
}
BDGu[dz][qy][qx] = BDGu_;
@@ -749,7 +739,7 @@ void SmemPAConvectionApply3D(const int ne,
double BBDGu_ = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const double w = Bt(dy,qy);
const double w = Bt(dy,qy);
BBDGu_ += w * BDGu[dz][qy][qx];
}
BBDGu[dz][dy][qx] = BBDGu_;
@@ -766,7 +756,7 @@ void SmemPAConvectionApply3D(const int ne,
double BBBDGu = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const double w = Bt(dx,qx);
const double w = Bt(dx,qx);
BBBDGu += w * BBDGu[dz][dy][qx];
}
y(dx,dy,dz,e) = BBBDGu;
@@ -776,117 +766,6 @@ void SmemPAConvectionApply3D(const int ne,
});
}
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
static void QEvalVGF2D(const int NE,
const double *b_,
const double *x_,
double *y_,
const int vdim = 1,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto X = Reshape(x_, D1D, D1D, VDIM, NE);
auto C = Reshape(y_, VDIM, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
{
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
MFEM_SHARED double B[MQ1*MD1];
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
MFEM_SHARED double DD[NBZ][MD1*MD1];
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
for (int c = 0; c < VDIM; c++)
{
mfem::kernels::LoadX<MD1,NBZ>(e,D1D,c,X,DD);
mfem::kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,DD,DQ);
mfem::kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
double G;
mfem::kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,G);
C(c,qx,qy,e) = G;
}
}
MFEM_SYNC_THREAD;
}
});
}
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
static void QEvalVGF3D(const int NE,
const double *b_,
const double *x_,
double *y_,
const int vdim = 1,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto X = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
auto C = Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
MFEM_SHARED double B[MQ1*MD1];
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
MFEM_SHARED double DDD[MD1*MD1*MD1];
MFEM_SHARED double DDQ[MD1*MD1*MQ1];
MFEM_SHARED double DQQ[MD1*MQ1*MQ1];
MFEM_SHARED double QQQ[MQ1*MQ1*MQ1];
for (int c = 0; c < VDIM; c++)
{
mfem::kernels::LoadX<MD1>(e,D1D,c,X,DDD);
mfem::kernels::EvalX<MD1,MQ1>(D1D,Q1D,B,DDD,DDQ);
mfem::kernels::EvalY<MD1,MQ1>(D1D,Q1D,B,DDQ,DQQ);
mfem::kernels::EvalZ<MD1,MQ1>(D1D,Q1D,B,DQQ,QQQ);
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
double G;
mfem::kernels::PullEval<MQ1>(qx,qy,qz,QQQ,G);
C(c,qx,qy,qz,e) = G;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
@@ -899,104 +778,16 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = fes.GetNE();
const DofToQuad::Mode mode = DofToQuad::TENSOR;
#ifdef MFEM_USE_UMPIRE
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
#else
const MemoryType temp_type = Device::GetDeviceMemoryType();
#endif
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode, temp_type);
maps = &el.GetDofToQuad(*ir, mode);
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(symmDims * nq * ne, temp_type);
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
Vector vel;
if (VectorConstantCoefficient *cQ =
dynamic_cast<VectorConstantCoefficient*>(Q))
if (VectorConstantCoefficient *cQ = dynamic_cast<VectorConstantCoefficient*>(Q))
{
vel = cQ->GetVec();
}
else if (VectorGridFunctionCoefficient *vgfQ =
dynamic_cast<VectorGridFunctionCoefficient*>(Q))
{
Vector xe;
vel.SetSize(dim * nq * ne, temp_type);
const GridFunction *gf = vgfQ->GetGridFunction();
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
const FiniteElementSpace &gf_fes = *gf->FESpace();
const int vdim = gf_fes.GetVDim();
const Operator *R = gf_fes.GetElementRestriction(ordering);
const FiniteElement &el_gf = *gf_fes.GetFE(0);
const DofToQuad *maps_gf = &el_gf.GetDofToQuad(*ir, mode);
const int D1D = maps_gf->ndof;
const int Q1D = maps_gf->nqpt;
MFEM_VERIFY(R,"");
MFEM_VERIFY(vdim == dim, "");
MFEM_VERIFY(dim==2 || dim==3,"");
xe.SetSize(R->Height(), Device::GetMemoryType());
xe.UseDevice(true);
R->Mult(*gf, xe);
const auto B = maps_gf->B.Read();
const auto x = xe.Read();
auto y = vel.Write();
const int id = (D1D << 4 ) | Q1D;
if (dim == 2)
{
switch (id)
{
case 0x22: QEvalVGF2D<2,2,2>(ne,B,x,y); break;
case 0x33: QEvalVGF2D<2,3,3>(ne,B,x,y); break;
case 0x34: QEvalVGF2D<2,3,4>(ne,B,x,y); break;
default:
{
constexpr int MAX_DQ = 8;
MFEM_VERIFY(D1D <= MAX_DQ, "");
MFEM_VERIFY(Q1D <= MAX_DQ, "");
QEvalVGF2D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
}
}
}
if (dim == 3)
{
switch (id)
{
case 0x23: QEvalVGF3D<3,2,3>(ne,B,x,y); break;
case 0x34: QEvalVGF3D<3,3,4>(ne,B,x,y); break;
case 0x35: QEvalVGF3D<3,3,5>(ne,B,x,y); break;
case 0x46: QEvalVGF3D<3,4,6>(ne,B,x,y); break;
case 0x48: QEvalVGF3D<3,4,8>(ne,B,x,y); break;
default:
{
constexpr int MAX_DQ = 6;
MFEM_VERIFY(D1D <= MAX_DQ, "");
MFEM_VERIFY(Q1D <= MAX_DQ, "");
QEvalVGF3D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
}
}
}
}
else if (VectorQuadratureFunctionCoefficient* cQ =
dynamic_cast<VectorQuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == dim * nq * ne,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
vel.SetSize(dim * nq * ne);
@@ -1036,12 +827,9 @@ static void PAConvectionApply(const int dim,
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPAConvectionApply2D<2,2,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x33: return SmemPAConvectionApply2D<3,3,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x34: return SmemPAConvectionApply2D<3,4,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x44: return SmemPAConvectionApply2D<4,4,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x46: return SmemPAConvectionApply2D<4,6,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x33: return SmemPAConvectionApply2D<3,3,3>(NE,B,G,Bt,Gt,op,x,y);
case 0x44: return SmemPAConvectionApply2D<4,4,2>(NE,B,G,Bt,Gt,op,x,y);
case 0x55: return SmemPAConvectionApply2D<5,5,2>(NE,B,G,Bt,Gt,op,x,y);
case 0x58: return SmemPAConvectionApply2D<5,8,2>(NE,B,G,Bt,Gt,op,x,y);
case 0x66: return SmemPAConvectionApply2D<6,6,1>(NE,B,G,Bt,Gt,op,x,y);
case 0x77: return SmemPAConvectionApply2D<7,7,1>(NE,B,G,Bt,Gt,op,x,y);
case 0x88: return SmemPAConvectionApply2D<8,8,1>(NE,B,G,Bt,Gt,op,x,y);
@@ -1054,12 +842,8 @@ static void PAConvectionApply(const int dim,
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPAConvectionApply3D<2,3>(NE,B,G,Bt,Gt,op,x,y);
case 0x24: return SmemPAConvectionApply3D<2,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x26: return SmemPAConvectionApply3D<2,6>(NE,B,G,Bt,Gt,op,x,y);
case 0x34: return SmemPAConvectionApply3D<3,4>(NE,B,G,Bt,Gt,op,x,y);
case 0x35: return SmemPAConvectionApply3D<3,5>(NE,B,G,Bt,Gt,op,x,y);
case 0x45: return SmemPAConvectionApply3D<4,5>(NE,B,G,Bt,Gt,op,x,y);
case 0x48: return SmemPAConvectionApply3D<4,8>(NE,B,G,Bt,Gt,op,x,y);
case 0x56: return SmemPAConvectionApply3D<5,6>(NE,B,G,Bt,Gt,op,x,y);
case 0x67: return SmemPAConvectionApply3D<6,7>(NE,B,G,Bt,Gt,op,x,y);
case 0x78: return SmemPAConvectionApply3D<7,8>(NE,B,G,Bt,Gt,op,x,y);
-27
View File
@@ -167,19 +167,6 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
r.SetSize(1);
r(0) = c_rho->constant;
}
else if (QuadratureFunctionCoefficient* c_rho =
dynamic_cast<QuadratureFunctionCoefficient*>(rho))
{
const QuadratureFunction &qFun = c_rho->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == nq * nf,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
r.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
r.SetSize(nq * nf);
@@ -213,20 +200,6 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
{
vel = c_u->GetVec();
}
else if (VectorQuadratureFunctionCoefficient* c_u =
dynamic_cast<VectorQuadratureFunctionCoefficient*>(u))
{
// Assumed to be in lexicographical ordering
const QuadratureFunction &qFun = c_u->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == dim * nq * nf,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
vel.SetSize(dim * nq * nf);
+6 -6
View File
@@ -31,7 +31,7 @@ static void EADiffusionAssemble1D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -53,7 +53,7 @@ static void EADiffusionAssemble1D(const int NE,
{
val += r_Gj[k1] * D(k1, e) * r_Gi[k1];
}
A(i1, j1, e) += val;
A(i1, j1, e) = val;
}
}
});
@@ -75,7 +75,7 @@ static void EADiffusionAssemble2D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, 3, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -120,7 +120,7 @@ static void EADiffusionAssemble2D(const int NE,
+ gbi * D11 * gbj;
}
}
A(i1, i2, j1, j2, e) += val;
A(i1, i2, j1, j2, e) = val;
}
}
}
@@ -144,7 +144,7 @@ static void EADiffusionAssemble3D(const int NE,
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 6, NE);
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -208,7 +208,7 @@ static void EADiffusionAssemble3D(const int NE,
}
}
}
A(i1, i2, i3, j1, j2, j3, e) += val;
A(i1, i2, i3, j1, j2, j3, e) = val;
}
}
}
+215 -301
View File
@@ -170,53 +170,47 @@ static void PADiffusionSetup3D(const int Q1D,
const Vector &c,
Vector &d)
{
const int NQ = Q1D*Q1D*Q1D;
const bool const_c = c.Size() == 1;
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
Reshape(c.Read(), Q1D,Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, 6, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 6, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
const double J11 = J(qx,qy,qz,0,0,e);
const double J21 = J(qx,qy,qz,1,0,e);
const double J31 = J(qx,qy,qz,2,0,e);
const double J12 = J(qx,qy,qz,0,1,e);
const double J22 = J(qx,qy,qz,1,1,e);
const double J32 = J(qx,qy,qz,2,1,e);
const double J13 = J(qx,qy,qz,0,2,e);
const double J23 = J(qx,qy,qz,1,2,e);
const double J33 = J(qx,qy,qz,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
const double c_detJ = W(qx,qy,qz) * coeff / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
D(qx,qy,qz,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
D(qx,qy,qz,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
D(qx,qy,qz,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
D(qx,qy,qz,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
D(qx,qy,qz,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
D(qx,qy,qz,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
}
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0) : C(q,e);
const double c_detJ = W[q] * coeff / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
D(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
D(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
D(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
D(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
D(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
D(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
@@ -259,7 +253,8 @@ static void PADiffusionSetup(const int dim,
}
}
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
const bool force)
{
// Assuming the same element type
fespace = &fes;
@@ -268,7 +263,7 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
@@ -276,16 +271,17 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
InitCeedCoeff(Q, ptr);
return CeedPADiffusionAssemble(fes, *ir, *ptr);
}
#else
MFEM_CONTRACT_VAR(force);
#endif
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
ne = fes.GetNE();
const DofToQuad::Mode mode = DofToQuad::TENSOR;
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
const int sdim = mesh->SpaceDimension();
maps = &el.GetDofToQuad(*ir, mode);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
@@ -300,19 +296,6 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
coeff.SetSize(1);
coeff(0) = cQ->constant;
}
else if (QuadratureFunctionCoefficient* cQ =
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == ne*nq,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
coeff.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
}
else
{
coeff.SetSize(nq * ne);
@@ -740,7 +723,6 @@ static void PADiffusionAssembleDiagonal(const int dim,
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,B,G,D,Y);
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,B,G,D,Y);
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,B,G,D,Y);
case 0x46: return SmemPADiffusionDiagonal3D<4,6>(NE,B,G,D,Y);
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,B,G,D,Y);
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,B,G,D,Y);
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,B,G,D,Y);
@@ -754,17 +736,9 @@ static void PADiffusionAssembleDiagonal(const int dim,
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAssembleDiagonalPA(ceedDataPtr, diag);
}
else
#endif
{
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
@@ -1333,33 +1307,7 @@ static void PADiffusionApply3D(const int NE,
});
}
// Half of B and G are stored in shared to get B, Bt, G and Gt.
// Indices computation for SmemPADiffusionApply3D.
static MFEM_HOST_DEVICE inline int qi(const int q, const int d, const int Q)
{
return (q<=d) ? q : Q-1-q;
}
static MFEM_HOST_DEVICE inline int dj(const int q, const int d, const int D)
{
return (q<=d) ? d : D-1-d;
}
static MFEM_HOST_DEVICE inline int qk(const int q, const int d, const int Q)
{
return (q<=d) ? Q-1-q : q;
}
static MFEM_HOST_DEVICE inline int dl(const int q, const int d, const int D)
{
return (q<=d) ? D-1-d : d;
}
static MFEM_HOST_DEVICE inline double sign(const int q, const int d)
{
return (q<=d) ? -1.0 : 1.0;
}
// Shared memory PA Diffusion Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
static void SmemPADiffusionApply3D(const int NE,
const Array<double> &b_,
@@ -1372,27 +1320,28 @@ static void SmemPADiffusionApply3D(const int NE,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= M1D, "");
MFEM_VERIFY(Q1D <= M1Q, "");
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
MFEM_VERIFY(D1D <= MD1, "");
MFEM_VERIFY(Q1D <= MQ1, "");
auto b = Reshape(b_.Read(), Q1D, D1D);
auto g = Reshape(g_.Read(), Q1D, D1D);
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, 6, NE);
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, 1,
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int tidz = MFEM_THREAD_ID(z);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
MFEM_SHARED double sBG[MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) sBG;
double (*G)[MD1] = (double (*)[MD1]) sBG;
double (*Bt)[MQ1] = (double (*)[MQ1]) sBG;
double (*Gt)[MQ1] = (double (*)[MQ1]) sBG;
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
MFEM_SHARED double sBG[2][MQ1*MD1];
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
double (*Gt)[MQ1] = (double (*)[MQ1]) (sBG+1);
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
@@ -1410,127 +1359,108 @@ static void SmemPADiffusionApply3D(const int NE,
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
}
MFEM_FOREACH_THREAD(qx,x,Q1D)
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
const int i = qi(qx,dy,Q1D);
const int j = dj(qx,dy,D1D);
const int k = qk(qx,dy,Q1D);
const int l = dl(qx,dy,D1D);
B[i][j] = b(qx,dy);
G[k][l] = g(qx,dy) * sign(qx,dy);
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B[q][d] = b(q,d);
G[q][d] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
double u[D1D], v[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dx = 0; dx < D1D; ++dx)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int i = qi(qx,dx,Q1D);
const int j = dj(qx,dx,D1D);
const int k = qk(qx,dx,Q1D);
const int l = dl(qx,dx,D1D);
const double s = sign(qx,dx);
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double coords = X[dz][dy][dx];
u[dz] += coords * B[i][j];
v[dz] += coords * G[k][l] * s;
u += coords * B[qx][dx];
v += coords * G[qx][dx];
}
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
DDQ0[dz][dy][qx] = u[dz];
DDQ1[dz][dy][qx] = v[dz];
DDQ0[dz][dy][qx] = u;
DDQ1[dz][dy][qx] = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
double u[D1D], v[D1D], w[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = w[dz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dy = 0; dy < D1D; ++dy)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const int i = qi(qy,dy,Q1D);
const int j = dj(qy,dy,D1D);
const int k = qk(qy,dy,Q1D);
const int l = dl(qy,dy,D1D);
const double s = sign(qy,dy);
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++)
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u[dz] += DDQ1[dz][dy][qx] * B[i][j];
v[dz] += DDQ0[dz][dy][qx] * G[k][l] * s;
w[dz] += DDQ0[dz][dy][qx] * B[i][j];
u += DDQ1[dz][dy][qx] * B[qy][dy];
v += DDQ0[dz][dy][qx] * G[qy][dy];
w += DDQ0[dz][dy][qx] * B[qy][dy];
}
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; dz++)
{
DQQ0[dz][qy][qx] = u[dz];
DQQ1[dz][qy][qx] = v[dz];
DQQ2[dz][qy][qx] = w[dz];
DQQ0[dz][qy][qx] = u;
DQQ1[dz][qy][qx] = v;
DQQ2[dz][qy][qx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
const int i = qi(qz,dz,Q1D);
const int j = dj(qz,dz,D1D);
const int k = qk(qz,dz,Q1D);
const int l = dl(qz,dz,D1D);
const double s = sign(qz,dz);
u[qz] += DQQ0[dz][qy][qx] * B[i][j];
v[qz] += DQQ1[dz][qy][qx] * B[i][j];
w[qz] += DQQ2[dz][qy][qx] * G[k][l] * s;
u += DQQ0[dz][qy][qx] * B[qz][dz];
v += DQQ1[dz][qy][qx] * B[qz][dz];
w += DQQ2[dz][qy][qx] * G[qz][dz];
}
QQQ0[qz][qy][qx] = u;
QQQ1[qz][qy][qx] = v;
QQQ2[qz][qy][qx] = w;
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const double O11 = d(qx,qy,qz,0,e);
const double O12 = d(qx,qy,qz,1,e);
const double O13 = d(qx,qy,qz,2,e);
const double O22 = d(qx,qy,qz,3,e);
const double O23 = d(qx,qy,qz,4,e);
const double O33 = d(qx,qy,qz,5,e);
const double gX = u[qz];
const double gY = v[qz];
const double gZ = w[qz];
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
const double O11 = d(q,0,e);
const double O12 = d(q,1,e);
const double O13 = d(q,2,e);
const double O22 = d(q,3,e);
const double O23 = d(q,4,e);
const double O33 = d(q,5,e);
const double gX = QQQ0[qz][qy][qx];
const double gY = QQQ1[qz][qy][qx];
const double gZ = QQQ2[qz][qy][qx];
QQQ0[qz][qy][qx] = (O11*gX) + (O12*gY) + (O13*gZ);
QQQ1[qz][qy][qx] = (O12*gX) + (O22*gY) + (O23*gZ);
QQQ2[qz][qy][qx] = (O13*gX) + (O23*gY) + (O33*gZ);
@@ -1538,112 +1468,78 @@ static void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(d,y,D1D)
if (tidz == 0)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
MFEM_FOREACH_THREAD(d,y,D1D)
{
const int i = qi(q,d,Q1D);
const int j = dj(q,d,D1D);
const int k = qk(q,d,Q1D);
const int l = dl(q,d,D1D);
Bt[j][i] = b(q,d);
Gt[l][k] = g(q,d) * sign(q,d);
MFEM_FOREACH_THREAD(q,x,Q1D)
{
Bt[d][q] = b(q,d);
Gt[d][q] = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qx = 0; qx < Q1D; ++qx)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
const int i = qi(qx,dx,Q1D);
const int j = dj(qx,dx,D1D);
const int k = qk(qx,dx,Q1D);
const int l = dl(qx,dx,D1D);
const double s = sign(qx,dx);
MFEM_UNROLL(MQ1)
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
u += QQQ0[qz][qy][qx] * Gt[dx][qx];
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
}
QQD0[qz][qy][dx] = u;
QQD1[qz][qy][dx] = v;
QQD2[qz][qy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += QQD0[qz][qy][dx] * Bt[dy][qy];
v += QQD1[qz][qy][dx] * Gt[dy][qy];
w += QQD2[qz][qy][dx] * Bt[dy][qy];
}
QDD0[qz][dy][dx] = u;
QDD1[qz][dy][dx] = v;
QDD2[qz][dy][dx] = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
u[qz] += QQQ0[qz][qy][qx] * Gt[l][k] * s;
v[qz] += QQQ1[qz][qy][qx] * Bt[j][i];
w[qz] += QQQ2[qz][qy][qx] * Bt[j][i];
u += QDD0[qz][dy][dx] * Bt[dz][qz];
v += QDD1[qz][dy][dx] * Bt[dz][qz];
w += QDD2[qz][dy][dx] * Gt[dz][qz];
}
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
QQD0[qz][qy][dx] = u[qz];
QQD1[qz][qy][dx] = v[qz];
QQD2[qz][qy][dx] = w[qz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u[Q1D], v[Q1D], w[Q1D];
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qy = 0; qy < Q1D; ++qy)
{
const int i = qi(qy,dy,Q1D);
const int j = dj(qy,dy,D1D);
const int k = qk(qy,dy,Q1D);
const int l = dl(qy,dy,D1D);
const double s = sign(qy,dy);
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
u[qz] += QQD0[qz][qy][dx] * Bt[j][i];
v[qz] += QQD1[qz][qy][dx] * Gt[l][k] * s;
w[qz] += QQD2[qz][qy][dx] * Bt[j][i];
}
}
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
QDD0[qz][dy][dx] = u[qz];
QDD1[qz][dy][dx] = v[qz];
QDD2[qz][dy][dx] = w[qz];
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double u[D1D], v[D1D], w[D1D];
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz) { u[dz] = v[dz] = w[dz] = 0.0; }
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
const int i = qi(qz,dz,Q1D);
const int j = dj(qz,dz,D1D);
const int k = qk(qz,dz,Q1D);
const int l = dl(qz,dz,D1D);
const double s = sign(qz,dz);
u[dz] += QDD0[qz][dy][dx] * Bt[j][i];
v[dz] += QDD1[qz][dy][dx] * Bt[j][i];
w[dz] += QDD2[qz][dy][dx] * Gt[l][k] * s;
}
}
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
y(dx,dy,dz,e) += (u[dz] + v[dz] + w[dz]);
y(dx,dy,dz,e) += (u + v + w);
}
}
}
@@ -1678,11 +1574,9 @@ static void PADiffusionApply(const int dim,
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
}
#endif // MFEM_USE_OCCA
const int ID = (D1D << 4 ) | Q1D;
if (dim == 2)
{
switch (ID)
switch ((D1D << 4 ) | Q1D)
{
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,D,X,Y);
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,D,X,Y);
@@ -1695,13 +1589,11 @@ static void PADiffusionApply(const int dim,
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
}
}
if (dim == 3)
else if (dim == 3)
{
switch (ID)
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,D,X,Y);
case 0x24: return SmemPADiffusionApply3D<2,4>(NE,B,G,D,X,Y);
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,D,X,Y);
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,D,X,Y);
case 0x46: return SmemPADiffusionApply3D<4,6>(NE,B,G,D,X,Y);
@@ -1722,7 +1614,29 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAddMultPA(ceedDataPtr, x, y);
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
+30 -1330
View File
File diff suppressed because it is too large Load Diff
+1 -11
View File
@@ -114,8 +114,6 @@ void PAHdivMassApply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
@@ -240,7 +238,6 @@ void PAHdivMassAssembleDiagonal2D(const int D1D,
Vector &_diag)
{
constexpr static int VDIM = 2;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
@@ -617,8 +614,6 @@ static void PADivDivApply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Bot = Reshape(_Bot.Read(), D1D-1, Q1D);
@@ -982,7 +977,6 @@ static void PADivDivAssembleDiagonal2D(const int D1D,
Vector &_diag)
{
constexpr static int VDIM = 2;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
@@ -1406,8 +1400,6 @@ static void PAHdivL2Apply2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
@@ -1674,8 +1666,6 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
Vector &_y)
{
constexpr static int VDIM = 2;
constexpr static int MAX_D1D = HDIV_MAX_D1D;
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
auto L2Bo = Reshape(_L2Bo.Read(), Q1D, L2D1D);
auto Gct = Reshape(_Gct.Read(), D1D, Q1D);
@@ -1734,7 +1724,7 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
for (int qy = 0; qy < Q1D; ++qy)
{
double aX[MAX_D1D];
double aX[HDIV_MAX_D1D];
int osc = 0;
for (int c = 0; c < VDIM; ++c) // loop over x, y components
+6 -6
View File
@@ -30,7 +30,7 @@ static void EAMassAssemble1D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -52,7 +52,7 @@ static void EAMassAssemble1D(const int NE,
{
val += r_Bi[k1] * r_Bj[k1] * D(k1, e);
}
M(i1, j1, e) += val;
M(i1, j1, e) = val;
}
}
});
@@ -72,7 +72,7 @@ static void EAMassAssemble2D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -114,7 +114,7 @@ static void EAMassAssemble2D(const int NE,
* s_D[k1][k2];
}
}
M(i1, i2, j1, j2, e) += val;
M(i1, i2, j1, j2, e) = val;
}
}
}
@@ -136,7 +136,7 @@ static void EAMassAssemble3D(const int NE,
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
auto B = Reshape(basis.Read(), Q1D, D1D);
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, NE);
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -189,7 +189,7 @@ static void EAMassAssemble3D(const int NE,
}
}
}
M(i1, i2, i3, j1, j2, j3, e) += val;
M(i1, i2, i3, j1, j2, j3, e) = val;
}
}
}
+57 -101
View File
@@ -23,9 +23,8 @@ namespace mfem
// PA Mass Assemble kernel
void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
{
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
@@ -34,7 +33,7 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
@@ -46,57 +45,27 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
const DofToQuad::Mode mode = DofToQuad::TENSOR;
const int flags = GeometricFactors::JACOBIANS |
GeometricFactors::COORDINATES;
#ifdef MFEM_USE_UMPIRE
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
#else
const MemoryType temp_type = Device::GetDeviceMemoryType();
#endif
geom = mesh->GetGeometricFactors(*ir, flags, mode, temp_type);
maps = &el.GetDofToQuad(*ir, mode);
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
Vector *coeff{nullptr};
bool own_coeff{true};
Vector coeff;
if (Q == nullptr)
{
coeff = new Vector;
coeff->SetSize(1);
(*coeff)(0) = 1.0;
coeff.SetSize(1);
coeff(0) = 1.0;
}
else if (ConstantCoefficient* cQ = dynamic_cast<ConstantCoefficient*>(Q))
{
coeff = new Vector;
coeff->SetSize(1);
(*coeff)(0) = cQ->constant;
}
else if (QuadratureCoefficient* cQ = dynamic_cast<QuadratureCoefficient*>(Q))
{
coeff = cQ->Data();
own_coeff = false;
}
else if (QuadratureFunctionCoefficient* cQ =
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
{
const QuadratureFunction &qFun = cQ->GetQuadFunction();
MFEM_VERIFY(qFun.Size() == nq * ne,
"Incompatible QuadratureFunction dimension \n");
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
"IntegrationRule used within integrator and in"
" QuadratureFunction appear to be different");
qFun.Read();
coeff->MakeRef(const_cast<QuadratureFunction &>(qFun),0);
coeff.SetSize(1);
coeff(0) = cQ->constant;
}
else
{
coeff = new Vector;
coeff->SetSize(nq * ne);
auto C = Reshape(coeff->HostWrite(), nq, ne);
coeff.SetSize(nq * ne);
auto C = Reshape(coeff.HostWrite(), nq, ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
@@ -111,11 +80,11 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
{
const int NE = ne;
const int NQ = nq;
const bool const_c = coeff->Size() == 1;
const bool const_c = coeff.Size() == 1;
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto C =
const_c ? Reshape(coeff->Read(), 1,1) : Reshape(coeff->Read(), NQ,NE);
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
MFEM_FORALL(e, NE,
{
@@ -134,43 +103,28 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
if (dim==3)
{
const int NE = ne;
const int Q1D = quad1D;
const bool const_c = coeff->Size() == 1;
const auto W = Reshape(ir->GetWeights().Read(),Q1D,Q1D,Q1D);
const auto J = Reshape(geom->J.Read(), Q1D,Q1D,Q1D,3,3,NE);
const auto C = const_c ?
Reshape(coeff->Read(), 1,1,1,1) :
Reshape(coeff->Read(), Q1D,Q1D,Q1D,NE);
auto V = Reshape(pa_data.Write(), Q1D,Q1D,Q1D,NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
const int NQ = nq;
const bool const_c = coeff.Size() == 1;
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto C =
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
const double J11 = J(qx,qy,qz,0,0,e);
const double J12 = J(qx,qy,qz,0,1,e);
const double J13 = J(qx,qy,qz,0,2,e);
const double J21 = J(qx,qy,qz,1,0,e);
const double J22 = J(qx,qy,qz,1,1,e);
const double J23 = J(qx,qy,qz,1,2,e);
const double J31 = J(qx,qy,qz,2,0,e);
const double J32 = J(qx,qy,qz,2,1,e);
const double J33 = J(qx,qy,qz,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
V(qx,qy,qz,e) = W(qx,qy,qz) * coeff * detJ;
}
}
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0) : C(q,e);
v(q,e) = W[q] * coeff * detJ;
}
});
}
if (own_coeff) { delete coeff; }
}
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
@@ -472,12 +426,8 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
switch ((D1D << 4 ) | Q1D)
{
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
case 0x24: return SmemPAMassAssembleDiagonal3D<2,4>(NE,B,D,Y);
case 0x26: return SmemPAMassAssembleDiagonal3D<2,6>(NE,B,D,Y);
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
case 0x35: return SmemPAMassAssembleDiagonal3D<3,5>(NE,B,D,Y);
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
case 0x48: return SmemPAMassAssembleDiagonal3D<4,8>(NE,B,D,Y);
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
@@ -490,16 +440,8 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAssembleDiagonalPA(ceedDataPtr, diag);
}
else
#endif
{
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
@@ -697,7 +639,6 @@ static void SmemPAMassApply2D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
@@ -961,7 +902,6 @@ static void SmemPAMassApply3D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
@@ -1211,13 +1151,10 @@ static void PAMassApply(const int dim,
case 0x24: return SmemPAMassApply2D<2,4,16>(NE,B,Bt,D,X,Y);
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
case 0x34: return SmemPAMassApply2D<3,4,16>(NE,B,Bt,D,X,Y);
case 0x35: return SmemPAMassApply2D<3,5,16>(NE,B,Bt,D,X,Y);
case 0x36: return SmemPAMassApply2D<3,6,16>(NE,B,Bt,D,X,Y);
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
case 0x46: return SmemPAMassApply2D<4,6,8>(NE,B,Bt,D,X,Y);
case 0x48: return SmemPAMassApply2D<4,8,4>(NE,B,Bt,D,X,Y);
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
case 0x57: return SmemPAMassApply2D<5,7,8>(NE,B,Bt,D,X,Y);
case 0x58: return SmemPAMassApply2D<5,8,2>(NE,B,Bt,D,X,Y);
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
@@ -1225,7 +1162,6 @@ static void PAMassApply(const int dim,
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
}
mfem::out << "Unknown 2D kernel 0x" << std::hex << id << std::endl;
}
else if (dim == 3)
{
@@ -1234,9 +1170,7 @@ static void PAMassApply(const int dim,
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
case 0x24: return SmemPAMassApply3D<2,4>(NE,B,Bt,D,X,Y);
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
case 0x35: return SmemPAMassApply3D<3,5>(NE,B,Bt,D,X,Y);
case 0x36: return SmemPAMassApply3D<3,6>(NE,B,Bt,D,X,Y);
case 0x37: return SmemPAMassApply3D<3,7>(NE,B,Bt,D,X,Y);
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
case 0x46: return SmemPAMassApply3D<4,6>(NE,B,Bt,D,X,Y);
case 0x48: return SmemPAMassApply3D<4,8>(NE,B,Bt,D,X,Y);
@@ -1248,8 +1182,8 @@ static void PAMassApply(const int dim,
case 0x9A: return SmemPAMassApply3D<9,10>(NE,B,Bt,D,X,Y);
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
}
mfem::out << "Unknown 3D kernel 0x" << std::hex << id << std::endl;
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Unknown kernel.");
}
@@ -1258,7 +1192,29 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAddMultPA(ceedDataPtr, x, y);
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
+1 -1
View File
@@ -25,7 +25,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
if (ne == 0) { return; }
const int dofs = fes.GetFE(0)->GetDof();
auto A = Reshape(ea_data_tmp.Write(), dofs, dofs, ne);
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
auto AT = Reshape(ea_data.Write(), dofs, dofs, ne);
MFEM_FORALL(e, ne,
{
for (int i = 0; i < dofs; i++)
+7 -30
View File
@@ -9,14 +9,12 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../general/forall.hpp"
#include "bilininteg.hpp"
namespace mfem
{
void PAHcurlSetup2D(const int Q1D,
const int coeffDim,
const int NE,
const Array<double> &w,
const Vector &j,
@@ -24,7 +22,6 @@ void PAHcurlSetup2D(const int Q1D,
Vector &op);
void PAHcurlSetup3D(const int Q1D,
const int coeffDim,
const int NE,
const Array<double> &w,
const Vector &j,
@@ -175,36 +172,16 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
const int coeffDim = VQ ? VQ->GetVDim() : 1;
Vector coeff(coeffDim * ne * nq);
Vector coeff(ne * nq);
coeff = 1.0;
auto coeffh = Reshape(coeff.HostWrite(), coeffDim, nq, ne);
if (Q || VQ)
if (Q)
{
Vector D(VQ ? coeffDim : 0);
if (VQ)
{
MFEM_VERIFY(coeffDim == dim, "");
}
for (int e=0; e<ne; ++e)
{
ElementTransformation *tr = mesh->GetElementTransformation(e);
for (int p=0; p<nq; ++p)
{
if (VQ)
{
VQ->Eval(D, *tr, ir->IntPoint(p));
for (int i=0; i<coeffDim; ++i)
{
coeffh(i, p, e) = D[i];
}
}
else
{
coeffh(0, p, e) = Q->Eval(*tr, ir->IntPoint(p));
}
coeff[p + (e * nq)] = Q->Eval(*tr, ir->IntPoint(p));
}
}
}
@@ -213,12 +190,12 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
{
PAHcurlSetup3D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
{
PAHcurlSetup2D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (el->GetDerivType() == mfem::FiniteElement::DIV && dim == 3)
@@ -371,12 +348,12 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
// Use the same setup functions as VectorFEMassIntegrator.
if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
{
PAHcurlSetup3D(quad1D, 1, ne, ir->GetWeights(), geom->J,
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
{
PAHcurlSetup2D(quad1D, 1, ne, ir->GetWeights(), geom->J,
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else
-8
View File
@@ -12,7 +12,6 @@
// Implementation of Coefficient class
#include "fem.hpp"
#include "../linalg/dtensor.hpp"
#include <cmath>
#include <limits>
@@ -22,13 +21,6 @@ namespace mfem
using namespace std;
double QuadratureCoefficient::Eval(ElementTransformation & T,
const IntegrationPoint & ip)
{
auto coeff = mfem::Reshape(qData->HostRead(), nip, NE);
return coeff(ip.index, T.ElementNo);
}
double PWConstCoefficient::Eval(ElementTransformation & T,
const IntegrationPoint & ip)
{
+1 -31
View File
@@ -30,10 +30,7 @@ class ParMesh;
/** @brief Base class Coefficients that optionally depend on space and time.
These are used by the BilinearFormIntegrator, LinearFormIntegrator, and
NonlinearFormIntegrator classes to represent the physical coefficients in
the PDEs that are being discretized. This class can also be used in a more
general way to represent functions that don't necessarily belong to a FE
space, e.g., to project onto GridFunctions to use as initial conditions,
exact solutions, etc. See, e.g., ex4 or ex22 for these uses. */
the PDEs that are being discretized. */
class Coefficient
{
protected:
@@ -87,33 +84,6 @@ public:
{ return (constant); }
};
/// class for quadrature coefficient
class QuadratureCoefficient : public Coefficient
{
private:
const int nip;
const int NE;
public:
Vector *qData{nullptr};
//Set external data
QuadratureCoefficient(Vector *Data, int in_nip, int in_NE)
: qData(Data), nip(in_nip), NE(in_NE)
{ }
virtual double Eval(ElementTransformation &T,
const IntegrationPoint &ip);
Vector *Data()
{
return qData;
}
};
/// class for piecewise constant coefficient
/** @brief A piecewise constant coefficient with the constants keyed
off the element attribute numbers. */
class PWConstCoefficient : public Coefficient
+53 -135
View File
@@ -342,10 +342,11 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
int ci)
{
FiniteElementSpace * fes = blfr->FESpace();
int vsize = fes->GetVSize();
// Allocate temporary vectors
Vector b_0(vsize); b_0 = 0.0;
Vector b_0(vsize); b_0 = 0.0;
// Extract the real and imaginary parts of the input vectors
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
@@ -359,7 +360,8 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
if (conv == ComplexOperator::BLOCK_SYMMETRIC) { b_i *= -1.0; }
int tvsize = fes->GetTrueVSize();
OperatorHandle A_r, A_i;
SparseMatrix * A_r = nullptr;
SparseMatrix * A_i = nullptr;
X.SetSize(2 * tvsize);
B.SetSize(2 * tvsize);
@@ -372,39 +374,42 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
if (RealInteg())
{
A_r = new SparseMatrix;
blfr->SetDiagonalPolicy(diag_policy);
b_0 = b_r;
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, A_r, X_0, B_0, ci);
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_r, X_0, B_0, ci);
X_r = X_0; B_r = B_0;
b_0 = b_i;
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, A_r, X_0, B_0, ci);
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_r, X_0, B_0, ci);
X_i = X_0; B_i = B_0;
if (ImagInteg())
{
A_i = new SparseMatrix;
blfi->SetDiagonalPolicy(mfem::Matrix::DiagonalPolicy::DIAG_ZERO);
b_0 = 0.0;
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, false);
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, false);
B_r -= B_0;
b_0 = 0.0;
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, false);
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, false);
B_i += B_0;
}
}
else if (ImagInteg())
{
A_i = new SparseMatrix;
blfi->SetDiagonalPolicy(diag_policy);
b_0 = b_i;
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, ci);
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, ci);
X_r = X_0; B_i = B_0;
b_0 = b_r; b_0 *= -1.0;
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, ci);
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, ci);
X_i = X_0; B_r = B_0; B_r *= -1.0;
}
else
@@ -412,55 +417,16 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
}
if (RealInteg() && ImagInteg())
{
// Modify RHS and offdiagonal blocks (imaginary parts of the matrix) to
// conform with standard essential BC treatment
if (A_i.Is<ConstrainedOperator>())
{
int n = ess_tdof_list.Size();
for (int k = 0; k < n; k++)
{
int j = ess_tdof_list[k];
B_r(j) = X_r(j);
B_i(j) = X_i(j);
}
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
}
}
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
{
B_i *= -1.0;
b_i *= -1.0;
}
// A = A_r + i A_i
A.Clear();
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
A_i.Type() == Operator::MFEM_SPARSEMAT )
{
ComplexSparseMatrix * A_sp =
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
A_i.As<SparseMatrix>(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexSparseMatrix>(A_sp, true);
}
else
{
ComplexOperator * A_op =
new ComplexOperator(A_r.Ptr(),
A_i.Ptr(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexOperator>(A_op, true);
}
A_r.SetOperatorOwner(false);
A_i.SetOperatorOwner(false);
ComplexSparseMatrix * A_sp;
A_sp = new ComplexSparseMatrix(A_r, A_i, true, true, conv);
A.Reset<ComplexSparseMatrix>(A_sp, true);
}
void
@@ -468,60 +434,31 @@ SesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
OperatorHandle &A)
{
OperatorHandle A_r, A_i;
SparseMatrix * A_r = nullptr;
SparseMatrix * A_i = nullptr;
if (RealInteg())
{
A_r = new SparseMatrix;
blfr->SetDiagonalPolicy(diag_policy);
blfr->FormSystemMatrix(ess_tdof_list, A_r);
blfr->FormSystemMatrix(ess_tdof_list, *A_r);
}
if (ImagInteg())
{
blfi->SetDiagonalPolicy(RealInteg() ?
mfem::Matrix::DiagonalPolicy::DIAG_ZERO :
diag_policy);
blfi->FormSystemMatrix(ess_tdof_list, A_i);
A_i = new SparseMatrix;
blfr->SetDiagonalPolicy(diag_policy);
blfi->FormSystemMatrix(ess_tdof_list, *A_i);
}
if (!RealInteg() && !ImagInteg())
{
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
}
if (RealInteg() && ImagInteg())
{
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
// with standard essential BC treatment
if (A_i.Is<ConstrainedOperator>())
{
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
}
}
// A = A_r + i A_i
A.Clear();
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
A_i.Type() == Operator::MFEM_SPARSEMAT )
{
ComplexSparseMatrix * A_sp =
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
A_i.As<SparseMatrix>(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexSparseMatrix>(A_sp, true);
}
else
{
ComplexOperator * A_op =
new ComplexOperator(A_r.Ptr(),
A_i.Ptr(),
A_r.OwnsOperator(),
A_i.OwnsOperator(),
conv);
A.Reset<ComplexOperator>(A_op, true);
}
A_r.SetOperatorOwner(false);
A_i.SetOperatorOwner(false);
ComplexSparseMatrix * A_sp =
new ComplexSparseMatrix(A_r, A_i, true, true, conv);
A.Reset<ComplexSparseMatrix>(A_sp, true);
}
void
@@ -709,7 +646,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
tdof_offsets = new HYPRE_Int[n+1];
for (int i = 0; i <= n; i++)
for (int i=0; i<=n; i++)
{
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
}
@@ -717,8 +654,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ParLinearForm *plf_r,
ParLinearForm *plf_i,
ParLinearForm *plf_r, ParLinearForm *plf_i,
ComplexOperator::Convention
convention)
: Vector(2*(pfes->GetVSize())),
@@ -734,7 +670,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
tdof_offsets = new HYPRE_Int[n+1];
for (int i = 0; i <= n; i++)
for (int i=0; i<=n; i++)
{
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
}
@@ -881,8 +817,7 @@ ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
{}
ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
ParBilinearForm *pbfr,
ParBilinearForm *pbfi,
ParBilinearForm *pbfr, ParBilinearForm *pbfi,
ComplexOperator::Convention convention)
: conv(convention),
pblfr(new ParBilinearForm(pf,pbfr)),
@@ -978,10 +913,9 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
int vsize = pfes->GetVSize();
// Allocate temporary vectors
Vector b_0(vsize); b_0 = 0.0;
Vector b_0(vsize); b_0 = 0.0;
// Extract the real and imaginary parts of the input vectors
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
Vector x_r(x.GetData(), vsize);
Vector x_i(&(x.GetData())[vsize], vsize);
@@ -1040,34 +974,25 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
}
// Modify RHS and offdiagonal blocks (Imaginary parts of the matrix) to
// conform with standard essential BC treatment i.e. zero out rows and
// columns and place ones on the diagonal.
if (RealInteg() && ImagInteg())
{
int n = ess_tdof_list.Size();
// Modify RHS to conform with standard essential BC treatment
for (int k = 0; k < n; k++)
{
int j=ess_tdof_list[k];
B_r(j) = X_r(j);
B_i(j) = X_i(j);
}
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
// with standard essential BC treatment
if ( A_i.Type() == Operator::Hypre_ParCSR )
{
HypreParMatrix * Ah;
A_i.Get(Ah);
hypre_ParCSRMatrix *Aih = *Ah;
for (int k = 0; k < n; k++)
HypreParMatrix * Ah; A_i.Get(Ah);
int n = ess_tdof_list.Size();
hypre_ParCSRMatrix * Aih =
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
for (int k=0; k<n; k++)
{
int j = ess_tdof_list[k];
int j=ess_tdof_list[k];
Aih->diag->data[Aih->diag->i[j]] = 0.0;
B_r(j) = X_r(j);
B_i(j) = X_i(j);
}
}
else
{
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
}
}
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
@@ -1075,7 +1000,6 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
B_i *= -1.0;
b_i *= -1.0;
}
// A = A_r + i A_i
A.Clear();
if ( A_r.Type() == Operator::Hypre_ParCSR ||
@@ -1099,8 +1023,6 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
conv);
A.Reset<ComplexOperator>(A_op, true);
}
A_r.SetOperatorOwner(false);
A_i.SetOperatorOwner(false);
}
void
@@ -1121,27 +1043,25 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
}
// Modify offdiagonal blocks (Imaginary parts of the matrix) to conform with
// standard essential BC treatment i.e. zero out rows and columns and place
// ones on the diagonal.
if (RealInteg() && ImagInteg())
{
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
// with standard essential BC treatment
if ( A_i.Type() == Operator::Hypre_ParCSR )
{
int n = ess_tdof_list.Size();
HypreParMatrix * Ah;
A_i.Get(Ah);
hypre_ParCSRMatrix * Aih = *Ah;
for (int k = 0; k < n; k++)
int j;
HypreParMatrix * Ah; A_i.Get(Ah);
hypre_ParCSRMatrix * Aih =
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
for (int k=0; k<n; k++)
{
int j = ess_tdof_list[k];
j=ess_tdof_list[k];
Aih->diag->data[Aih->diag->i[j]] = 0.0;
}
}
else
{
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
}
}
// A = A_r + i A_i
@@ -1167,8 +1087,6 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
conv);
A.Reset<ComplexOperator>(A_op, true);
}
A_r.SetOperatorOwner(false);
A_i.SetOperatorOwner(false);
}
void
+1 -31
View File
@@ -219,21 +219,6 @@ public:
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::FULL (default)
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
blfr->SetAssemblyLevel(assembly_level);
blfi->SetAssemblyLevel(assembly_level);
}
BilinearForm & real() { return *blfr; }
BilinearForm & imag() { return *blfi; }
const BilinearForm & real() const { return *blfr; }
@@ -493,7 +478,7 @@ public:
/** Class for a parallel sesquilinear form
A sesquilinear form is a generalization of a bilinear form to complex-valued
fields. Sesquilinear forms are linear in the second argument but the
fields. Sesquilinear forms are linear in the second argument but but the
first argument involves a complex conjugate in the sense that:
a(alpha u, beta v) = conj(alpha) beta a(u, v)
@@ -539,21 +524,6 @@ public:
void SetConvention(const ComplexOperator::Convention &
convention) { conv = convention; }
/// Set the desired assembly level.
/** Valid choices are:
- AssemblyLevel::FULL (default)
- AssemblyLevel::PARTIAL
- AssemblyLevel::ELEMENT
- AssemblyLevel::NONE
This method must be called before assembly. */
void SetAssemblyLevel(AssemblyLevel assembly_level)
{
pblfr->SetAssemblyLevel(assembly_level);
pblfi->SetAssemblyLevel(assembly_level);
}
ParBilinearForm & real() { return *pblfr; }
ParBilinearForm & imag() { return *pblfi; }
const ParBilinearForm & real() const { return *pblfr; }
+8 -40
View File
@@ -415,6 +415,9 @@ void VisItDataCollection::SetMesh(MPI_Comm comm, Mesh *new_mesh)
void VisItDataCollection::RegisterField(const std::string& name,
GridFunction *gf)
{
DataCollection::RegisterField(name, gf);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim());
int LOD = 1;
if (gf->FESpace()->GetNURBSext())
{
@@ -428,27 +431,6 @@ void VisItDataCollection::RegisterField(const std::string& name,
}
}
DataCollection::RegisterField(name, gf);
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD);
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
void VisItDataCollection::RegisterQField(const std::string& name,
QuadratureFunction *qf)
{
int LOD = -1;
Mesh *mesh = qf->GetSpace()->GetMesh();
for (int e=0; e<qf->GetSpace()->GetNE(); e++)
{
int locLOD = GlobGeometryRefiner.GetRefinementLevelFromElems(
mesh->GetElementBaseGeometry(e),
qf->GetElementIntRule(e).GetNPoints());
LOD = std::max(LOD,locLOD);
}
DataCollection::RegisterQField(name, qf);
field_info_map[name] = VisItFieldInfo("elements", 1, LOD);
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
}
@@ -616,28 +598,14 @@ void VisItDataCollection::LoadFields()
// TODO: 1) load parallel GridFunction on one processor
if (serial)
{
if ((it->second).association == "nodes")
{
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
}
else if ((it->second).association == "elements")
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
}
else
{
#ifdef MFEM_USE_MPI
if ((it->second).association == "nodes")
{
field_map.Register(
it->first,
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
}
else if ((it->second).association == "elements")
{
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
}
field_map.Register(
it->first,
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
#else
error = READ_ERROR;
MFEM_WARNING("Reading parallel format in serial is not supported");
@@ -672,7 +640,7 @@ std::string VisItDataCollection::GetVisItRootString()
{
ftags["assoc"] = picojson::value((it->second).association);
ftags["comps"] = picojson::value(to_string((it->second).num_components));
ftags["lod"] = picojson::value(to_string((it->second).lod));
ftags["lod"] = picojson::value(to_string(visit_levels_of_detail));
field["path"] = picojson::value(path_str + it->first + file_ext_format);
field["tags"] = picojson::value(ftags);
fields[it->first] = picojson::value(field);
+3 -10
View File
@@ -391,10 +391,9 @@ class VisItFieldInfo
public:
std::string association;
int num_components;
int lod;
VisItFieldInfo() { association = ""; num_components = 0; lod = 1;}
VisItFieldInfo(std::string _association, int _num_components, int _lod = 1)
{ association = _association; num_components = _num_components; lod =_lod;}
VisItFieldInfo() { association = ""; num_components = 0; }
VisItFieldInfo(std::string _association, int _num_components)
{ association = _association; num_components = _num_components; }
};
/// Data collection with VisIt I/O routines
@@ -446,12 +445,6 @@ public:
/// Add a grid function to the collection and update the root file
virtual void RegisterField(const std::string& field_name, GridFunction *gf);
/// Add a quadrature function to the collection and update the root file.
/** Visualization of quadrature function is not supported in VisIt(3.12).
A patch has been sent to VisIt developers in June 2020. */
virtual void RegisterQField(const std::string& q_field_name,
QuadratureFunction *qf);
/// Set VisIt parameter: default levels of detail for the MultiresControl
void SetLevelsOfDetail(int levels_of_detail);
+1 -4
View File
@@ -37,8 +37,7 @@ public:
ClosedUniform = 4, ///< Nodes: x_i = i/(n-1), i=0,...,n-1
OpenHalfUniform = 5, ///< Nodes: x_i = (i+1/2)/n, i=0,...,n-1
Serendipity = 6, ///< Serendipity basis (squares / cubes)
ClosedGL = 7, ///< Closed GaussLegendre
NumBasisTypes = 8 /**< Keep track of maximum types to prevent
NumBasisTypes = 7 /**< Keep track of maximum types to prevent
hard-coding */
};
/** @brief If the input does not represents a valid BasisType, abort with an
@@ -70,7 +69,6 @@ public:
case ClosedUniform: return Quadrature1D::ClosedUniform;
case OpenHalfUniform: return Quadrature1D::OpenHalfUniform;
case Serendipity: return Quadrature1D::GaussLobatto;
case ClosedGL: return Quadrature1D::ClosedGL;
}
return Quadrature1D::Invalid;
}
@@ -84,7 +82,6 @@ public:
case Quadrature1D::OpenUniform: return OpenUniform;
case Quadrature1D::ClosedUniform: return ClosedUniform;
case Quadrature1D::OpenHalfUniform: return OpenHalfUniform;
case Quadrature1D::ClosedGL: return ClosedGL;
}
return Invalid;
}
+6 -6
View File
@@ -944,7 +944,7 @@ const Operator *FiniteElementSpace::GetFaceRestriction(
}
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const IntegrationRule &ir, const DofToQuad::Mode mode) const
const IntegrationRule &ir) const
{
for (int i = 0; i < E2Q_array.Size(); i++)
{
@@ -952,13 +952,13 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
if (qi->IntRule == &ir) { return qi; }
}
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir, mode);
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir);
E2Q_array.Append(qi);
return qi;
}
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const QuadratureSpace &qs, const DofToQuad::Mode mode) const
const QuadratureSpace &qs) const
{
for (int i = 0; i < E2Q_array.Size(); i++)
{
@@ -966,7 +966,7 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
if (qi->qspace == &qs) { return qi; }
}
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs, mode);
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs);
E2Q_array.Append(qi);
return qi;
}
@@ -983,8 +983,8 @@ const FaceQuadratureInterpolator
if (qi->IntRule == &ir) { return qi; }
}
FaceQuadratureInterpolator *qi =
new FaceQuadratureInterpolator(*this, ir, type);
FaceQuadratureInterpolator *qi = new FaceQuadratureInterpolator(*this, ir,
type);
E2IFQ_array.Append(qi);
return qi;
}
+2 -8
View File
@@ -367,7 +367,7 @@ public:
All elements will use the same IntegrationRule, @a ir as the target
quadrature points. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const IntegrationRule &ir, const DofToQuad::Mode = DofToQuad::FULL) const;
const IntegrationRule &ir) const;
/** @brief Return a QuadratureInterpolator that interpolates E-vectors to
quadrature point values and/or derivatives (Q-vectors). */
@@ -378,7 +378,7 @@ public:
The target quadrature points in the elements are described by the given
QuadratureSpace, @a qs. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const QuadratureSpace &qs, const DofToQuad::Mode = DofToQuad::FULL) const;
const QuadratureSpace &qs) const;
/** @brief Return a FaceQuadratureInterpolator that interpolates E-vectors to
quadrature point values and/or derivatives (Q-vectors). */
@@ -756,12 +756,6 @@ public:
/// Return the total number of quadrature points.
int GetSize() const { return size; }
/// Returns the mesh
inline Mesh *GetMesh() const { return mesh; }
/// Returns number of elements in the mesh.
inline int GetNE() const { return mesh->GetNE(); }
/// Get the IntegrationRule associated with mesh element @a idx.
const IntegrationRule &GetElementIntRule(int idx) const
{ return *int_rule[mesh->GetElementBaseGeometry(idx)]; }
+33 -141
View File
@@ -1344,14 +1344,15 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, Times-1);
if (ir) { return ir; }
ir = new IntegrationRule(Times-1);
for (int i = 1; i < Times; i++)
if (ir == NULL)
{
IntegrationPoint &ip = ir->IntPoint(i-1);
ip.x = double(i) / Times;
ip.y = ip.z = 0.0;
ir = new IntegrationRule(Times-1);
for (int i = 1; i < Times; i++)
{
IntegrationPoint &ip = ir->IntPoint(i-1);
ip.x = double(i) / Times;
ip.y = ip.z = 0.0;
}
}
}
break;
@@ -1363,17 +1364,18 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, ((Times-1)*(Times-2))/2);
if (ir) { return ir; }
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
for (int k = 0, j = 1; j < Times-1; j++)
for (int i = 1; i < Times-j; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
if (ir == NULL)
{
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
for (int k = 0, j = 1; j < Times-1; j++)
for (int i = 1; i < Times-j; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
}
break;
@@ -1384,17 +1386,18 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
return NULL;
}
ir = FindInIntPts(Geom, (Times-1)*(Times-1));
if (ir) { return ir; }
ir = new IntegrationRule((Times-1)*(Times-1));
for (int k = 0, j = 1; j < Times; j++)
for (int i = 1; i < Times; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
if (ir == NULL)
{
ir = new IntegrationRule((Times-1)*(Times-1));
for (int k = 0, j = 1; j < Times; j++)
for (int i = 1; i < Times; i++, k++)
{
IntegrationPoint &ip = ir->IntPoint(k);
ip.x = double(i) / Times;
ip.y = double(j) / Times;
ip.z = 0.0;
}
}
}
break;
@@ -1402,121 +1405,10 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
mfem_error("GeometryRefiner::RefineInterior(...)");
}
MFEM_ASSERT(ir != NULL, "Failed to construct the refined IntegrationRule.");
IntPts[Geom].Append(ir);
if (ir) { IntPts[Geom].Append(ir); }
return ir;
}
int GeometryRefiner::GetRefinementLevelFromPoints(Geometry::Type geom, int Npts)
{
switch (geom)
{
case Geometry::POINT:
{
return -1;
}
case Geometry::SEGMENT:
{
return Npts -1;
}
case Geometry::TRIANGLE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+2)/2;
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::SQUARE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1);
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::CUBE:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1)*(n+1);
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::TETRAHEDRON:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+3)*(n+2)*(n+1)/6;
if (np == Npts) { return n; }
}
return -1;
}
case Geometry::PRISM:
{
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
{
np = (n+1)*(n+1)*(n+2)/2;
if (np == Npts) { return n; }
}
return -1;
}
default:
{
mfem_error("Non existing Geometry.");
}
}
return -1;
}
int GeometryRefiner::GetRefinementLevelFromElems(Geometry::Type geom, int Nels)
{
switch (geom)
{
case Geometry::POINT:
{
return -1;
}
case Geometry::SEGMENT:
{
return Nels;
}
case Geometry::TRIANGLE:
case Geometry::SQUARE:
{
for (int n = 0; (n < 15) && (n*n < Nels+1) ; n++)
{
if (n*n == Nels) { return n-1; }
}
return -1;
}
case Geometry::CUBE:
case Geometry::TETRAHEDRON:
case Geometry::PRISM:
{
for (int n = 0; (n < 15) && (n*n*n < Nels+1) ; n++)
{
if (n*n*n == Nels) { return n-1; }
}
return -1;
}
default:
{
mfem_error("Non existing Geometry.");
}
}
return -1;
}
GeometryRefiner GlobGeometryRefiner;
}
-6
View File
@@ -273,12 +273,6 @@ public:
/// @note This method always uses Quadrature1D::OpenUniform points.
const IntegrationRule *RefineInterior(Geometry::Type Geom, int Times);
/// Get the Refinement level based on number of points
virtual int GetRefinementLevelFromPoints(Geometry::Type Geom, int Npts);
/// Get the Refinement level based on number of elements
virtual int GetRefinementLevelFromElems(Geometry::Type geom, int Npts);
~GeometryRefiner();
};
+2 -2
View File
@@ -2984,7 +2984,7 @@ double GridFunction::ComputeLpError(const double p, Coefficient &exsol,
}
else
{
int intorder = 2*fe->GetOrder() + 3; // <----------
int intorder = 2*fe->GetOrder() + 1; // <----------
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
GetValues(i, *ir, vals);
@@ -3117,7 +3117,7 @@ double GridFunction::ComputeLpError(const double p, VectorCoefficient &exsol,
}
else
{
int intorder = 2*fe->GetOrder() + 3; // <----------
int intorder = 2*fe->GetOrder() + 1; // <----------
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
T = fes->GetElementTransformation(i);
+2 -2
View File
@@ -105,7 +105,7 @@ public:
have the same size.
@note Defining this method overwrites the implicitly defined copy
assignment operator. */
assignemnt operator. */
GridFunction &operator=(const GridFunction &rhs)
{ return operator=((const Vector &)rhs); }
@@ -714,7 +714,7 @@ public:
the same size.
@note Defining this method overwrites the implicitly defined copy
assignment operator. */
assignemnt operator. */
QuadratureFunction &operator=(const QuadratureFunction &v);
/// Get the IntegrationRule associated with mesh element @a idx.
-25
View File
@@ -618,26 +618,6 @@ void QuadratureFunctions1D::OpenHalfUniform(const int np, IntegrationRule* ir)
CalculateUniformWeights(ir, Quadrature1D::OpenHalfUniform);
}
void QuadratureFunctions1D::ClosedGL(const int np, IntegrationRule* ir)
{
ir->SetSize(np);
ir->IntPoint(0).x = 0.0;
ir->IntPoint(np-1).x = 1.0;
if ( np > 2 )
{
IntegrationRule gl_ir;
GaussLegendre(np-1, &gl_ir);
for (int i = 1; i < np-1; ++i)
{
ir->IntPoint(i).x = (gl_ir.IntPoint(i-1).x + gl_ir.IntPoint(i).x)/2;
}
}
CalculateUniformWeights(ir, Quadrature1D::ClosedGL);
}
void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
const int type)
{
@@ -670,11 +650,6 @@ void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
OpenHalfUniform(np, &ir);
break;
}
case Quadrature1D::ClosedGL:
{
ClosedGL(np, &ir);
break;
}
default:
{
MFEM_ABORT("Asking for an unknown type of 1D Quadrature points, "
+1 -3
View File
@@ -272,7 +272,6 @@ public:
void OpenUniform(const int np, IntegrationRule *ir);
void ClosedUniform(const int np, IntegrationRule *ir);
void OpenHalfUniform(const int np, IntegrationRule *ir);
void ClosedGL(const int np, IntegrationRule *ir);
///@}
/// A helper function that will play nice with Poly_1D::OpenPoints and
@@ -294,8 +293,7 @@ public:
GaussLobatto = 1,
OpenUniform = 2, ///< aka open Newton-Cotes
ClosedUniform = 3, ///< aka closed Newton-Cotes
OpenHalfUniform = 4, ///< aka "open half" Newton-Cotes
ClosedGL = 5 ///< aka closed Gauss Legendre
OpenHalfUniform = 4 ///< aka "open half" Newton-Cotes
};
/** @brief If the Quadrature1D type is not closed return Invalid; otherwise
return type. */
-1465
View File
File diff suppressed because it is too large Load Diff
-53
View File
@@ -415,59 +415,6 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
void CeedAddMultPA(const CeedData *ceedDataPtr,
const Vector &x,
Vector &y)
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorTakeArray(ceedDataPtr->u, mem, const_cast<CeedScalar**>(&x_ptr));
CeedVectorTakeArray(ceedDataPtr->v, mem, &y_ptr);
}
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
Vector &diag)
{
CeedScalar *d_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
d_ptr = diag.ReadWrite();
}
else
{
d_ptr = diag.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, d_ptr);
CeedOperatorLinearAssembleAddDiagonal(ceedDataPtr->oper, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorTakeArray(ceedDataPtr->v, mem, &d_ptr);
}
} // namespace mfem
#endif // MFEM_USE_CEED
-10
View File
@@ -16,7 +16,6 @@
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../linalg/vector.hpp"
#include <ceed.h>
namespace mfem
@@ -145,15 +144,6 @@ const std::string &GetCeedPath();
void CeedPAAssemble(const CeedPAOperator& op,
CeedData& ceedData);
/** @brief Function that applies a libCEED PA operator. */
void CeedAddMultPA(const CeedData *ceedDataPtr,
const Vector &x,
Vector &y);
/** @brief Function that assembles a libCEED PA operator diagonal. */
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
Vector &diag);
/** @brief Function that determines if a CEED kernel should be used, based on
the current mfem::Device configuration. */
inline bool DeviceCanUseCeed()
+1 -2
View File
@@ -199,8 +199,7 @@ void LinearForm::Assemble()
void LinearForm::Update(FiniteElementSpace *f, Vector &v, int v_offset)
{
fes = f;
NewMemoryAndSize(Memory<double>(v.GetMemory(), v_offset, f->GetVSize()),
f->GetVSize(), false);
NewDataAndSize((double *)v + v_offset, fes->GetVSize());
ResetDeltaLocations();
}
+4 -11
View File
@@ -135,18 +135,11 @@ void Multigrid::SetOperator(const Operator& op)
MFEM_ABORT("SetOperator not supported in Multigrid");
}
void Multigrid::SmoothingStep(int level, bool transpose) const
void Multigrid::SmoothingStep(int level) const
{
GetOperatorAtLevel(level)->Mult(*Y[level], *R[level]); // r = A x
subtract(*X[level], *R[level], *R[level]); // r = b - A x
if (transpose)
{
GetSmootherAtLevel(level)->MultTranspose(*R[level], *Z[level]); // z = S r
}
else
{
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
}
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
add(*Y[level], 1.0, *Z[level], *Y[level]); // x = x + S (b - A x)
}
@@ -160,7 +153,7 @@ void Multigrid::Cycle(int level) const
for (int i = 0; i < preSmoothingSteps; i++)
{
SmoothingStep(level, false);
SmoothingStep(level);
}
// Compute residual
@@ -194,7 +187,7 @@ void Multigrid::Cycle(int level) const
// Post-smooth
for (int i = 0; i < postSmoothingSteps; i++)
{
SmoothingStep(level, true);
SmoothingStep(level);
}
}
+1 -1
View File
@@ -108,7 +108,7 @@ public:
private:
/// Application of a smoothing step at particular level
void SmoothingStep(int level, bool transpose) const;
void SmoothingStep(int level) const;
/// Application of a cycle at particular level
void Cycle(int level) const;
+14 -63
View File
@@ -10,7 +10,6 @@
// CONTRIBUTING.md for details.
#include "fem.hpp"
#include "../general/forall.hpp"
namespace mfem
{
@@ -28,7 +27,7 @@ void NonlinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
// This is the default behavior.
break;
case AssemblyLevel::PARTIAL:
ext = new PANonlinearForm(this);
ext = new PANonlinearFormExtension(this);
break;
default:
mfem_error("Unknown assembly level for this form.");
@@ -81,13 +80,6 @@ void NonlinearForm::SetEssentialVDofs(const Array<int> &ess_vdofs_list)
double NonlinearForm::GetGridFunctionEnergy(const Vector &x) const
{
if (ext)
{
MFEM_VERIFY(!fnfi.Size(), "Interior faces terms not yet implemented!");
MFEM_VERIFY(!bfnfi.Size(), "Boundary face terms not yet implemented!");
return ext->GetGridFunctionEnergy(x);
}
Array<int> vdofs;
Vector el_x;
const FiniteElement *fe;
@@ -146,14 +138,6 @@ void NonlinearForm::Mult(const Vector &x, Vector &y) const
if (ext)
{
ext->Mult(px, py);
if (Serial())
{
if (cP) { cP->MultTranspose(py, y); }
const int N = ess_tdof_list.Size();
const auto tdof = ess_tdof_list.Read();
auto Y = y.ReadWrite();
MFEM_FORALL(i, N, Y[tdof[i]] = 0.0; );
}
return;
}
@@ -280,16 +264,7 @@ Operator &NonlinearForm::GetGradient(const Vector &x) const
{
if (ext)
{
Operator &grad = ext->GetGradient(Prolongate(x));
hGrad.Reset(&grad, false);
if (Serial())
{
Operator *Gop;
if (cP) { hGrad.Reset(new RAPOperator(*cP, grad, *cP)); }
hGrad.Ptr()->Operator::FormSystemOperator(ess_tdof_list, Gop);
hGrad.Reset(Gop);
}
return *hGrad.Ptr();
MFEM_ABORT("Not yet implemented!");
}
const int skip_zeros = 0;
@@ -451,31 +426,7 @@ void NonlinearForm::Update()
void NonlinearForm::Setup()
{
if (ext) { return ext->Setup(); }
}
void NonlinearForm::AssembleGradientDiagonal(Vector &diag) const
{
if (ext)
{
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
"Vector for holding diagonal has wrong size!");
const Operator *P = fes->GetProlongationMatrix();
if (!IsIdentityProlongation(P))
{
Vector local_diag(P->Height());
ext->AssembleGradientDiagonal(local_diag);
P->MultTranspose(local_diag, diag);
}
else
{
ext->AssembleGradientDiagonal(diag);
}
}
else
{
MFEM_ABORT("Not implemented. Can be obtained through GetGradient().");
}
if (ext) { return ext->AssemblePA(); }
}
NonlinearForm::~NonlinearForm()
@@ -982,17 +933,6 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
}
}
if (!Grads(0,0)->Finalized())
{
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
{
Grads(i,j)->Finalize(skip_zeros);
}
}
}
for (int s=0; s<fes.Size(); ++s)
{
for (int i = 0; i < ess_vdofs[s]->Size(); ++i)
@@ -1012,6 +952,17 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
}
}
if (!Grads(0,0)->Finalized())
{
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
{
Grads(i,j)->Finalize(skip_zeros);
}
}
}
for (int i=0; i<fes.Size(); ++i)
{
for (int j=0; j<fes.Size(); ++j)
-10
View File
@@ -45,7 +45,6 @@ protected:
Array<Array<int>*> bfnfi_marker; // not owned
mutable SparseMatrix *Grad, *cGrad; // owned
mutable OperatorHandle hGrad;
/// A list of all essential true dofs
Array<int> ess_tdof_list;
@@ -166,15 +165,6 @@ public:
/// Setup the NonlinearForm
virtual void Setup();
/** @brief Assemble the diagonal of the gradient into diag
For adaptively refined meshes, this returns P^T d_e, where d_e is the
locally assembled diagonal on each element and P^T is the transpose of
the conforming prolongation. In general this is not the correct diagonal
for an AMR mesh. */
void AssembleGradientDiagonal(Vector &diag) const;
/// Get the finite element space prolongation matrix
virtual const Operator *GetProlongation() const { return P; }
/// Get the finite element space restriction matrix
+38 -77
View File
@@ -13,101 +13,62 @@
// PABilinearFormExtension and MFBilinearFormExtension.
#include "nonlinearform.hpp"
#include "../general/forall.hpp"
namespace mfem
{
NonlinearFormExtension::NonlinearFormExtension(const NonlinearForm *nlf)
: Operator(nlf->FESpace()->GetTrueVSize()), nlf(nlf) { }
PANonlinearForm::PANonlinearForm(NonlinearForm *nlf):
NonlinearFormExtension(nlf),
x_grad(NULL),
fes(*nlf->FESpace()),
dnfi(*nlf->GetDNFI()),
R(fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC))
NonlinearFormExtension::NonlinearFormExtension(NonlinearForm *form)
: Operator(form->FESpace()->GetTrueVSize()), n(form)
{
MFEM_VERIFY(R, "Not yet implemented!");
xe.SetSize(R->Height(), Device::GetMemoryType());
ye.SetSize(R->Height(), Device::GetMemoryType());
ye.UseDevice(true);
// empty
}
double PANonlinearForm::GetGridFunctionEnergy(const Vector &x) const
PANonlinearFormExtension::PANonlinearFormExtension(NonlinearForm *form):
NonlinearFormExtension(form), fes(*form->FESpace())
{
double energy = 0.0;
R->Mult(x, xe);
for (int i = 0; i < dnfi.Size(); i++)
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
elem_restrict_lex = fes.GetElementRestriction(ordering);
if (elem_restrict_lex)
{
energy += dnfi[i]->GetGridFunctionEnergyPA(xe);
localX.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
localY.UseDevice(true); // ensure 'localY = 0.0' is done on device
}
return energy;
}
void PANonlinearForm::Setup()
void PANonlinearFormExtension::AssemblePA()
{
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(fes); }
}
void PANonlinearForm::Mult(const Vector &x, Vector &y) const
{
ye = 0.0;
R->Mult(x, xe);
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultPA(xe, ye); }
R->MultTranspose(ye, y);
}
void PANonlinearForm::AssembleGradientDiagonal(Vector &diag) const
{
MFEM_VERIFY(x_grad, "GetGradient() has not been called");
R->Mult(*x_grad, xe);
ye = 0.0;
for (int i = 0; i < dnfi.Size(); ++i)
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int Ni = integrators.Size();
for (int i = 0; i < Ni; ++i)
{
dnfi[i]->AssembleGradientDiagonalPA(xe, ye);
integrators[i]->AssemblePA(*n->FESpace());
}
R->MultTranspose(ye, diag);
}
Operator &PANonlinearForm::GetGradient(const Vector &x) const
void PANonlinearFormExtension::Mult(const Vector &x, Vector &y) const
{
// Store the last x that was used to compute the gradient.
x_grad = &x;
Grad.Reset(new PANonlinearForm::Gradient(x, *this));
return *Grad.Ptr();
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
const int iSz = integrators.Size();
if (elem_restrict_lex)
{
elem_restrict_lex->Mult(x, localX);
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(localX, localY);
}
elem_restrict_lex->MultTranspose(localY, y);
}
else
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
for (int i = 0; i < iSz; ++i)
{
integrators[i]->AddMultPA(x, y);
}
}
}
PANonlinearForm::Gradient::Gradient(const Vector &x, const PANonlinearForm &e):
Operator(e.fes.GetVSize()), R(e.R), dnfi(e.dnfi)
{
ge.UseDevice(true);
ge.SetSize(R->Height(), Device::GetMemoryType());
R->Mult(x, ge);
xe.UseDevice(true);
xe.SetSize(R->Height(), Device::GetMemoryType());
ye.UseDevice(true);
ye.SetSize(R->Height(), Device::GetMemoryType());
ze.UseDevice(true);
ze.SetSize(R->Height(), Device::GetMemoryType());
// Do we still need to do this?
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(e.fes); }
}
void PANonlinearForm::Gradient::Mult(const Vector &x, Vector &y) const
{
ze = x;
ye = 0.0;
R->Mult(ze, xe);
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultGradPA(ge, xe, ye); }
R->MultTranspose(ye, y);
}
} // namespace mfem
+9 -41
View File
@@ -17,60 +17,28 @@
namespace mfem
{
class NonlinearForm;
class NonlinearFormIntegrator;
/** @brief Class extending the NonlinearForm class to support the different
AssemblyLevel%s. */
class NonlinearFormExtension : public Operator
{
protected:
const NonlinearForm *nlf;
NonlinearForm *n; ///< Not owned
public:
NonlinearFormExtension(const NonlinearForm*);
virtual void Setup() = 0;
virtual Operator &GetGradient(const Vector&) const = 0;
virtual double GetGridFunctionEnergy(const Vector &x) const = 0;
virtual void AssembleGradientDiagonal(Vector &diag) const
{
MFEM_ABORT("Not implemented for this assembly level!");
}
NonlinearFormExtension(NonlinearForm *form);
virtual void AssemblePA() = 0;
};
class PANonlinearForm;
/// Data and methods for partially-assembled nonlinear forms
class PANonlinearForm : public NonlinearFormExtension
class PANonlinearFormExtension : public NonlinearFormExtension
{
private:
class Gradient : public Operator
{
protected:
const Operator *R;
mutable Vector ge, xe, ye, ze;
const Array<NonlinearFormIntegrator*> &dnfi;
public:
Gradient(const Vector &x, const PANonlinearForm &ext);
virtual void Mult(const Vector &x, Vector &y) const;
};
protected:
mutable Vector xe, ye;
mutable const Vector *x_grad;
mutable OperatorHandle Grad;
const FiniteElementSpace &fes;
const Array<NonlinearFormIntegrator*> &dnfi;
const Operator *R;
const FiniteElementSpace &fes; // Not owned
mutable Vector localX, localY;
const Operator *elem_restrict_lex; // Not owned
public:
PANonlinearForm(NonlinearForm *nlf);
void Setup();
PANonlinearFormExtension(NonlinearForm*);
void AssemblePA();
void Mult(const Vector &x, Vector &y) const;
Operator &GetGradient(const Vector &x) const;
double GetGridFunctionEnergy(const Vector &x) const;
void AssembleGradientDiagonal(Vector &diag) const;
};
}
#endif // NONLINEARFORM_EXT_HPP
-21
View File
@@ -15,13 +15,6 @@
namespace mfem
{
double NonlinearFormIntegrator::GetGridFunctionEnergyPA(const Vector &x) const
{
mfem_error ("NonlinearFormIntegrator::GetGridFunctionEnergyPA(...)\n"
" is not implemented for this class.");
return 0.0;
}
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
{
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
@@ -41,20 +34,6 @@ void NonlinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AddMultGradPA(const Vector&,
const Vector&, Vector&) const
{
mfem_error ("NonlinearFormIntegrator::AddMultGradPA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssembleGradientDiagonalPA(const mfem::Vector &x,
mfem::Vector &diag) const
{
mfem_error ("NonlinearFormIntegrator::AssembleDiagonalPA(...)\n"
" is not implemented for this class.");
}
void NonlinearFormIntegrator::AssembleElementVector(
const FiniteElement &el, ElementTransformation &Tr,
const Vector &elfun, Vector &elvect)
-9
View File
@@ -68,9 +68,6 @@ public:
ElementTransformation &Tr,
const Vector &elfun);
/// Compute the local energy with partial assembly.
virtual double GetGridFunctionEnergyPA(const Vector &x) const;
/// Method defining partial assembly.
/** The result of the partial assembly is stored internally so that it can be
used later in the methods AddMultPA(). */
@@ -91,12 +88,6 @@ public:
called. */
virtual void AddMultPA(const Vector &x, Vector &y) const;
/// Method for partially assembled gradient action.
virtual void AddMultGradPA(const Vector &g,
const Vector &x, Vector &y) const;
virtual void AssembleGradientDiagonalPA(const Vector &x, Vector &diag) const;
virtual ~NonlinearFormIntegrator() { }
};
+1 -1
View File
@@ -241,7 +241,7 @@ void ParBilinearForm::Assemble(int skip_zeros)
BilinearForm::Assemble(skip_zeros);
if (!ext && fbfi.Size() > 0)
if (fbfi.Size() > 0)
{
AssembleSharedFaces(skip_zeros);
}
+2 -2
View File
@@ -3147,7 +3147,7 @@ static void SetSubVector(const int N,
const Array<int> &indices,
const Vector &in, Vector &out)
{
auto y = out.ReadWrite();
auto y = out.Write();
const auto x = in.Read();
const auto I = indices.Read();
MFEM_FORALL(i, N, y[I[i]] = x[i];);
@@ -3234,7 +3234,7 @@ static void AddSubVector(const int num_unique_dst_indices,
const Vector &src,
Vector &dst)
{
auto y = dst.ReadWrite();
auto y = dst.Write();
const auto x = src.Read();
const auto DST_I = unique_dst_indices.Read();
const auto SRC_O = unique_to_src_offsets.Read();
-2
View File
@@ -711,9 +711,7 @@ void ParGridFunction::SaveAsOne(std::ostream &out)
int *nfdofs = new int[NRanks];
int *nrdofs = new int[NRanks];
HostReadWrite();
values[0] = data;
nv[0] = pfes -> GetVSize();
nvdofs[0] = pfes -> GetNVDofs();
nedofs[0] = pfes -> GetNEDofs();
+9 -16
View File
@@ -14,7 +14,6 @@
#ifdef MFEM_USE_MPI
#include "fem.hpp"
#include "../general/forall.hpp"
namespace mfem
{
@@ -50,7 +49,6 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
if (fnfi.Size())
{
MFEM_VERIFY(!NonlinearForm::ext,"");
// Terms over shared interior faces in parallel.
ParFiniteElementSpace *pfes = ParFESpace();
ParMesh *pmesh = pfes->GetParMesh();
@@ -88,16 +86,15 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
P->MultTranspose(aux2, y);
const int N = ess_tdof_list.Size();
const auto idx = ess_tdof_list.Read();
auto Y = y.ReadWrite();
MFEM_FORALL(i, N, Y[idx[i]] = 0.0; );
y.HostReadWrite();
for (int i = 0; i < ess_tdof_list.Size(); i++)
{
y(ess_tdof_list[i]) = 0.0;
}
}
const SparseMatrix &ParNonlinearForm::GetLocalGradient(const Vector &x) const
{
if (NonlinearForm::ext) { MFEM_ABORT("Not yet implemented!"); }
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
return *Grad;
@@ -107,20 +104,16 @@ Operator &ParNonlinearForm::GetGradient(const Vector &x) const
{
ParFiniteElementSpace *pfes = ParFESpace();
Operator &grad = NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
pGrad.Clear();
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
OperatorHandle dA(pGrad.Type()), Ph(pGrad.Type());
if (fnfi.Size() == 0)
{
if (NonlinearForm::ext) { dA.Reset(&grad, false); }
else
{
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
pfes->GetDofOffsets(), Grad);
}
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
pfes->GetDofOffsets(), Grad);
}
else
{
+57 -136
View File
@@ -27,24 +27,35 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
ElementDofOrdering e_ordering,
FaceType type,
L2FaceValues m)
: L2FaceRestriction(fes, type, m)
: fes(fes),
nf(fes.GetNFbyType(type)),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndofs(fes.GetNDofs()),
dof(nf>0 ?
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
: 0),
m(m),
nfdofs(nf*dof),
scatter_indices1(nf*dof),
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
offsets(ndofs+1),
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
{
if (nf==0) { return; }
// If fespace == L2
const ParFiniteElementSpace &pfes =
static_cast<const ParFiniteElementSpace&>(this->fes);
const FiniteElement *fe = pfes.GetFE(0);
const FiniteElement *fe = fes.GetFE(0);
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement*>(fe);
MFEM_VERIFY(tfe != NULL &&
(tfe->GetBasisType()==BasisType::GaussLobatto ||
tfe->GetBasisType()==BasisType::Positive),
"Only Gauss-Lobatto and Bernstein basis are supported in "
"ParL2FaceRestriction.");
MFEM_VERIFY(pfes.GetMesh()->Conforming(),
MFEM_VERIFY(fes.GetMesh()->Conforming(),
"Non-conforming meshes not yet supported with partial assembly.");
// Assuming all finite elements are using Gauss-Lobatto dofs
height = (m==L2FaceValues::DoubleValued? 2 : 1)*vdim*nf*dof;
width = pfes.GetVSize();
width = fes.GetVSize();
const bool dof_reorder = (e_ordering == ElementDofOrdering::LEXICOGRAPHIC);
if (!dof_reorder)
{
@@ -52,32 +63,32 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
}
if (dof_reorder && nf > 0)
{
for (int f = 0; f < pfes.GetNF(); ++f)
for (int f = 0; f < fes.GetNF(); ++f)
{
const FiniteElement *fe =
pfes.GetTraceElement(f, pfes.GetMesh()->GetFaceBaseGeometry(f));
fes.GetTraceElement(f, fes.GetMesh()->GetFaceBaseGeometry(f));
const TensorBasisElement* el =
dynamic_cast<const TensorBasisElement*>(fe);
if (el) { continue; }
mfem_error("Finite element not suitable for lexicographic ordering");
}
}
const Table& e2dTable = pfes.GetElementToDofTable();
const Table& e2dTable = fes.GetElementToDofTable();
const int* elementMap = e2dTable.GetJ();
Array<int> faceMap1(dof), faceMap2(dof);
int e1, e2;
int inf1, inf2;
int face_id1, face_id2;
int orientation;
const int dof1d = pfes.GetFE(0)->GetOrder()+1;
const int elem_dofs = pfes.GetFE(0)->GetDof();
const int dim = pfes.GetMesh()->SpaceDimension();
const int dof1d = fes.GetFE(0)->GetOrder()+1;
const int elem_dofs = fes.GetFE(0)->GetDof();
const int dim = fes.GetMesh()->SpaceDimension();
// Computation of scatter indices
int f_ind=0;
for (int f = 0; f < pfes.GetNF(); ++f)
for (int f = 0; f < fes.GetNF(); ++f)
{
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if (dof_reorder)
{
orientation = inf1 % 64;
@@ -125,7 +136,7 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
{
const int se2 = -1 - e2;
Array<int> sharedDofs;
pfes.GetFaceNbrElementVDofs(se2, sharedDofs);
fes.GetFaceNbrElementVDofs(se2, sharedDofs);
for (int d = 0; d < dof; ++d)
{
const int pd = PermuteFaceL2(dim, face_id1, face_id2,
@@ -169,10 +180,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
offsets[i] = 0;
}
f_ind = 0;
for (int f = 0; f < pfes.GetNF(); ++f)
for (int f = 0; f < fes.GetNF(); ++f)
{
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
(type==FaceType::Boundary && e2<0 && inf2<0) )
{
@@ -211,10 +222,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
offsets[i] += offsets[i - 1];
}
f_ind = 0;
for (int f = 0; f < pfes.GetNF(); ++f)
for (int f = 0; f < fes.GetNF(); ++f)
{
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
(type==FaceType::Boundary && e2<0 && inf2<0) )
{
@@ -261,10 +272,8 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
{
const ParFiniteElementSpace &pfes =
static_cast<const ParFiniteElementSpace&>(this->fes);
ParGridFunction x_gf;
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&pfes),
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&fes),
const_cast<Vector&>(x), 0);
x_gf.ExchangeFaceNbrData();
@@ -328,122 +337,34 @@ void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
}
}
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
void ParL2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
{
int val = AtomicAdd(I[iE],dofs);
return val;
}
void ParL2FaceRestriction::FillI(SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
const int Ndofs = ndofs;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
auto I_face = face_mat.ReadWriteI();
MFEM_FORALL(i, ne*elemDofs*vdim+1,
// Assumes all elements have the same number of dofs
const int nd = dof;
const int vd = vdim;
const bool t = byvdim;
const int dofs = nfdofs;
auto d_offsets = offsets.Read();
auto d_indices = gather_indices.Read();
auto d_x = Reshape(x.Read(), nd, vd, 2, nf);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
{
I_face[i] = 0;
});
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
if (iE1 < Ndofs)
const int offset = d_offsets[i];
const int nextOffset = d_offsets[i + 1];
for (int c = 0; c < vd; ++c)
{
for (int jF = 0; jF < face_dofs; jF++)
double dofValue = 0;
for (int j = offset; j < nextOffset; ++j)
{
const int jE2 = d_indices2[f*face_dofs+jF];
if (jE2 < Ndofs)
{
AddNnz(iE1,I,1);
}
else
{
AddNnz(iE1,I_face,1);
}
}
}
const int iE2 = d_indices2[f*face_dofs+iF];
if (iE2 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
if (jE1 < Ndofs)
{
AddNnz(iE2,I,1);
}
else
{
AddNnz(iE2,I_face,1);
}
}
}
});
}
void ParL2FaceRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
const int Ndofs = ndofs;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
auto I = mat.ReadWriteI();
auto I_face = face_mat.ReadWriteI();
auto J = mat.WriteJ();
auto J_face = face_mat.WriteJ();
auto Data = mat.WriteData();
auto Data_face = face_mat.WriteData();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
if (iE1 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE2 = d_indices2[f*face_dofs+jF];
if (jE2 < Ndofs)
{
const int offset = AddNnz(iE1,I,1);
J[offset] = jE2;
Data[offset] = mat_fea(jF,iF,1,f);
}
else
{
const int offset = AddNnz(iE1,I_face,1);
J_face[offset] = jE2-Ndofs;
Data_face[offset] = mat_fea(jF,iF,1,f);
}
}
}
const int iE2 = d_indices2[f*face_dofs+iF];
if (iE2 < Ndofs)
{
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
if (jE1 < Ndofs)
{
const int offset = AddNnz(iE2,I,1);
J[offset] = jE1;
Data[offset] = mat_fea(jF,iF,0,f);
}
else
{
const int offset = AddNnz(iE2,I_face,1);
J_face[offset] = jE1-Ndofs;
Data_face[offset] = mat_fea(jF,iF,0,f);
}
int idx_j = d_indices[j];
bool isE1 = idx_j < dofs;
idx_j = isE1 ? idx_j : idx_j - dofs;
dofValue += isE1 ?
d_x(idx_j % nd, c, 0, idx_j / nd)
:d_x(idx_j % nd, c, 1, idx_j / nd);
}
d_y(t?c:i,t?i:c) += dofValue;
}
});
}
+16 -9
View File
@@ -26,21 +26,28 @@ class ParFiniteElementSpace;
/// Operator that extracts Face degrees of freedom in parallel.
/** Objects of this type are typically created and owned by FiniteElementSpace
objects, see FiniteElementSpace::GetFaceRestriction(). */
class ParL2FaceRestriction : public L2FaceRestriction
class ParL2FaceRestriction : public Operator
{
protected:
const ParFiniteElementSpace &fes;
const int nf;
const int vdim;
const bool byvdim;
const int ndofs;
const int dof;
const L2FaceValues m;
const int nfdofs;
Array<int> scatter_indices1;
Array<int> scatter_indices2;
Array<int> offsets;
Array<int> gather_indices;
public:
ParL2FaceRestriction(const ParFiniteElementSpace&, ElementDofOrdering,
FaceType type,
L2FaceValues m = L2FaceValues::DoubleValued);
void Mult(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ParL2FaceRestriction. */
void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this ParL2FaceRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const;
void MultTranspose(const Vector &x, Vector &y) const;
};
}
+1121 -396
View File
File diff suppressed because it is too large Load Diff
+15 -33
View File
@@ -41,11 +41,10 @@ protected:
const FiniteElementSpace *fespace; ///< Not owned
const QuadratureSpace *qspace; ///< Not owned
const IntegrationRule *IntRule; ///< Not owned
mutable QVectorLayout q_layout; ///< Output Q-vector layout
mutable bool use_tensor_products; ///< Tensor product evaluation mmode
public:
mutable bool use_tensor_products;
static const int MAX_NQ2D = 100;
static const int MAX_ND2D = 100;
static const int MAX_VDIM2D = 3;
@@ -54,6 +53,7 @@ public:
static const int MAX_ND3D = 1000;
static const int MAX_VDIM3D = 3;
public:
enum EvalFlags
{
VALUES = 1 << 0, ///< Evaluate the values at quadrature points
@@ -61,28 +61,21 @@ public:
/** @brief Assuming the derivative at quadrature points form a matrix,
this flag can be used to compute and store their determinants. This
flag can only be used in Mult(). */
DETERMINANTS = 1 << 2,
PHYSICAL_DERIVATIVES = 1 << 3 ///< Evaluate the physical derivatives
DETERMINANTS = 1 << 2
};
QuadratureInterpolator(const FiniteElementSpace &fes,
const IntegrationRule &ir,
const bool use_tensor_products = false);
const IntegrationRule &ir);
QuadratureInterpolator(const FiniteElementSpace &fes,
const QuadratureSpace &qs,
const bool use_tensor_products = false);
const QuadratureSpace &qs);
/** @brief Disable the use of tensor product evaluations, for tensor-product
elements, e.g. quads and hexes. */
void DisableTensorProducts() const { use_tensor_products = false; }
/** @brief Enable the use of tensor product evaluations, for tensor-product
elements, e.g. quads and hexes. */
void EnableTensorProducts() const { use_tensor_products = true; }
/** @brief Query the current evaluation mode. */
bool UseTensorProducts() const { return use_tensor_products; }
/** Currently, tensor product evaluations are not implemented and this method
has no effect. */
void DisableTensorProducts(bool disable = true) const
{ use_tensor_products = !disable; }
/** @brief Query the current output Q-vector layout. The default value is
QVectorLayout::byNODES. */
@@ -90,7 +83,8 @@ public:
/** @brief Set the desired output Q-vector layout. The default value is
QVectorLayout::byNODES. */
void SetOutputLayout(QVectorLayout layout) const { q_layout = layout; }
void SetOutputLayout(QVectorLayout out_layout) const
{ q_layout = out_layout; }
/// Interpolate the E-vector @a e_vec to quadrature points.
/** The @a eval_flags are a bitwise mask of constants from the EvalFlags
@@ -105,36 +99,26 @@ public:
Vector &q_val, Vector &q_der, Vector &q_det) const;
/// Interpolate the values of the E-vector @a e_vec at quadrature points.
template <QVectorLayout>
void Values(const Vector &e_vec, Vector &q_val) const;
void Values(const Vector &e_vec, Vector &q_val) const;
/** @brief Interpolate the derivatives of the E-vector @a e_vec at quadrature
points. */
template <QVectorLayout>
void Derivatives(const Vector &e_vec, Vector &q_der) const;
void Derivatives(const Vector &e_vec, Vector &q_der) const;
/** @brief Interpolate the derivatives in physical space of the E-vector
@a e_vec at quadrature points. */
template <QVectorLayout>
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
/// Compute the determinant of the E-vector @a e_vec at quadrature points.
void Determinants(const Vector &e_vec, Vector &q_det) const;
/// Perform the transpose operation of Mult(). (TODO)
void MultTranspose(unsigned eval_flags, const Vector &q_val,
const Vector &q_der, Vector &e_vec) const;
// Compute kernels follow (cannot be private or protected with nvcc)
/// Template compute kernel for 2D.
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
static void Mult2D(const int NE,
static void Eval2D(const int NE,
const int vdim,
const QVectorLayout q_layout,
const GeometricFactors *geom,
const DofToQuad &maps,
const Vector &e_vec,
Vector &q_val,
@@ -144,10 +128,8 @@ public:
/// Template compute kernel for 3D.
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
static void Mult3D(const int NE,
static void Eval3D(const int NE,
const int vdim,
const QVectorLayout q_layout,
const GeometricFactors *geom,
const DofToQuad &maps,
const Vector &e_vec,
Vector &q_val,
-208
View File
@@ -1,208 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop_pa.hpp"
#include "quadinterpolator.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../fem/kernels.hpp"
#include "../linalg/kernels.hpp"
using namespace mfem;
namespace mfem
{
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
static void Det2D(const int NE,
const double *b,
const double *g,
const double *x,
double *y,
const int vdim = 1,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 2;
constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_SHARED double BG[2][MQ1*MD1];
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double J[4];
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,J);
Y(qx,qy,e) = kernels::Det<2>(J);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
static void Det3D(const int NE,
const double *b,
const double *g,
const double *x,
double *y,
const int vdim = 1,
const int d1d = 0,
const int q1d = 0)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
MFEM_SHARED double BG[2][MQ1*MD1];
MFEM_SHARED double sm0[9][MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[9][MDQ*MDQ*MDQ];
double (*DDD)[MD1*MD1*MD1] = (double (*)[MD1*MD1*MD1]) (sm0);
double (*DDQ)[MD1*MD1*MQ1] = (double (*)[MD1*MD1*MQ1]) (sm1);
double (*DQQ)[MD1*MQ1*MQ1] = (double (*)[MD1*MQ1*MQ1]) (sm0);
double (*QQQ)[MQ1*MQ1*MQ1] = (double (*)[MQ1*MQ1*MQ1]) (sm1);
kernels::LoadX<MD1>(e,D1D,X,DDD);
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
kernels::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
kernels::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double J[9];
kernels::PullGrad<MQ1>(qx,qy,qz, QQQ, J);
Y(qx,qy,qz,e) = kernels::Det<3>(J);
}
}
}
});
}
void QuadratureInterpolator::Determinants(const Vector &e_vec,
Vector &q_det) const
{
if (use_tensor_products)
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *G = maps.G.Read();
const double *X = e_vec.Read();
double *Y = q_det.Write();
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
switch (id)
{
case 0x2222: return Det2D<2,2>(NE,B,G,X,Y);
case 0x2223: return Det2D<2,3>(NE,B,G,X,Y);
case 0x2224: return Det2D<2,4>(NE,B,G,X,Y);
case 0x2226: return Det2D<2,6>(NE,B,G,X,Y);
case 0x2234: return Det2D<3,4>(NE,B,G,X,Y);
case 0x2236: return Det2D<3,6>(NE,B,G,X,Y);
case 0x2244: return Det2D<4,4>(NE,B,G,X,Y);
case 0x2246: return Det2D<4,6>(NE,B,G,X,Y);
case 0x2256: return Det2D<5,6>(NE,B,G,X,Y);
case 0x3324: return Det3D<2,4>(NE,B,G,X,Y);
case 0x3333: return Det3D<3,3>(NE,B,G,X,Y);
case 0x3335: return Det3D<3,5>(NE,B,G,X,Y);
case 0x3336: return Det3D<3,6>(NE,B,G,X,Y);
//case 0x3348: return Det3D<4,8>(NE,B,G,X,Y);
default:
{
if (dim == 2)
{
constexpr int MD1 = 8;
constexpr int MQ1 = 8;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
return Det2D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
}
if (dim == 3)
{
constexpr int MD1 = 6;
constexpr int MQ1 = 6;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
return Det3D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
}
}
}
MFEM_ABORT("Kernel " << std::hex << id << std::dec << " not supported yet");
}
else
{
Vector empty;
Mult(e_vec, DETERMINANTS, empty, empty, q_det);
}
}
} // namespace mfem
-233
View File
@@ -1,233 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
static void Eval2D(const int NE,
const double *b_,
const double *x_,
double *y_,
const int vdim = 0,
const int d1d = 0,
const int q1d = 0)
{
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, VDIM, NE):
Reshape(y_, VDIM, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1*MD1];
DeviceTensor<2,double> B(s_B, Q1D, D1D);
MFEM_SHARED double s_DD[NBZ][MD1*MD1];
DeviceTensor<2,double> DD((double*)(s_DD+tidz), MD1, MD1);
MFEM_SHARED double s_DQ[NBZ][MD1*MQ1];
DeviceTensor<2,double> DQ((double*)(s_DQ+tidz), MD1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; c++)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
DD(dx,dy) = x(dx,dy,c,e);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
u += B(qx,dx) * DD(dx,dy);
}
DQ(dy,qx) = u;
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DQ(dy,qx) * B(qy,dy);
}
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,e) = u; }
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,c,e) = u; }
}
}
MFEM_SYNC_THREAD;
}
});
}
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int MAX_D1D = 0, int MAX_Q1D = 0>
static void Eval3D(const int NE,
const double *b_,
const double *x_,
double *y_,
const int vdim = 0,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, Q1D, VDIM, NE):
Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1*MD1];
DeviceTensor<2,double> B(s_B, Q1D, D1D);
MFEM_SHARED double sm0[MDQ*MDQ*MDQ];
MFEM_SHARED double sm1[MDQ*MDQ*MDQ];
DeviceTensor<3,double> DDD(sm0, MD1, MD1, MD1);
DeviceTensor<3,double> DDQ(sm1, MD1, MD1, MQ1);
DeviceTensor<3,double> DQQ(sm0, MD1, MQ1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; c++)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
DDD(dx,dy,dz) = x(dx,dy,dz,c,e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
u += B(qx,dx) * DDD(dx,dy,dz);
}
DDQ(dz,dy,qx) = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ(dz,dy,qx) * B(qy,dy);
}
DQQ(dz,qy,qx) = u;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ(dz,qy,qx) * B(qz,dz);
}
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,qz,e) = u; }
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,qz,c,e) = u; }
}
}
}
MFEM_SYNC_THREAD;
}
});
}
} // namespace mfem
-110
View File
@@ -1,110 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_eval.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::Values<QVectorLayout::byNODES>(
const Vector &e_vec, Vector &q_val) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *X = e_vec.Read();
double *Y = q_val.Write();
constexpr QVectorLayout L = QVectorLayout::byNODES;
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
switch (id)
{
case 0x2133: return Eval2D<L,1,3,3>(NE,B,X,Y);
case 0x2124: return Eval2D<L,1,2,4>(NE,B,X,Y);
case 0x2132: return Eval2D<L,1,3,2>(NE,B,X,Y);
case 0x2134: return Eval2D<L,1,3,4>(NE,B,X,Y);
case 0x2143: return Eval2D<L,1,4,3>(NE,B,X,Y);
case 0x2144: return Eval2D<L,1,4,4>(NE,B,X,Y);
case 0x2222: return Eval2D<L,2,2,2>(NE,B,X,Y);
case 0x2223: return Eval2D<L,2,2,3>(NE,B,X,Y);
case 0x2224: return Eval2D<L,2,2,4>(NE,B,X,Y);
case 0x2225: return Eval2D<L,2,2,5>(NE,B,X,Y);
case 0x2226: return Eval2D<L,2,2,6>(NE,B,X,Y);
case 0x2233: return Eval2D<L,2,3,3>(NE,B,X,Y);
case 0x2234: return Eval2D<L,2,3,4>(NE,B,X,Y);
case 0x2236: return Eval2D<L,2,3,6>(NE,B,X,Y);
case 0x2243: return Eval2D<L,2,4,3>(NE,B,X,Y);
case 0x2244: return Eval2D<L,2,4,4>(NE,B,X,Y);
case 0x2245: return Eval2D<L,2,4,5>(NE,B,X,Y);
case 0x2246: return Eval2D<L,2,4,6>(NE,B,X,Y);
case 0x2247: return Eval2D<L,2,4,7>(NE,B,X,Y);
case 0x2256: return Eval2D<L,2,5,6>(NE,B,X,Y);
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
case 0x3133: return Eval3D<L,1,3,3>(NE,B,X,Y);
case 0x3134: return Eval3D<L,1,3,4>(NE,B,X,Y);
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
case 0x3143: return Eval3D<L,1,4,3>(NE,B,X,Y);
case 0x3144: return Eval3D<L,1,4,4>(NE,B,X,Y);
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
case 0x3222: return Eval3D<L,2,2,2>(NE,B,X,Y);
case 0x3223: return Eval3D<L,2,2,3>(NE,B,X,Y);
case 0x3234: return Eval3D<L,2,3,4>(NE,B,X,Y);
case 0x3323: return Eval3D<L,3,2,3>(NE,B,X,Y);
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
case 0x3325: return Eval3D<L,3,2,5>(NE,B,X,Y);
case 0x3326: return Eval3D<L,3,2,6>(NE,B,X,Y);
case 0x3333: return Eval3D<L,3,3,3>(NE,B,X,Y);
case 0x3334: return Eval3D<L,3,3,4>(NE,B,X,Y);
case 0x3335: return Eval3D<L,3,3,5>(NE,B,X,Y);
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
case 0x3343: return Eval3D<L,3,4,3>(NE,B,X,Y);
case 0x3344: return Eval3D<L,3,4,4>(NE,B,X,Y);
case 0x3346: return Eval3D<L,3,4,6>(NE,B,X,Y);
case 0x3347: return Eval3D<L,3,4,7>(NE,B,X,Y);
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
default:
{
constexpr int MD1 = 8;
constexpr int MQ1 = 8;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
return;
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Kernel not supported yet");
}
} // namespace mfem
-79
View File
@@ -1,79 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_eval.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::Values<QVectorLayout::byVDIM>(
const Vector &e_vec, Vector &q_val) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *X = e_vec.Read();
double *Y = q_val.Write();
constexpr QVectorLayout L = QVectorLayout::byVDIM;
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
switch (id)
{
case 0x2124: return Eval2D<L,1,2,4,8>(NE,B,X,Y);
case 0x2136: return Eval2D<L,1,3,6,4>(NE,B,X,Y);
case 0x2148: return Eval2D<L,1,4,8,2>(NE,B,X,Y);
case 0x2224: return Eval2D<L,2,2,4,8>(NE,B,X,Y);
case 0x2234: return Eval2D<L,2,3,4,8>(NE,B,X,Y);
case 0x2236: return Eval2D<L,2,3,6,4>(NE,B,X,Y);
case 0x2248: return Eval2D<L,2,4,8,2>(NE,B,X,Y);
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
default:
{
constexpr int MD1 = 8;
constexpr int MQ1 = 8;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
return;
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Kernel not supported yet");
}
} // namespace mfem
+2 -3
View File
@@ -495,9 +495,8 @@ void FaceQuadratureInterpolator::Mult(
}
}
void FaceQuadratureInterpolator::Values(const Vector &e_vec,
Vector &q_val) const
void FaceQuadratureInterpolator::Values(
const Vector &e_vec, Vector &q_val) const
{
Vector q_der, q_det, q_nor;
Mult(e_vec, VALUES, q_val, q_der, q_det, q_nor);
-282
View File
@@ -1,282 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
static void Grad2D(const int NE,
const double *b_,
const double *g_,
const double *x_,
double *y_,
const int vdim = 0,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto g = Reshape(g_, Q1D, D1D);
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1*MD1];
MFEM_SHARED double s_G[MQ1*MD1];
DeviceTensor<2,double> B(s_B, Q1D, D1D);
DeviceTensor<2,double> G(s_G, Q1D, D1D);
MFEM_SHARED double s_X[NBZ][MD1*MD1];
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
MFEM_SHARED double s_DQ[2][NBZ][MD1*MQ1];
DeviceTensor<2,double> DQ0((double*)(s_DQ[0]+tidz), MD1, MQ1);
DeviceTensor<2,double> DQ1((double*)(s_DQ[1]+tidz), MD1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
G(q,d) = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
X(dx,dy) = x(dx,dy,c,e);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double input = X(dx,dy);
u += input * B(qx,dx);
v += input * G(qx,dx);
}
DQ0(dy,qx) = u;
DQ1(dy,qx) = v;
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DQ1(dy,qx) * B(qy,dy);
v += DQ0(dy,qx) * G(qy,dy);
}
if (Q_LAYOUT == QVectorLayout::byNODES)
{
y(qx,qy,c,0,e) = u;
y(qx,qy,c,1,e) = v;
}
if (Q_LAYOUT == QVectorLayout::byVDIM)
{
y(c,0,qx,qy,e) = u;
y(c,1,qx,qy,e) = v;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int MAX_D1D = 0, int MAX_Q1D = 0>
static void Grad3D(const int NE,
const double *b_,
const double *g_,
const double *x_,
double *y_,
const int vdim = 0,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto g = Reshape(g_, Q1D, D1D);
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1*MD1];
MFEM_SHARED double s_G[MQ1*MD1];
DeviceTensor<2,double> B(s_B, Q1D, D1D);
DeviceTensor<2,double> G(s_G, Q1D, D1D);
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
G(q,d) = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
X(dx,dy,dz) = x(dx,dy,dz,c,e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double input = X(dx,dy,dz);
u += input * B(qx,dx);
v += input * G(qx,dx);
}
DDQ0(dz,dy,qx) = u;
DDQ1(dz,dy,qx) = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1(dz,dy,qx) * B(qy,dy);
v += DDQ0(dz,dy,qx) * G(qy,dy);
w += DDQ0(dz,dy,qx) * B(qy,dy);
}
DQQ0(dz,qy,qx) = u;
DQQ1(dz,qy,qx) = v;
DQQ2(dz,qy,qx) = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0(dz,qy,qx) * B(qz,dz);
v += DQQ1(dz,qy,qx) * B(qz,dz);
w += DQQ2(dz,qy,qx) * G(qz,dz);
}
if (Q_LAYOUT == QVectorLayout::byNODES)
{
y(qx,qy,qz,c,0,e) = u;
y(qx,qy,qz,c,1,e) = v;
y(qx,qy,qz,c,2,e) = w;
}
if (Q_LAYOUT == QVectorLayout::byVDIM)
{
y(c,0,qx,qy,qz,e) = u;
y(c,1,qx,qy,qz,e) = v;
y(c,2,qx,qy,qz,e) = w;
}
}
}
}
MFEM_SYNC_THREAD;
}
});
}
} // namespace mfem
-109
View File
@@ -1,109 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_grad.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::Derivatives<QVectorLayout::byNODES>(
const Vector &e_vec, Vector &q_der) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *G = maps.G.Read();
const double *X = e_vec.Read();
double *Y = q_der.Write();
constexpr QVectorLayout L = QVectorLayout::byNODES;
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
switch (id)
{
case 0x2133: return Grad2D<L,1,3,3,16>(NE,B,G,X,Y);
case 0x2134: return Grad2D<L,1,3,4,16>(NE,B,G,X,Y);
case 0x2143: return Grad2D<L,1,4,3,16>(NE,B,G,X,Y);
case 0x2144: return Grad2D<L,1,4,4,16>(NE,B,G,X,Y);
case 0x2222: return Grad2D<L,2,2,2,16>(NE,B,G,X,Y);
case 0x2223: return Grad2D<L,2,2,3,8>(NE,B,G,X,Y);
case 0x2224: return Grad2D<L,2,2,4,4>(NE,B,G,X,Y);
case 0x2225: return Grad2D<L,2,2,5,4>(NE,B,G,X,Y);
case 0x2226: return Grad2D<L,2,2,6,2>(NE,B,G,X,Y);
case 0x2233: return Grad2D<L,2,3,3,2>(NE,B,G,X,Y);
case 0x2234: return Grad2D<L,2,3,4,4>(NE,B,G,X,Y);
case 0x2243: return Grad2D<L,2,4,3,4>(NE,B,G,X,Y);
case 0x2236: return Grad2D<L,2,3,6,2>(NE,B,G,X,Y);
case 0x2244: return Grad2D<L,2,4,4,2>(NE,B,G,X,Y);
case 0x2245: return Grad2D<L,2,4,5,2>(NE,B,G,X,Y);
case 0x2246: return Grad2D<L,2,4,6,2>(NE,B,G,X,Y);
case 0x2247: return Grad2D<L,2,4,7,2>(NE,B,G,X,Y);
case 0x2256: return Grad2D<L,2,5,6,2>(NE,B,G,X,Y);
case 0x3124: return Grad3D<L,1,2,4>(NE,B,G,X,Y);
case 0x3133: return Grad3D<L,1,3,3>(NE,B,G,X,Y);
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
case 0x3136: return Grad3D<L,1,3,6>(NE,B,G,X,Y);
case 0x3144: return Grad3D<L,1,4,4>(NE,B,G,X,Y);
case 0x3148: return Grad3D<L,1,4,8>(NE,B,G,X,Y);
case 0x3323: return Grad3D<L,3,2,3>(NE,B,G,X,Y);
case 0x3324: return Grad3D<L,3,2,4>(NE,B,G,X,Y);
case 0x3325: return Grad3D<L,3,2,5>(NE,B,G,X,Y);
case 0x3326: return Grad3D<L,3,2,6>(NE,B,G,X,Y);
case 0x3333: return Grad3D<L,3,3,3>(NE,B,G,X,Y);
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
case 0x3335: return Grad3D<L,3,3,5>(NE,B,G,X,Y);
case 0x3336: return Grad3D<L,3,3,6>(NE,B,G,X,Y);
case 0x3344: return Grad3D<L,3,4,4>(NE,B,G,X,Y);
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
case 0x3347: return Grad3D<L,3,4,7>(NE,B,G,X,Y);
case 0x3348: return Grad3D<L,3,4,8>(NE,B,G,X,Y);
default:
{
constexpr int MD1 = 8;
constexpr int MQ1 = 8;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
if (dim == 2)
{
return Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
}
if (dim == 3)
{
return Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
}
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Kernel not supported yet");
}
} // namespace mfem
-76
View File
@@ -1,76 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_grad.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::Derivatives<QVectorLayout::byVDIM>(
const Vector &e_vec, Vector &q_der) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *G = maps.G.Read();
const double *X = e_vec.Read();
double *Y = q_der.Write();
constexpr QVectorLayout L = QVectorLayout::byVDIM;
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
switch (id)
{
case 0x2134: return Grad2D<L,1,3,4,8>(NE,B,G,X,Y);
case 0x2146: return Grad2D<L,1,4,6,4>(NE,B,G,X,Y);
case 0x2158: return Grad2D<L,1,5,8,2>(NE,B,G,X,Y);
case 0x2234: return Grad2D<L,2,3,4,8>(NE,B,G,X,Y);
case 0x2246: return Grad2D<L,2,4,6,4>(NE,B,G,X,Y);
case 0x2258: return Grad2D<L,2,5,8,2>(NE,B,G,X,Y);
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
case 0x3146: return Grad3D<L,1,4,6>(NE,B,G,X,Y);
case 0x3158: return Grad3D<L,1,5,8>(NE,B,G,X,Y);
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
case 0x3358: return Grad3D<L,3,5,8>(NE,B,G,X,Y);
default:
{
constexpr int MD1 = 8;
constexpr int MQ1 = 8;
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
<< MQ1 << " 1D points are not supported!");
if (dim == 2) { Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
if (dim == 3) { Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
return;
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Kernel not supported yet");
}
} // namespace mfem
-303
View File
@@ -1,303 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "../general/forall.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
static void PhysGrad2D(const int NE,
const double *b_,
const double *g_,
const double *j_,
const double *x_,
double *y_,
const int vdim = 0,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
const auto b = Reshape(b_, Q1D, D1D);
const auto g = Reshape(g_, Q1D, D1D);
const auto j = Reshape(j_, Q1D, Q1D, 2, 2, NE);
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1][MD1];
MFEM_SHARED double s_G[MQ1][MD1];
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
MFEM_SHARED double s_X[NBZ][MD1*MD1];
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
MFEM_SHARED double sm[2][NBZ][MD1*MQ1];
DeviceTensor<2,double> DQ0((double*)(sm[0]+tidz), MD1, MQ1);
DeviceTensor<2,double> DQ1((double*)(sm[1]+tidz), MD1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
G(q,d) = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
X(dx,dy) = x(dx,dy,c,e);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double input = X(dx,dy);
u += input * B(qx,dx);
v += input * G(qx,dx);
}
DQ0(dy,qx) = u;
DQ1(dy,qx) = v;
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DQ1(dy,qx) * B(qy,dy);
v += DQ0(dy,qx) * G(qy,dy);
}
double Jloc[4], Jinv[4];
Jloc[0] = j(qx,qy,0,0,e);
Jloc[1] = j(qx,qy,1,0,e);
Jloc[2] = j(qx,qy,0,1,e);
Jloc[3] = j(qx,qy,1,1,e);
kernels::CalcInverse<2>(Jloc, Jinv);
if (Q_LAYOUT == QVectorLayout::byVDIM)
{
y(c,0,qx,qy,e) = Jinv[0]*u + Jinv[1]*v;
y(c,1,qx,qy,e) = Jinv[2]*u + Jinv[3]*v;
}
if (Q_LAYOUT == QVectorLayout::byNODES)
{
y(qx,qy,c,0,e) = Jinv[0]*u + Jinv[1]*v;
y(qx,qy,c,1,e) = Jinv[2]*u + Jinv[3]*v;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
template<QVectorLayout Q_LAYOUT,
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
int MAX_D = 0, int MAX_Q = 0>
static void PhysGrad3D(const int NE,
const double *b_,
const double *g_,
const double *j_,
const double *x_,
double *y_,
const int vdim = 1,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
const auto b = Reshape(b_, Q1D, D1D);
const auto g = Reshape(g_, Q1D, D1D);
const auto j = Reshape(j_, Q1D, Q1D, Q1D, 3, 3, NE);
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED double s_B[MQ1][MD1];
MFEM_SHARED double s_G[MQ1][MD1];
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
if (tidz == 0)
{
MFEM_FOREACH_THREAD(d,y,D1D)
{
MFEM_FOREACH_THREAD(q,x,Q1D)
{
B(q,d) = b(q,d);
G(q,d) = g(q,d);
}
}
}
MFEM_SYNC_THREAD;
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
X(dx,dy,dz) = x(dx,dy,dz,c,e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const double coords = X(dx,dy,dz);
u += coords * B(qx,dx);
v += coords * G(qx,dx);
}
DDQ0(dz,dy,qx) = u;
DDQ1(dz,dy,qx) = v;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dy = 0; dy < D1D; ++dy)
{
u += DDQ1(dz,dy,qx) * B(qy,dy);
v += DDQ0(dz,dy,qx) * G(qy,dy);
w += DDQ0(dz,dy,qx) * B(qy,dy);
}
DQQ0(dz,qy,qx) = u;
DQQ1(dz,qy,qx) = v;
DQQ2(dz,qy,qx) = w;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
double u = 0.0;
double v = 0.0;
double w = 0.0;
for (int dz = 0; dz < D1D; ++dz)
{
u += DQQ0(dz,qy,qx) * B(qz,dz);
v += DQQ1(dz,qy,qx) * B(qz,dz);
w += DQQ2(dz,qy,qx) * G(qz,dz);
}
double Jloc[9], Jinv[9];
for (int col = 0; col < 3; col++)
{
for (int row = 0; row < 3; row++)
{
Jloc[row+3*col] = j(qx,qy,qz,row,col,e);
}
}
kernels::CalcInverse<3>(Jloc, Jinv);
if (Q_LAYOUT == QVectorLayout::byNODES)
{
y(qx,qy,qz,c,0,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
y(qx,qy,qz,c,1,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
y(qx,qy,qz,c,2,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
}
if (Q_LAYOUT == QVectorLayout::byVDIM)
{
y(c,0,qx,qy,qz,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
y(c,1,qx,qy,qz,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
y(c,2,qx,qy,qz,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
}
}
}
}
MFEM_SYNC_THREAD;
}
});
}
} // namespace mfem
-110
View File
@@ -1,110 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_grad_phys.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byNODES>(
const Vector &e_vec, Vector &q_der) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
Mesh *mesh = fespace->GetMesh();
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
const GeometricFactors *geom =
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *G = maps.G.Read();
const double *J = geom->J.Read();
const double *X = e_vec.Read();
double *Y = q_der.Write();
constexpr QVectorLayout L = QVectorLayout::byNODES;
const int id = (vdim<<8) | (D1D<<4) | Q1D;
if (dim == 2)
{
switch (id)
{
case 0x133: return PhysGrad2D<L,1,3,3,8>(NE,B,G,J,X,Y);
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE,B,G,J,X,Y);
case 0x143: return PhysGrad2D<L,1,4,3,4>(NE,B,G,J,X,Y);
case 0x144: return PhysGrad2D<L,1,4,4,4>(NE,B,G,J,X,Y);
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE,B,G,J,X,Y);
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE,B,G,J,X,Y);
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE,B,G,J,X,Y);
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE,B,G,J,X,Y);
case 0x243: return PhysGrad2D<L,2,4,3,4>(NE,B,G,J,X,Y);
case 0x244: return PhysGrad2D<L,2,4,4,4>(NE,B,G,J,X,Y);
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE,B,G,J,X,Y);
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE,B,G,J,X,Y);
default:
{
constexpr int MD = MAX_D1D;
constexpr int MQ = MAX_Q1D;
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
<< " 1D points are not supported!");
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
return;
}
}
}
if (dim == 3)
{
switch (id)
{
case 0x133: return PhysGrad3D<L,1,3,3>(NE,B,G,J,X,Y);
case 0x134: return PhysGrad3D<L,1,3,4>(NE,B,G,J,X,Y);
case 0x144: return PhysGrad3D<L,1,4,4>(NE,B,G,J,X,Y);
case 0x146: return PhysGrad3D<L,1,4,6>(NE,B,G,J,X,Y);
case 0x158: return PhysGrad3D<L,1,5,8>(NE,B,G,J,X,Y);
case 0x333: return PhysGrad3D<L,3,3,3>(NE,B,G,J,X,Y);
case 0x334: return PhysGrad3D<L,3,3,4>(NE,B,G,J,X,Y);
case 0x344: return PhysGrad3D<L,3,4,4>(NE,B,G,J,X,Y);
case 0x346: return PhysGrad3D<L,3,4,6>(NE,B,G,J,X,Y);
case 0x358: return PhysGrad3D<L,3,5,8>(NE,B,G,J,X,Y);
default:
{
constexpr int MD = 8;
constexpr int MQ = 8;
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
<< " 1D points are not supported!");
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
return;
}
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Unknown kernel");
}
} // namespace mfem
-101
View File
@@ -1,101 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "quadinterpolator.hpp"
#include "quadinterpolator_grad_phys.hpp"
namespace mfem
{
template<>
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byVDIM>(
const Vector &e_vec, Vector &q_der) const
{
const int NE = fespace->GetNE();
if (NE == 0) { return; }
Mesh *mesh = fespace->GetMesh();
const int vdim = fespace->GetVDim();
const int dim = fespace->GetMesh()->Dimension();
const FiniteElement *fe = fespace->GetFE(0);
const IntegrationRule *ir =
IntRule ? IntRule : &qspace->GetElementIntRule(0);
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
const GeometricFactors *geom =
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
const int D1D = maps.ndof;
const int Q1D = maps.nqpt;
const double *B = maps.B.Read();
const double *G = maps.G.Read();
const double *J = geom->J.Read();
const double *X = e_vec.Read();
double *Y = q_der.Write();
constexpr QVectorLayout L = QVectorLayout::byVDIM;
const int id = (vdim<<8) | (D1D<<4) | Q1D;
if (dim == 2)
{
switch (id)
{
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE, B, G, J, X, Y);
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE, B, G, J, X, Y);
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE, B, G, J, X, Y);
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE, B, G, J, X, Y);
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE, B, G, J, X, Y);
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE, B, G, J, X, Y);
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE, B, G, J, X, Y);
default:
{
constexpr int MD = MAX_D1D;
constexpr int MQ = MAX_Q1D;
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
<< " 1D points are not supported!");
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
return;
}
}
}
if (dim == 3)
{
switch (id)
{
case 0x134: return PhysGrad3D<L,1,3,4>(NE, B, G, J, X, Y);
case 0x146: return PhysGrad3D<L,1,4,6>(NE, B, G, J, X, Y);
case 0x158: return PhysGrad3D<L,1,5,8>(NE, B, G, J, X, Y);
case 0x334: return PhysGrad3D<L,3,3,4>(NE, B, G, J, X, Y);
case 0x346: return PhysGrad3D<L,3,4,6>(NE, B, G, J, X, Y);
case 0x358: return PhysGrad3D<L,3,5,8>(NE, B, G, J, X, Y);
default:
{
constexpr int MD = 8;
constexpr int MQ = 8;
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
<< " are not supported!");
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
<< " 1D points are not supported!");
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
return;
}
}
}
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
MFEM_ABORT("Unknown kernel");
}
} // namespace mfem
+52 -421
View File
@@ -17,6 +17,55 @@
namespace mfem
{
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
: ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
ndofs(fes.GetNDofs())
{
height = vdim*ne*ndof;
width = vdim*ne*ndof;
}
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
auto d_y = Reshape(y.Write(), nd, vd, ne);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
}
});
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), nd, vd, ne);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
}
});
}
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
ElementDofOrdering e_ordering)
: fes(f),
@@ -232,309 +281,7 @@ void ElementRestriction::BooleanMask(Vector& y) const
}
}
void ElementRestriction::FillSparseMatrix(const Vector &mat_ea,
SparseMatrix &mat) const
{
mat.GetMemoryI().New(mat.Height()+1, mat.GetMemoryI().GetMemoryType());
const int nnz = FillI(mat);
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
FillJAndData(mat_ea, mat);
}
template <int MaxNbNbr>
static MFEM_HOST_DEVICE int GetMinElt(const int *my_elts, const int nbElts,
const int *nbr_elts, const int nbrNbElts)
{
// Building the intersection
int inter[MaxNbNbr];
int cpt = 0;
for (int i = 0; i < nbElts; i++)
{
const int e_i = my_elts[i];
for (int j = 0; j < nbrNbElts; j++)
{
if (e_i==nbr_elts[j])
{
inter[cpt] = e_i;
cpt++;
}
}
}
// Finding the minimum
int min = inter[0];
for (int i = 1; i < cpt; i++)
{
if (inter[i] < min)
{
min = inter[i];
}
}
return min;
}
/** Returns the index where a non-zero entry should be added and increment the
number of non-zeros for the row i_L. */
static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
{
int ind = AtomicAdd(I[i_L],1);
return ind;
}
int ElementRestriction::FillI(SparseMatrix &mat) const
{
static constexpr int Max = MaxNbNbr;
const int all_dofs = ndofs;
const int vd = vdim;
const int elt_dofs = dof;
auto I = mat.ReadWriteI();
auto d_offsets = offsets.Read();
auto d_indices = indices.Read();
auto d_gatherMap = gatherMap.Read();
MFEM_FORALL(i_L, vd*all_dofs+1,
{
I[i_L] = 0;
});
MFEM_FORALL(e, ne,
{
for (int i = 0; i < elt_dofs; i++)
{
int i_elts[Max];
const int i_E = e*elt_dofs + i;
const int i_L = d_gatherMap[i_E];
const int i_offset = d_offsets[i_L];
const int i_nextOffset = d_offsets[i_L+1];
const int i_nbElts = i_nextOffset - i_offset;
for (int e_i = 0; e_i < i_nbElts; ++e_i)
{
const int i_E = d_indices[i_offset+e_i];
i_elts[e_i] = i_E/elt_dofs;
}
for (int j = 0; j < elt_dofs; j++)
{
const int j_E = e*elt_dofs + j;
const int j_L = d_gatherMap[j_E];
const int j_offset = d_offsets[j_L];
const int j_nextOffset = d_offsets[j_L+1];
const int j_nbElts = j_nextOffset - j_offset;
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
{
GetAndIncrementNnzIndex(i_L, I);
}
else // assembly required
{
int j_elts[Max];
for (int e_j = 0; e_j < j_nbElts; ++e_j)
{
const int j_E = d_indices[j_offset+e_j];
const int elt = j_E/elt_dofs;
j_elts[e_j] = elt;
}
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
if (e == min_e) // add the nnz only once
{
GetAndIncrementNnzIndex(i_L, I);
}
}
}
}
});
// We need to sum the entries of I, we do it on CPU as it is very sequential.
auto h_I = mat.HostReadWriteI();
const int nTdofs = vd*all_dofs;
int sum = 0;
for (int i = 0; i < nTdofs; i++)
{
const int nnz = h_I[i];
h_I[i] = sum;
sum+=nnz;
}
h_I[nTdofs] = sum;
// We return the number of nnz
return h_I[nTdofs];
}
void ElementRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat) const
{
static constexpr int Max = MaxNbNbr;
const int all_dofs = ndofs;
const int vd = vdim;
const int elt_dofs = dof;
auto I = mat.ReadWriteI();
auto J = mat.WriteJ();
auto Data = mat.WriteData();
auto d_offsets = offsets.Read();
auto d_indices = indices.Read();
auto d_gatherMap = gatherMap.Read();
auto mat_ea = Reshape(ea_data.Read(), elt_dofs, elt_dofs, ne);
MFEM_FORALL(e, ne,
{
for (int i = 0; i < elt_dofs; i++)
{
int i_elts[Max];
int i_B[Max];
const int i_E = e*elt_dofs + i;
const int i_L = d_gatherMap[i_E];
const int i_offset = d_offsets[i_L];
const int i_nextOffset = d_offsets[i_L+1];
const int i_nbElts = i_nextOffset - i_offset;
for (int e_i = 0; e_i < i_nbElts; ++e_i)
{
const int i_E = d_indices[i_offset+e_i];
i_elts[e_i] = i_E/elt_dofs;
i_B[e_i] = i_E%elt_dofs;
}
for (int j = 0; j < elt_dofs; j++)
{
const int j_E = e*elt_dofs + j;
const int j_L = d_gatherMap[j_E];
const int j_offset = d_offsets[j_L];
const int j_nextOffset = d_offsets[j_L+1];
const int j_nbElts = j_nextOffset - j_offset;
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
{
const int nnz = GetAndIncrementNnzIndex(i_L, I);
J[nnz] = j_L;
Data[nnz] = mat_ea(j,i,e);
}
else // assembly required
{
int j_elts[Max];
int j_B[Max];
for (int e_j = 0; e_j < j_nbElts; ++e_j)
{
const int j_E = d_indices[j_offset+e_j];
const int elt = j_E/elt_dofs;
j_elts[e_j] = elt;
j_B[e_j] = j_E%elt_dofs;
}
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
if (e == min_e) // add the nnz only once
{
double val = 0.0;
for (int i = 0; i < i_nbElts; i++)
{
const int e_i = i_elts[i];
const int i_Bloc = i_B[i];
for (int j = 0; j < j_nbElts; j++)
{
const int e_j = j_elts[j];
const int j_Bloc = j_B[j];
if (e_i == e_j)
{
val += mat_ea(j_Bloc, i_Bloc, e_i);
}
}
}
const int nnz = GetAndIncrementNnzIndex(i_L, I);
J[nnz] = j_L;
Data[nnz] = val;
}
}
}
}
});
// We need to shift again the entries of I, we do it on CPU as it is very
// sequential.
auto h_I = mat.HostReadWriteI();
const int size = vd*all_dofs;
for (int i = 0; i < size; i++)
{
h_I[size-i] = h_I[size-(i+1)];
}
h_I[0] = 0;
}
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
: ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
ndofs(fes.GetNDofs())
{
height = vdim*ne*ndof;
width = vdim*ne*ndof;
}
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
auto d_y = Reshape(y.Write(), nd, vd, ne);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
}
});
}
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
{
const int nd = ndof;
const int vd = vdim;
const bool t = byvdim;
auto d_x = Reshape(x.Read(), nd, vd, ne);
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
MFEM_FORALL(i, ndofs,
{
const int idx = i;
const int dof = idx % nd;
const int e = idx / nd;
for (int c = 0; c < vd; ++c)
{
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
}
});
}
void L2ElementRestriction::FillI(SparseMatrix &mat) const
{
const int elem_dofs = ndof;
const int vd = vdim;
auto I = mat.WriteI();
MFEM_FORALL(dof, ne*elem_dofs*vd,
{
I[dof] = elem_dofs;
});
}
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
{
int val = AtomicAdd(I[iE],dofs);
return val;
}
void L2ElementRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat) const
{
const int elem_dofs = ndof;
const int vd = vdim;
auto I = mat.ReadWriteI();
auto J = mat.WriteJ();
auto Data = mat.WriteData();
auto mat_ea = Reshape(ea_data.Read(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(iE, ne*elem_dofs*vd,
{
const int offset = AddNnz(iE,I,elem_dofs);
const int e = iE/elem_dofs;
const int i = iE%elem_dofs;
for (int j = 0; j < elem_dofs; j++)
{
J[offset+j] = e*elem_dofs+j;
Data[offset+j] = mat_ea(j,i,e);
}
});
}
// Return the face degrees of freedom returned in Lexicographic order.
/// Return the face degrees of freedom returned in Lexicographic order.
void GetFaceDofs(const int dim, const int face_id,
const int dof1d, Array<int> &faceMap)
{
@@ -953,7 +700,7 @@ static int PermuteFace3D(const int face_id1, const int face_id2,
return ToLexOrdering3D(face_id2, size1d, new_i, new_j);
}
// Permute dofs or quads on a face for e2 to match with the ordering of e1
/// Permute dofs or quads on a face for e2 to match with the ordering of e1
int PermuteFaceL2(const int dim, const int face_id1,
const int face_id2, const int orientation,
const int size1d, const int index)
@@ -973,32 +720,23 @@ int PermuteFaceL2(const int dim, const int face_id1,
}
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
const ElementDofOrdering e_ordering,
const FaceType type,
const L2FaceValues m)
: fes(fes),
nf(fes.GetNFbyType(type)),
ne(fes.GetNE()),
vdim(fes.GetVDim()),
byvdim(fes.GetOrdering() == Ordering::byVDIM),
ndofs(fes.GetNDofs()),
dof(nf > 0 ?
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
: 0),
elemDofs(fes.GetFE(0)->GetDof()),
m(m),
nfdofs(nf*dof),
scatter_indices1(nf*dof),
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
offsets(ndofs+1),
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
{
}
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
const ElementDofOrdering e_ordering,
const FaceType type,
const L2FaceValues m)
: L2FaceRestriction(fes, type, m)
{
// If fespace == L2
const FiniteElement *fe = fes.GetFE(0);
@@ -1296,113 +1034,6 @@ void L2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
}
}
void L2FaceRestriction::FillI(SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int iE1 = d_indices1[fdof];
const int iE2 = d_indices2[fdof];
AddNnz(iE1,I,face_dofs);
AddNnz(iE2,I,face_dofs);
});
}
void L2FaceRestriction::FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const
{
const int face_dofs = dof;
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto I = mat.ReadWriteI();
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
auto J = mat.WriteJ();
auto Data = mat.WriteData();
MFEM_FORALL(fdof, nf*face_dofs,
{
const int f = fdof/face_dofs;
const int iF = fdof%face_dofs;
const int iE1 = d_indices1[f*face_dofs+iF];
const int iE2 = d_indices2[f*face_dofs+iF];
const int offset1 = AddNnz(iE1,I,face_dofs);
const int offset2 = AddNnz(iE2,I,face_dofs);
for (int jF = 0; jF < face_dofs; jF++)
{
const int jE1 = d_indices1[f*face_dofs+jF];
const int jE2 = d_indices2[f*face_dofs+jF];
J[offset2+jF] = jE1;
J[offset1+jF] = jE2;
Data[offset2+jF] = mat_fea(jF,iF,0,f);
Data[offset1+jF] = mat_fea(jF,iF,1,f);
}
});
}
void L2FaceRestriction::AddFaceMatricesToElementMatrices(Vector &fea_data,
Vector &ea_data) const
{
const int face_dofs = dof;
const int elem_dofs = elemDofs;
const int NE = ne;
if (m==L2FaceValues::DoubleValued)
{
auto d_indices1 = scatter_indices1.Read();
auto d_indices2 = scatter_indices2.Read();
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, 2, nf);
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(f, nf,
{
const int e1 = d_indices1[f*face_dofs]/elem_dofs;
const int e2 = d_indices2[f*face_dofs]/elem_dofs;
for (int j = 0; j < face_dofs; j++)
{
const int jB1 = d_indices1[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iB1 = d_indices1[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iB1,jB1,e1), mat_fea(i,j,0,f));
}
}
if (e2 < NE)
{
for (int j = 0; j < face_dofs; j++)
{
const int jB2 = d_indices2[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iB2 = d_indices2[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iB2,jB2,e2), mat_fea(i,j,1,f));
}
}
}
});
}
else
{
auto d_indices = scatter_indices1.Read();
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, nf);
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
MFEM_FORALL(f, nf,
{
const int e = d_indices[f*face_dofs]/elem_dofs;
for (int j = 0; j < face_dofs; j++)
{
const int jE = d_indices[f*face_dofs+j]%elem_dofs;
for (int i = 0; i < face_dofs; i++)
{
const int iE = d_indices[f*face_dofs+i]%elem_dofs;
AtomicAdd(mat_ea(iE,jE,e), mat_fea(i,j,f));
}
}
});
}
}
int ToLexOrdering(const int dim, const int face_id, const int size1d,
const int index)
{
+1 -39
View File
@@ -30,11 +30,6 @@ enum class L2FaceValues : bool {SingleValued, DoubleValued};
objects, see FiniteElementSpace::GetElementRestriction(). */
class ElementRestriction : public Operator
{
private:
/** This number defines the maximum number of elements any dof can belong to
for the FillSparseMatrix method. */
static const int MaxNbNbr = 16;
protected:
const FiniteElementSpace &fes;
const int ne;
@@ -64,16 +59,6 @@ public:
emulate SetSubVector and its transpose on GPUs. This method is running on
the host, since the `processed` array requires a large shared memory. */
void BooleanMask(Vector& y) const;
/// Fill a Sparse Matrix with Element Matrices.
void FillSparseMatrix(const Vector &mat_ea, SparseMatrix &mat) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ElementRestriction. */
int FillI(SparseMatrix &mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this ElementRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
};
/// Operator that converts L2 FiniteElementSpace L-vectors to E-vectors.
@@ -92,12 +77,6 @@ public:
L2ElementRestriction(const FiniteElementSpace&);
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this ElementRestriction. */
void FillI(SparseMatrix &mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this L2FaceRestriction, and the values of ea_data. */
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
};
/// Operator that extracts Face degrees of freedom.
@@ -132,12 +111,10 @@ class L2FaceRestriction : public Operator
protected:
const FiniteElementSpace &fes;
const int nf;
const int ne;
const int vdim;
const bool byvdim;
const int ndofs;
const int dof;
const int elemDofs;
const L2FaceValues m;
const int nfdofs;
Array<int> scatter_indices1;
@@ -145,27 +122,12 @@ protected:
Array<int> offsets;
Array<int> gather_indices;
L2FaceRestriction(const FiniteElementSpace&,
const FaceType,
const L2FaceValues m = L2FaceValues::DoubleValued);
public:
L2FaceRestriction(const FiniteElementSpace&, const ElementDofOrdering,
const FaceType,
const L2FaceValues m = L2FaceValues::DoubleValued);
virtual void Mult(const Vector &x, Vector &y) const;
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
given by this L2FaceRestriction. */
virtual void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
pattern given by this L2FaceRestriction, and the values of ea_data. */
virtual void FillJAndData(const Vector &ea_data,
SparseMatrix &mat,
SparseMatrix &face_mat) const;
/// This methods adds the DG face matrices to the element matrices.
void AddFaceMatricesToElementMatrices(Vector &fea_data,
Vector &ea_data) const;
};
// Return the face degrees of freedom returned in Lexicographic order.
+75 -573
View File
@@ -13,7 +13,6 @@
#include "linearform.hpp"
#include "pgridfunc.hpp"
#include "tmop_tools.hpp"
#include "../general/forall.hpp"
namespace mfem
{
@@ -442,8 +441,8 @@ void TMOP_Metric_058::AssembleH(const DenseMatrix &Jpt,
double TMOP_Metric_077::EvalW(const DenseMatrix &Jpt) const
{
ie.SetJacobian(Jpt.GetData());
const double I2b = ie.Get_I2b();
return 0.5*(I2b*I2b + 1./(I2b*I2b) - 2.);
const double I2 = ie.Get_I2b();
return 0.5*(I2*I2 + 1./(I2*I2) - 2.);
}
void TMOP_Metric_077::EvalP(const DenseMatrix &Jpt, DenseMatrix &P) const
@@ -928,20 +927,9 @@ void TargetConstructor::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void TargetConstructor::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
// TODO: Compute derivative for targets with GIVEN_SHAPE or/and GIVEN_SIZE
for (int i = 0; i < Tpr.GetFE()->GetDim()*ir.GetNPoints(); i++) { dJtr(i) = 0.; }
}
void AnalyticAdaptTC::SetAnalyticTargetSpec(Coefficient *sspec,
VectorCoefficient *vspec,
TMOPMatrixCoefficient *mspec)
MatrixCoefficient *mspec)
{
scalar_tspec = sspec;
vector_tspec = vspec;
@@ -982,39 +970,6 @@ void AnalyticAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void AnalyticAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
const FiniteElement *fe = Tpr.GetFE();
DenseMatrix point_mat;
point_mat.UseExternalData(elfun.GetData(), fe->GetDof(), fe->GetDim());
switch (target_type)
{
case GIVEN_FULL:
{
MFEM_VERIFY(matrix_tspec != NULL,
"Target type GIVEN_FULL requires a TMOPMatrixCoefficient.");
for (int d = 0; d < fe->GetDim(); d++)
{
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
Tpr.SetIntPoint(&ip);
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
matrix_tspec->EvalGrad(dJtr_i, Tpr, ip, d);
}
}
break;
}
default:
MFEM_ABORT("Incompatible target type for analytic adaptation!");
}
}
#ifdef MFEM_USE_MPI
void DiscreteAdaptTC::FinalizeParDiscreteTargetSpec(const ParGridFunction
&tspec_)
@@ -1039,10 +994,11 @@ void DiscreteAdaptTC::SetTspecAtIndex(int idx, const ParGridFunction &tspec_)
{
const int vdim = tspec_.FESpace()->GetVDim(),
dof_cnt = tspec_.Size()/vdim;
const auto tspec__d = tspec_.Read();
auto tspec_d = tspec.ReadWrite();
const int offset = idx*dof_cnt;
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
for (int i = 0; i < dof_cnt*vdim; i++)
{
tspec(i+idx*dof_cnt) = tspec_(i);
}
FinalizeParDiscreteTargetSpec(tspec_);
}
@@ -1102,33 +1058,34 @@ void DiscreteAdaptTC::SetDiscreteTargetBase(const GridFunction &tspec_)
// make a copy of tspec->tspec_temp, increase its size, and
// copy data from tspec_temp -> tspec, then add new entries
Vector tspec_temp = tspec;
tspec.UseDevice(true);
tspec_sav.UseDevice(true);
tspec.SetSize(ncomp*dof_cnt);
const auto tspec_temp_d = tspec_temp.Read();
auto tspec_d = tspec.ReadWrite();
MFEM_FORALL(i, tspec_temp.Size(), tspec_d[i] = tspec_temp_d[i];);
for (int i = 0; i < tspec_temp.Size(); i++)
{
tspec(i) = tspec_temp(i);
}
const auto tspec__d = tspec_.Read();
const int offset = (ncomp-vdim)*dof_cnt;
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
for (int i = 0; i < dof_cnt*vdim; i++)
{
tspec(i+(ncomp-vdim)*dof_cnt) = tspec_(i);
}
}
void DiscreteAdaptTC::SetTspecAtIndex(int idx, const GridFunction &tspec_)
{
const int vdim = tspec_.FESpace()->GetVDim(),
dof_cnt = tspec_.Size()/vdim;
for (int i = 0; i < dof_cnt*vdim; i++)
{
tspec(i+idx*dof_cnt) = tspec_(i);
}
const auto tspec__d = tspec_.Read();
auto tspec_d = tspec.ReadWrite();
const int offset = idx*dof_cnt;
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
FinalizeSerialDiscreteTargetSpec();
}
void DiscreteAdaptTC::SetSerialDiscreteTargetSize(const GridFunction &tspec_)
{
if (sizeidx > -1) { SetTspecAtIndex(sizeidx, tspec_); return; }
sizeidx = ncomp;
SetDiscreteTargetBase(tspec_);
@@ -1211,7 +1168,7 @@ void DiscreteAdaptTC::UpdateTargetSpecificationAtNode(const FiniteElement &el,
Array<int> dofs;
tspec_fes->GetElementDofs(T.ElementNo, dofs);
const int cnt = tspec.Size()/ncomp; // dofs per scalar-field
const int cnt = tspec.Size()/ncomp; //dofs per scalar-field
for (int i = 0; i < ncomp; i++)
{
@@ -1239,9 +1196,6 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
DenseTensor &Jtr) const
{
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
const int dim = fe.GetDim(),
nqp = ir.GetNPoints();
Jtrcomp.SetSize(dim, dim, 4*nqp);
switch (target_type)
{
@@ -1251,44 +1205,36 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
const DenseMatrix &Wideal =
Geometries.GetGeomToPerfGeomJac(fe.GetGeomType());
const int dim = Wideal.Height(),
ndofs = tspec_fes->GetFE(e_id)->GetDof(),
ndofs = tspec_fes->GetFE(0)->GetDof(),
ntspec_dofs = ndofs*ncomp;
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
par_vals_c1, par_vals_c2, par_vals_c3;
Array<int> dofs;
DenseMatrix D_rho(dim), Q_phi(dim), R_theta(dim);
tspec_fesv->GetElementVDofs(e_id, dofs);
tspec.UseDevice(true);
tspec.GetSubVector(dofs, tspec_vals);
for (int q = 0; q < nqp; q++)
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
const IntegrationPoint &ip = ir.IntPoint(i);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
Jtr(i) = Wideal; //Initialize to identity
Jtr(q) = Wideal; // Initialize to identity
for (int d = 0; d < 4; d++)
{
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(d + 4*q), dim, dim);
Jtrcomp_q = Wideal; // Initialize to identity
}
if (sizeidx != -1) // Set size
if (sizeidx != -1) //Set size
{
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
const double min_size = par_vals.Min();
MFEM_VERIFY(min_size > 0.0,
"Non-positive size propagated in the target definition.");
const double size = std::max(shape * par_vals, min_size);
Jtr(q).Set(std::pow(size, 1.0/dim), Jtr(q));
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(0 + 4*q), dim, dim);
Jtrcomp_q = Jtr(q);
} // Done size
Jtr(i).Set(std::pow(size, 1.0/dim), Jtr(i));
} //Done size
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
if (aspectratioidx != -1) // Set aspect ratio
if (aspectratioidx != -1) //Set aspect ratio
{
if (dim == 2)
{
@@ -1316,13 +1262,12 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
D_rho(1,1) = pow(rho2,2./3.);
D_rho(2,2) = pow(rho3,2./3.);
}
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(1 + 4*q), dim, dim);
Jtrcomp_q = D_rho;
DenseMatrix Temp = Jtr(q);
Mult(D_rho, Temp, Jtr(q));
} // Done aspect ratio
if (skewidx != -1) // Set skew
DenseMatrix Temp = Jtr(i);
Mult(D_rho, Temp, Jtr(i));
} //Done aspect ratio
if (skewidx != -1) //Set skew
{
if (dim == 2)
{
@@ -1358,13 +1303,12 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
Q_phi(2,2) = sin(phi13)*sin(chi);
}
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*q), dim, dim);
Jtrcomp_q = Q_phi;
DenseMatrix Temp = Jtr(q);
Mult(Q_phi, Temp, Jtr(q));
} // Done skew
if (orientationidx != -1) // Set orientation
DenseMatrix Temp = Jtr(i);
Mult(Q_phi, Temp, Jtr(i));
} // done skew
if (orientationidx != -1) //Set orientation
{
if (dim == 2)
{
@@ -1389,28 +1333,33 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
const double psi = shape * par_vals_c2;
const double beta = shape * par_vals_c3;
DenseMatrix R_tp(dim), R_beta(dim), R_theta(dim);
double ct = cos(theta), st = sin(theta),
cp = cos(psi), sp = sin(psi),
cb = cos(beta), sb = sin(beta);
cp = cos(psi), sp = sin(psi);
R_tp(0,0) = ct*sp;
R_tp(1,0) = st*sp;
R_tp(2,0) = cp;
R_theta = 0.;
R_theta(0,0) = ct*sp;
R_theta(1,0) = st*sp;
R_theta(2,0) = cp;
R_tp(0,1) = -(ct*st*sp*sp)/(1+cp);
R_tp(1,1) = cp+(pow(ct,2.)*pow(sp,2.))/(1+cp);
R_tp(2,1) = -st*sp;
R_theta(0,1) = -st*cb + ct*cp*sb;
R_theta(1,1) = ct*cb + st*cp*sb;
R_theta(2,1) = -sp*sb;
R_tp(0,2) = -cp-(pow(st,2.)*pow(sp,2.))/(1+cp);
R_tp(1,2) = -R_tp(0,1);
R_tp(2,2) = ct*sp;
R_theta(0,0) = -st*sb - ct*cp*cb;
R_theta(1,0) = ct*sb - st*cp*cb;
R_theta(2,0) = sp*cb;
R_beta = 0.;
R_beta(0,0) = 1.;
R_beta(1,1) = cos(beta);
R_beta(1,2) = -sin(beta);
R_beta(2,1) = sin(beta);
R_beta(2,2) = cos(beta);
Mult(R_tp, R_beta, R_theta);
}
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(3 + 4*q), dim, dim);
Jtrcomp_q = R_theta;
DenseMatrix Temp = Jtr(q);
Mult(R_theta, Temp, Jtr(q));
} // Done orientation
DenseMatrix Temp = Jtr(i);
Mult(R_theta, Temp, Jtr(i));
} // done orientation
}
break;
}
@@ -1419,353 +1368,6 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
}
}
void DiscreteAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const
{
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
dJtr = 0.;
const int e_id = Tpr.ElementNo;
const FiniteElement *fe = Tpr.GetFE();
switch (target_type)
{
case IDEAL_SHAPE_GIVEN_SIZE:
case GIVEN_SHAPE_AND_SIZE:
{
const DenseMatrix &Wideal =
Geometries.GetGeomToPerfGeomJac(fe->GetGeomType());
const int dim = Wideal.Height(),
ndofs = fe->GetDof(),
ntspec_dofs = ndofs*ncomp;
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
par_vals_c1(ndofs), par_vals_c2(ndofs), par_vals_c3(ndofs);
Array<int> dofs;
DenseMatrix dD_rho(dim), dQ_phi(dim), dR_theta(dim);
DenseMatrix dQ_phi13(dim), dQ_phichi(dim); // dQ_phi is used for dQ/dphi12 in 3D
DenseMatrix dR_psi(dim), dR_beta(dim);
tspec_fesv->GetElementVDofs(e_id, dofs);
tspec.GetSubVector(dofs, tspec_vals);
DenseMatrix grad_e_c1(ndofs, dim),
grad_e_c2(ndofs, dim),
grad_e_c3(ndofs, dim);
Vector grad_ptr_c1(grad_e_c1.GetData(), ndofs*dim),
grad_ptr_c2(grad_e_c2.GetData(), ndofs*dim),
grad_ptr_c3(grad_e_c3.GetData(), ndofs*dim);
DenseMatrix grad_phys; // This will be (dof x dim, dof).
fe->ProjectGrad(*fe, Tpr, grad_phys);
for (int i = 0; i < ir.GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
DenseMatrix Jtrcomp_s(Jtrcomp.GetData(0 + 4*i), dim, dim); // size
DenseMatrix Jtrcomp_d(Jtrcomp.GetData(1 + 4*i), dim, dim); // aspect-ratio
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*i), dim, dim); // skew
DenseMatrix Jtrcomp_r(Jtrcomp.GetData(3 + 4*i), dim, dim); // orientation
DenseMatrix work1(dim), work2(dim), work3(dim);
if (sizeidx != -1) // Set size
{
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double min_size = par_vals.Min();
MFEM_VERIFY(min_size > 0.0,
"Non-positive size propagated in the target definition.");
const double size = std::max(shape * par_vals, min_size);
double dz_dsize = (1./dim)*pow(size, 1./dim - 1.);
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_r, work1, work2); // R*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = Wideal;
work1.Set(dz_dsize, work1); // dz/dsize
work1 *= grad_q(d); // dz/dsize*dsize/dx
AddMult(work1, work2, dJtr_i); // dz/dx*R*Q*D
}
} // Done size
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
if (aspectratioidx != -1) // Set aspect ratio
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
aspectratioidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double aspectratio = shape * par_vals;
dD_rho = 0.;
dD_rho(0,0) = -0.5*pow(aspectratio,-1.5);
dD_rho(1,1) = 0.5*pow(aspectratio,-0.5);
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
Mult(work1, Jtrcomp_q, work2); // z*R*Q
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dD_rho;
work1 *= grad_q(d); // work1 = dD/drho*drho/dx
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
}
}
else // 3D
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
aspectratioidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double rho1 = shape * par_vals_c1;
const double rho2 = shape * par_vals_c2;
const double rho3 = shape * par_vals_c3;
dD_rho = 0.;
dD_rho(0,0) = (2./3.)*pow(rho1,-1./3.);
dD_rho(1,1) = (2./3.)*pow(rho2,-1./3.);
dD_rho(2,2) = (2./3.)*pow(rho3,-1./3.);
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
Mult(work1, Jtrcomp_q, work2); // z*R*Q
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dD_rho;
work1(0,0) *= grad_q1(d);
work1(1,2) *= grad_q2(d);
work1(2,2) *= grad_q3(d);
// work1 = dD/dx = dD/drho1*drho1/dx + dD/drho2*drho2/dx
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
}
}
} // Done aspect ratio
if (skewidx != -1) // Set skew
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
skewidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double skew = shape * par_vals;
dQ_phi = 0.;
dQ_phi(0,0) = 1.;
dQ_phi(0,1) = -sin(skew);
dQ_phi(1,1) = cos(skew);
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dQ_phi;
work1 *= grad_q(d); // work1 = dQ/dphi*dphi/dx
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
}
}
else
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
skewidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double phi12 = shape * par_vals_c1;
const double phi13 = shape * par_vals_c2;
const double chi = shape * par_vals_c3;
dQ_phi = 0.;
dQ_phi(0,0) = 1.;
dQ_phi(0,1) = -sin(phi12);
dQ_phi(1,1) = cos(phi12);
dQ_phi13 = 0.;
dQ_phi13(0,2) = -sin(phi13);
dQ_phi13(1,2) = cos(phi13)*cos(chi);
dQ_phi13(2,2) = cos(phi13)*sin(chi);
dQ_phichi = 0.;
dQ_phichi(1,2) = -sin(phi13)*sin(chi);
dQ_phichi(2,2) = sin(phi13)*cos(chi);
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dQ_phi;
work1 *= grad_q1(d); // work1 = dQ/dphi12*dphi12/dx
work1.Add(grad_q2(d), dQ_phi13); // + dQ/dphi13*dphi13/dx
work1.Add(grad_q3(d), dQ_phichi); // + dQ/dchi*dchi/dx
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
}
}
} // Done skew
if (orientationidx != -1) // Set orientation
{
if (dim == 2)
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
orientationidx*ndofs, ndofs);
grad_phys.Mult(par_vals, grad_ptr_c1);
Vector grad_q(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q);
const double theta = shape * par_vals;
dR_theta(0,0) = -sin(theta);
dR_theta(0,1) = -cos(theta);
dR_theta(1,0) = cos(theta);
dR_theta(1,1) = -sin(theta);
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_s, work1, work2); // z*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dR_theta;
work1 *= grad_q(d); // work1 = dR/dtheta*dtheta/dx
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
}
}
else
{
par_vals.SetDataAndSize(tspec_vals.GetData()+
orientationidx*ndofs, ndofs*3);
par_vals_c1.SetData(par_vals.GetData());
par_vals_c2.SetData(par_vals.GetData()+ndofs);
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
grad_e_c1.MultTranspose(shape, grad_q1);
grad_e_c2.MultTranspose(shape, grad_q2);
grad_e_c3.MultTranspose(shape, grad_q3);
const double theta = shape * par_vals_c1;
const double psi = shape * par_vals_c2;
const double beta = shape * par_vals_c3;
const double ct = cos(theta), st = sin(theta),
cp = cos(psi), sp = sin(psi),
cb = cos(beta), sb = sin(beta);
dR_theta = 0.;
dR_theta(0,0) = -st*sp;
dR_theta(1,0) = ct*sp;
dR_theta(2,0) = 0;
dR_theta(0,1) = -ct*cb - st*cp*sb;
dR_theta(1,1) = -st*cb + ct*cp*sb;
dR_theta(2,1) = 0.;
dR_theta(0,0) = -ct*sb + st*cp*cb;
dR_theta(1,0) = -st*sb - ct*cp*cb;
dR_theta(2,0) = 0.;
dR_beta = 0.;
dR_beta(0,0) = 0.;
dR_beta(1,0) = 0.;
dR_beta(2,0) = 0.;
dR_beta(0,1) = st*sb + ct*cp*cb;
dR_beta(1,1) = -ct*sb + st*cp*cb;
dR_beta(2,1) = -sp*cb;
dR_beta(0,0) = -st*cb + ct*cp*sb;
dR_beta(1,0) = ct*cb + st*cp*sb;
dR_beta(2,0) = 0.;
dR_psi = 0.;
dR_psi(0,0) = ct*cp;
dR_psi(1,0) = st*cp;
dR_psi(2,0) = -sp;
dR_psi(0,1) = 0. - ct*sp*sb;
dR_psi(1,1) = 0. + st*sp*sb;
dR_psi(2,1) = -cp*sb;
dR_psi(0,0) = 0. + ct*sp*cb;
dR_psi(1,0) = 0. + st*sp*cb;
dR_psi(2,0) = cp*cb;
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
Mult(Jtrcomp_s, work1, work2); // z*Q*D
for (int d = 0; d < dim; d++)
{
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
work1 = dR_theta;
work1 *= grad_q1(d); // work1 = dR/dtheta*dtheta/dx
work1.Add(grad_q2(d), dR_psi); // +dR/dpsi*dpsi/dx
work1.Add(grad_q3(d), dR_beta); // +dR/dbeta*dbeta/dx
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
}
}
} // Done orientation
}
break;
}
default:
MFEM_ABORT("Incompatible target type for discrete adaptation!");
}
Jtrcomp.Clear();
}
void DiscreteAdaptTC::UpdateGradientTargetSpecification(const Vector &x,
const double dx,
bool use_flag)
@@ -1872,17 +1474,6 @@ void AdaptivityEvaluator::SetParMetaInfo(const ParMesh &m,
}
#endif
void AdaptivityEvaluator::ClearGeometricFactors()
{
#ifdef MFEM_USE_MPI
if (pmesh) pmesh->DeleteGeometricFactors();
if (pfes) pfes->GetParMesh()->DeleteGeometricFactors();
#else
if (mesh) mesh->DeleteGeometricFactors();
if (fes) fes->GetMesh()->DeleteGeometricFactors();
#endif
}
AdaptivityEvaluator::~AdaptivityEvaluator()
{
delete fes;
@@ -1910,7 +1501,6 @@ void TMOP_Integrator::EnableLimiting(const GridFunction &n0,
{
EnableLimiting(n0, w0, lfunc);
lim_dist = &dist;
if (PA.enabled) { EnableLimitingPA(n0); }
}
void TMOP_Integrator::EnableLimiting(const GridFunction &n0, Coefficient &w0,
TMOP_LimiterFunction *lfunc)
@@ -2056,8 +1646,7 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
PMatI.MultTranspose(shape, p);
pos0.MultTranspose(shape, p0);
val += lim_normal *
lim_func->Eval(p, p0, d_vals(i)) *
coeff0->Eval(*Tpr, ip);
lim_func->Eval(p, p0, d_vals(i)) * coeff0->Eval(*Tpr, ip);
}
if (adaptive_limiting)
@@ -2108,7 +1697,6 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
{
const int dof = el.GetDof(), dim = el.GetDim();
DenseMatrix Amat(dim), work1(dim), work2(dim);
DSh.SetSize(dof, dim);
DS.SetSize(dof, dim);
Jrt.SetSize(dim);
@@ -2124,15 +1712,14 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
elvect = 0.0;
Vector weights(nqp);
DenseTensor Jtr(dim, dim, nqp);
DenseTensor dJtr(dim, dim, dim*nqp);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
// Limited case.
DenseMatrix pos0;
Vector shape, p, p0, d_vals, grad;
shape.SetSize(dof);
if (coeff0)
{
shape.SetSize(dof);
p.SetSize(dim);
p0.SetSize(dim);
pos0.SetSize(dof, dim);
@@ -2152,7 +1739,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
// Define ref->physical transformation, when a Coefficient is specified.
IsoparametricTransformation *Tpr = NULL;
if (coeff1 || coeff0 || zeta || exact_action)
if (coeff1 || coeff0 || zeta)
{
Tpr = new IsoparametricTransformation;
Tpr->SetFE(&el);
@@ -2160,16 +1747,8 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
Tpr->ElementType = ElementTransformation::ELEMENT;
Tpr->Attribute = T.Attribute;
Tpr->GetPointMat().Transpose(PMatI); // PointMat = PMatI^T
if (exact_action)
{
targetC->ComputeElementTargetsGradient(*ir, elfun, *Tpr, dJtr);
}
}
Vector d_detW_dx(dim);
Vector d_Winv_dx(dim);
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir->IntPoint(q);
@@ -2188,44 +1767,13 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
if (coeff1) { weight_m *= coeff1->Eval(*Tpr, ip); }
P *= weight_m;
AddMultABt(DS, P, PMatO); // w_q det(W) dmu/dx : dA/dx Winv
AddMultABt(DS, P, PMatO);
if (exact_action)
{
el.CalcShape(ip, shape);
// Derivatives of adaptivity-based targets.
// First term: w_q d*(Det W)/dx * mu(T)
// d(Det W)/dx = det(W)*Tr[Winv*dW/dx]
DenseMatrix dwdx(dim);
for (int d = 0; d < dim; d++)
{
const DenseMatrix &dJtr_q = dJtr(q + d*ir->GetNPoints());
Mult(Jrt, dJtr_q, dwdx );
d_detW_dx(d) = dwdx.Trace();
}
d_detW_dx *= weight_m*metric->EvalW(Jpt); // *[w_q*det(W)]*mu(T)
// Second term: w_q det(W) dmu/dx : AdWinv/dx
// dWinv/dx = -Winv*dW/dx*Winv
MultAtB(PMatI, DSh, Amat);
for (int d = 0; d < dim; d++)
{
const DenseMatrix &dJtr_q = dJtr(q + d*nqp);
Mult(Jrt, dJtr_q, work1); // Winv*dw/dx
Mult(work1, Jrt, work2); // Winv*dw/dx*Winv
Mult(Amat, work2, work1); // A*Winv*dw/dx*Winv
MultAtB(P, work1, work2); // dmu/dT^T*A*Winv*dw/dx*Winv
d_Winv_dx(d) = work2.Trace(); // Tr[dmu/dT : AWinv*dw/dx*Winv]
}
d_Winv_dx *= -weight_m; // Include (-) factor as well
d_detW_dx += d_Winv_dx;
AddMultVWt(shape, d_detW_dx, PMatO);
}
// TODO: derivatives of adaptivity-based targets.
if (coeff0)
{
if (!exact_action) { el.CalcShape(ip, shape); }
el.CalcShape(ip, shape);
PMatI.MultTranspose(shape, p);
pos0.MultTranspose(shape, p0);
lim_func->Eval_d1(p, p0, d_vals(q), grad);
@@ -2666,8 +2214,7 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
Jpt.SetSize(dim);
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
const int nqp = ir->GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
DenseTensor Jtr(dim, dim, ir->GetNPoints());
metric_energy = 0.0;
lim_energy = 0.0;
@@ -2680,12 +2227,12 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
targetC->ComputeElementTargets(i, *fe, *ir, x_vals, Jtr);
for (int q = 0; q < nqp; q++)
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(q);
metric->SetTargetJacobian(Jtr(q));
CalcInverse(Jtr(q), Jrt);
const double weight = ip.weight * Jtr(q).Det();
const IntegrationPoint &ip = ir->IntPoint(i);
metric->SetTargetJacobian(Jtr(i));
CalcInverse(Jtr(i), Jrt);
const double weight = ip.weight * Jtr(i).Det();
fe->CalcDShape(ip, DSh);
MultAtB(PMatI, DSh, Jpr);
@@ -2737,8 +2284,6 @@ void TMOP_Integrator::ComputeMinJac(const Vector &x,
void TMOP_Integrator::UpdateAfterMeshChange(const Vector &new_x)
{
PA.setup_Jtr = false;
PA.setup_Grad = false;
// Update zeta if adaptive limiting is enabled.
if (zeta) { adapt_eval->ComputeAtNewPosition(new_x, *zeta); }
}
@@ -2894,49 +2439,6 @@ void TMOPComboIntegrator::ParEnableNormalization(const ParGridFunction &x)
}
#endif
void TMOPComboIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
for (int i = 0; i < tmopi.Size(); i++)
{
tmopi[i]->AssemblePA(fes);
}
}
void TMOPComboIntegrator::AssembleGradientDiagonalPA(const Vector &xe,
Vector &de) const
{
for (int i = 0; i < tmopi.Size(); i++)
{
tmopi[i]->AssembleGradientDiagonalPA(xe, de);
}
}
void TMOPComboIntegrator::AddMultPA(const Vector &xe, Vector &ye) const
{
for (int i = 0; i < tmopi.Size(); i++)
{
tmopi[i]->AddMultPA(xe, ye);
}
}
void TMOPComboIntegrator::AddMultGradPA(const Vector &xe, const Vector &re,
Vector &ce) const
{
for (int i = 0; i < tmopi.Size(); i++)
{
tmopi[i]->AddMultGradPA(xe, re, ce);
}
}
double TMOPComboIntegrator::GetGridFunctionEnergyPA(const Vector &xe) const
{
double energy = 0.0;
for (int i = 0; i < tmopi.Size(); i++)
{
energy += tmopi[i]->GetGridFunctionEnergyPA(xe);
}
return energy;
}
void InterpolateTMOP_QualityMetric(TMOP_QualityMetric &metric,
const TargetConstructor &tc,
+11 -178
View File
@@ -68,10 +68,6 @@ public:
*/
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const = 0;
/** @brief Return the metric ID.
*/
virtual int Id() const { return 0; }
};
@@ -89,8 +85,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 1; }
};
/// Skew metric, 2D.
@@ -182,8 +176,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 2; }
};
/// Shape & area, ideal barrier metric, 2D
@@ -200,8 +192,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 7; }
};
/// Shape & area metric, 2D
@@ -288,6 +278,7 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
};
/// Shape, ideal barrier metric, 2D
@@ -305,6 +296,7 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
};
/// Area, ideal barrier metric, 2D
@@ -322,7 +314,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 77; }
};
/// Shape & orientation metric, 2D.
@@ -409,8 +400,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 302; }
};
/// Shape, ideal barrier metric, 3D
@@ -427,8 +416,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 303; }
};
/// Volume metric, 3D
@@ -445,8 +432,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 315; }
};
/// Volume, ideal barrier metric, 3D
@@ -481,8 +466,6 @@ public:
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const double weight, DenseMatrix &A) const;
virtual int Id() const { return 321; }
};
/// Shifted barrier form of 3D metric 16 (volume, ideal barrier metric), 3D
@@ -606,8 +589,6 @@ public:
virtual void ComputeAtNewPosition(const Vector &new_nodes,
Vector &new_field) = 0;
void ClearGeometricFactors();
};
/** @brief Base class representing target-matrix construction algorithms for
@@ -683,14 +664,9 @@ public:
nodes are used by all target types except IDEAL_SHAPE_UNIT_SIZE. */
void SetNodes(const GridFunction &n) { nodes = &n; avg_volume = 0.0; }
/** @brief Get the nodes to be used in the target-matrix construction. */
const GridFunction *GetNodes() const { return nodes; }
/// Used by target type IDEAL_SHAPE_EQUAL_SIZE. The default volume scale is 1.
void SetVolumeScale(double vol_scale) { volume_scale = vol_scale; }
const TargetType &Type() const { return target_type; }
/// Checks if the target matrices contain non-trivial size specification.
virtual bool ContainsVolumeInfo() const;
@@ -701,35 +677,6 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
template<int DIM>
bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe = Vector()) const;
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe = Vector()) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
class TMOPMatrixCoefficient : public MatrixCoefficient
{
public:
explicit TMOPMatrixCoefficient(int dim) : MatrixCoefficient(dim, dim) { }
/** @brief Evaluate the derivative of the matrix coefficient with respect to
@a comp in the element described by @a T at the point @a ip, storing the
result in @a K. */
virtual void EvalGrad(DenseMatrix &K, ElementTransformation &T,
const IntegrationPoint &ip, int comp) = 0;
virtual ~TMOPMatrixCoefficient() { }
};
class AnalyticAdaptTC : public TargetConstructor
@@ -738,7 +685,7 @@ protected:
// Analytic target specification.
Coefficient *scalar_tspec;
VectorCoefficient *vector_tspec;
TMOPMatrixCoefficient *matrix_tspec;
MatrixCoefficient *matrix_tspec;
public:
AnalyticAdaptTC(TargetType ttype)
@@ -747,7 +694,7 @@ public:
virtual void SetAnalyticTargetSpec(Coefficient *sspec,
VectorCoefficient *vspec,
TMOPMatrixCoefficient *mspec);
MatrixCoefficient *mspec);
/** @brief Given an element and quadrature rule, computes ref->target
transformation Jacobians for each quadrature point in the element.
@@ -756,16 +703,6 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe = Vector()) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
#ifdef MFEM_USE_MPI
@@ -787,36 +724,25 @@ protected:
// eta1(x+h,y), eta2(x+h,y) ... etan(x+h,y), eta1(x,y+h), eta2(x,y+h) ...
// same for tspec_pert2h and tspec_pertmix.
// Components of Target Jacobian at each quadrature point of an element. This
// is required for computation of the derivative using chain rule.
mutable DenseTensor Jtrcomp;
// Note: do not use the Nodes of this space as they may not be on the
// positions corresponding to the values of tspec.
const FiniteElementSpace *tspec_fes;
const FiniteElementSpace *tspec_fesv;
// These flags can be used by outside functions to avoid recomputing the
// tspec and tspec_perth fields again on the same mesh.
// These flags can be used by outside functions to avoid recomputing
// the tspec and tspec_perth fields again on the same mesh.
bool good_tspec, good_tspec_grad, good_tspec_hess;
// Evaluation of the discrete target specification on different meshes.
// Owned.
AdaptivityEvaluator *adapt_eval;
// PA extension
struct { mutable Vector tspec_e; } PA;
void FinalizeSerialDiscreteTargetSpec();
#ifdef MFEM_USE_MPI
void FinalizeParDiscreteTargetSpec(const ParGridFunction &tspec_);
#endif
public: // MFEM_FORALL nvcc restriction that it must be public
void SetDiscreteTargetBase(const GridFunction &tspec_);
void SetTspecAtIndex(int idx, const GridFunction &tspec_);
void FinalizeSerialDiscreteTargetSpec();
#ifdef MFEM_USE_MPI
void SetTspecAtIndex(int idx, const ParGridFunction &tspec_);
void FinalizeParDiscreteTargetSpec(const ParGridFunction &tspec_);
#endif
public:
@@ -901,7 +827,6 @@ public:
const Vector &GetTspecPert1H() { return tspec_pert1h; }
const Vector &GetTspecPert2H() { return tspec_pert2h; }
const Vector &GetTspecPertMixH() { return tspec_pertmix; }
const FiniteElementSpace *GetTspecFesv() const { return tspec_fesv; }
/** @brief Given an element and quadrature rule, computes ref->target
transformation Jacobians for each quadrature point in the element.
@@ -912,16 +837,6 @@ public:
const IntegrationRule &ir,
const Vector &elfun,
DenseTensor &Jtr) const;
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe = Vector()) const;
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
const Vector &elfun,
IsoparametricTransformation &Tpr,
DenseTensor &dJtr) const;
};
class TMOPNewtonSolver;
@@ -974,9 +889,6 @@ protected:
// Specifies that ComputeElementTargets is being called by a FD function.
// It's used to skip terms that have exact derivative calculations.
bool fd_call_flag;
// Compute the exact action of the Integrator (includes derivative of the
// target with respect to spatial position)
bool exact_action;
Array <Vector *> ElemDer; //f'(x)
Array <Vector *> ElemPertEnergy; //f(x+h)
@@ -993,25 +905,10 @@ protected:
// output - the result of AssembleElementVector() (dof x dim).
DenseMatrix DSh, DS, Jrt, Jpr, Jpt, P, PMatI, PMatO;
// PA extension
struct
{
bool enabled;
int dim, ne, nq;
mutable DenseTensor Jtr;
mutable bool setup_Grad, setup_Jtr;
mutable Vector E, O, W, X0, H, C0, LD, H0;
const DofToQuad *maps;
const DofToQuad *maps_lim = nullptr;
const GeometricFactors *geom;
const FiniteElementSpace *fes;
const Operator *R;
const IntegrationRule *ir;
} PA;
void ComputeNormalizationEnergies(const GridFunction &x,
double &metric_energy, double &lim_energy);
void AssembleElementVectorExact(const FiniteElement &el,
ElementTransformation &T,
const Vector &elfun, Vector &elvect);
@@ -1081,8 +978,8 @@ public:
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
zeta_0(NULL), zeta(NULL), coeff_zeta(NULL), adapt_eval(NULL),
discr_tc(dynamic_cast<DiscreteAdaptTC *>(tc)),
fdflag(false), dxscale(1.0e3), fd_call_flag(false), exact_action(false)
{ PA.enabled = false; }
fdflag(false), dxscale(1.0e3), fd_call_flag(false)
{ }
~TMOP_Integrator();
@@ -1150,45 +1047,6 @@ public:
virtual void AssembleElementGrad(const FiniteElement &el,
ElementTransformation &T,
const Vector &elfun, DenseMatrix &elmat);
/// PA extension
void SetupGradPA(const Vector &xe) const;
void EnableLimitingPA(const GridFunction &n0);
void ComputeElementTargetsPA(const Vector &xe = Vector()) const;
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
double GetGridFunctionEnergyPA_2D(const Vector&) const;
double GetGridFunctionEnergyPA_C0_2D(const Vector&) const;
double GetGridFunctionEnergyPA_3D(const Vector&) const;
double GetGridFunctionEnergyPA_C0_3D(const Vector&) const;
virtual double GetGridFunctionEnergyPA(const Vector&) const;
using NonlinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
void AssembleDiagonalPA_2D(Vector&) const;
void AssembleDiagonalPA_3D(Vector&) const;
void AssembleDiagonalPA_C0_2D(Vector&) const;
void AssembleDiagonalPA_C0_3D(Vector&) const;
using NonlinearFormIntegrator::AddMultPA;
void AddMultPA_2D(const Vector&, Vector&) const;
void AddMultPA_3D(const Vector&, Vector&) const;
void AddMultPA_C0_2D(const Vector&, Vector&) const;
void AddMultPA_C0_3D(const Vector&, Vector&) const;
virtual void AddMultPA(const Vector&, Vector&) const;
using NonlinearFormIntegrator::AddMultGradPA;
void AddMultGradPA_2D(const Vector&, Vector&) const;
void AddMultGradPA_3D(const Vector&, const Vector&, Vector&) const;
void AddMultGradPA_C0_2D(const Vector&, const Vector&, Vector&) const;
void AddMultGradPA_C0_3D(const Vector&, const Vector&, Vector&) const;
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
void AssembleGradPA_2D(const Vector&) const;
void AssembleGradPA_3D(const Vector&) const;
void AssembleGradPA_C0_2D(const Vector&) const;
void AssembleGradPA_C0_3D(const Vector&) const;
DiscreteAdaptTC *GetDiscreteAdaptTC() const { return discr_tc; }
@@ -1208,20 +1066,6 @@ public:
void SetFDhScale(double _dxscale) { dxscale = _dxscale; }
bool GetFDFlag() const { return fdflag; }
double GetFDh() const { return dx; }
/** @brief Flag to control if exact action of Integration is effected. */
void SetExactActionFlag(bool flag_) { exact_action = flag_; }
void ReleaseTemporaryMemory()
{
if (PA.enabled)
{
PA.H.GetMemory().DeleteDevice();
PA.H0.GetMemory().DeleteDevice();
//PA.Jtr.GetMemory().DeleteDevice();
//PA.setup_Jtr = false;
}
}
};
class TMOPComboIntegrator : public NonlinearFormIntegrator
@@ -1270,17 +1114,6 @@ public:
#ifdef MFEM_USE_MPI
void ParEnableNormalization(const ParGridFunction &x);
#endif
/// PA extension
using NonlinearFormIntegrator::AssemblePA;
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
using NonlinearFormIntegrator::AddMultPA;
virtual void AddMultPA(const Vector&, Vector&) const;
using NonlinearFormIntegrator::AddMultGradPA;
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
virtual double GetGridFunctionEnergyPA(const Vector&) const;
};
/// Interpolates the @a metric's values at the nodes of @a metric_gf.
-363
View File
@@ -1,363 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "linearform.hpp"
#include "pgridfunc.hpp"
#include "tmop_tools.hpp"
#include "quadinterpolator.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
void TMOP_Integrator::SetupGradPA(const Vector &xe) const
{
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
PA.setup_Grad = true;
if (PA.dim == 2)
{
AssembleGradPA_2D(xe);
if (coeff0) { AssembleGradPA_C0_2D(xe); }
}
if (PA.dim == 3)
{
AssembleGradPA_3D(xe);
if (coeff0) { AssembleGradPA_C0_3D(xe); }
}
}
// We might come here w/o knowing that PA will be used.
// It is the case when EnableLimiting is called before the Setup => AssemblePA.
void TMOP_Integrator::EnableLimitingPA(const GridFunction &n0)
{
MFEM_VERIFY(PA.enabled, "EnableLimitingPA but PA is not enabled!");
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
// Nodes0
const FiniteElementSpace *n0_fes = n0.FESpace();
const Operator *n0_R = n0_fes->GetElementRestriction(ordering);
PA.X0.SetSize(n0_R->Height(), Device::GetMemoryType());
PA.X0.UseDevice(true);
n0_R->Mult(n0, PA.X0);
// Get the 1D maps for the distance FE space.
const IntegrationRule &ir = *EnergyIntegrationRule(*n0.FESpace()->GetFE(0));
PA.maps_lim =
&lim_dist->FESpace()->GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
// lim_dist & lim_func checks
MFEM_VERIFY(lim_dist, "No lim_dist!")
const FiniteElementSpace *ld_fes = lim_dist->FESpace();
const Operator *ld_R = ld_fes->GetElementRestriction(ordering);
MFEM_VERIFY(ld_R, "No lim_dist restriction operator found!");
PA.LD.SetSize(ld_R->Height(), Device::GetMemoryType());
PA.LD.UseDevice(true);
ld_R->Mult(*lim_dist, PA.LD);
// Only TMOP_QuadraticLimiter is supported
MFEM_VERIFY(lim_func, "No lim_func!")
MFEM_VERIFY(dynamic_cast<TMOP_QuadraticLimiter*>(lim_func),
"Only TMOP_QuadraticLimiter is supported");
}
bool TargetConstructor::ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe) const
{
MFEM_VERIFY(Jtr.SizeI() == Jtr.SizeJ() && Jtr.SizeI() > 1, "");
const int dim = Jtr.SizeI();
if (dim == 2) { return ComputeElementTargetsPA<2>(fes, ir, Jtr, xe); }
if (dim == 3) { return ComputeElementTargetsPA<3>(fes, ir, Jtr, xe); }
return false;
}
bool AnalyticAdaptTC::ComputeElementTargetsPA(const FiniteElementSpace *fes,
const IntegrationRule *ir,
DenseTensor &Jtr,
const Vector &xe) const
{
return false;
}
// Code paths leading to ComputeElementTargets:
// - GetElementEnergy(elfun) which is done through GetGridFunctionEnergyPA(x)
// - AssembleElementVectorExact(elfun)
// - AssembleElementGradExact(elfun)
// - EnableNormalization(x) -> ComputeNormalizationEnergies(x)
// - (AssembleElementVectorFD(elfun))
// - (AssembleElementGradFD(elfun))
// ============================================================================
// - TargetConstructor():
// - IDEAL_SHAPE_UNIT_SIZE: Wideal
// - IDEAL_SHAPE_EQUAL_SIZE: α * Wideal
// - IDEAL_SHAPE_GIVEN_SIZE: β * Wideal
// - GIVEN_SHAPE_AND_SIZE: β * Wideal
// - AnalyticAdaptTC(elfun):
// - GIVEN_FULL: matrix_tspec->Eval(Jtr(elfun))
// - DiscreteAdaptTC():
// - IDEAL_SHAPE_GIVEN_SIZE: size^{1.0/dim} * Jtr(i) (size)
// - GIVEN_SHAPE_AND_SIZE: Jtr(i) *= D_rho (ratio)
// Jtr(i) *= Q_phi (skew)
// Jtr(i) *= R_theta (orientation)
void TMOP_Integrator::ComputeElementTargetsPA(const Vector &xe) const
{
PA.setup_Jtr = false;
const FiniteElementSpace *fes = PA.fes;
const IntegrationRule *ir = EnergyIntegrationRule(*fes->GetFE(0));
const TargetConstructor::TargetType &target_type = targetC->Type();
const DiscreteAdaptTC *discr_tc = GetDiscreteAdaptTC();
// Skip when TargetConstructor needs the nodes but have not been set
const bool use_nodes =
target_type == TargetConstructor::IDEAL_SHAPE_EQUAL_SIZE ||
target_type == TargetConstructor::IDEAL_SHAPE_GIVEN_SIZE ||
target_type == TargetConstructor::GIVEN_SHAPE_AND_SIZE;
if (targetC && !discr_tc && use_nodes && !targetC->GetNodes()) { return; }
// Try to use the TargetConstructor ComputeElementTargetsPA
PA.setup_Jtr = targetC->ComputeElementTargetsPA(fes, ir, PA.Jtr);
if (PA.setup_Jtr) { return; }
// Defaulting to host version
PA.Jtr.HostWrite();
const int NE = PA.ne;
const int NQ = PA.nq;
const int dim = PA.dim;
DenseTensor &Jtr = PA.Jtr;
Vector x;
const bool useable_input_vector = xe.Size() > 0;
const bool use_input_vector = target_type == TargetConstructor::GIVEN_FULL;
if (use_input_vector && !useable_input_vector) { return; }
if (discr_tc && !discr_tc->GetTspecFesv()) { return; }
if (use_input_vector)
{
x.SetSize(PA.R->Width(), Device::GetMemoryType());
x.UseDevice(true);
PA.R->MultTranspose(xe, x);
// Scale by weights
const int N = PA.W.Size();
const auto W = Reshape(PA.W.Read(), N);
auto X = Reshape(x.ReadWrite(), N);
MFEM_FORALL(i, N, X(i) /= W(i););
}
// Use TargetConstructor::ComputeElementTargets to fill the PA.Jtr
Vector elfun;
Array<int> vdofs;
DenseTensor J;
for (int e = 0; e < NE; e++)
{
const FiniteElement &fe = *fes->GetFE(e);
if (use_input_vector)
{
fes->GetElementVDofs(e, vdofs);
x.GetSubVector(vdofs, elfun);
}
J.UseExternalData(Jtr(e*NQ).Data(), dim, dim, NQ);
targetC->ComputeElementTargets(e, fe, *ir, elfun, J);
}
PA.setup_Jtr = true;
}
void TMOP_Integrator::AssemblePA(const FiniteElementSpace &fes)
{
PA.enabled = true;
MFEM_ASSERT(fes.GetMesh()->GetNE() > 0, "");
PA.ir = EnergyIntegrationRule(*fes.GetFE(0));
const IntegrationRule *ir = PA.ir;
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
"PA Only supports Ordering::byNODES!");
PA.fes = &fes;
Mesh *mesh = fes.GetMesh();
const int nq = PA.nq = ir->GetNPoints();
const int ne = PA.ne = fes.GetMesh()->GetNE();
const int dim = PA.dim = mesh->Dimension();
MFEM_VERIFY(PA.dim == 2 || PA.dim == 3, "Not yet implemented!");
const DofToQuad::Mode mode = DofToQuad::TENSOR;
PA.maps = &fes.GetFE(0)->GetDofToQuad(*ir, mode);
PA.geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
// Energy vector
PA.E.UseDevice(true);
PA.E.SetSize(ne*nq, Device::GetDeviceMemoryType());
// Setup initialization
PA.setup_Jtr = false;
PA.setup_Grad = false;
#ifdef MFEM_USE_UMPIRE
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
#else
const MemoryType temp_type = Device::GetDeviceMemoryType();
#endif
// H for Grad
PA.H.SetSize(dim*dim * dim*dim * nq*ne, temp_type);
// H0 for coeff0
PA.H0.SetSize(dim * dim * nq*ne, temp_type);
// Restriction setup
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
PA.R = fes.GetElementRestriction(ordering);
MFEM_VERIFY(PA.R, "Not yet implemented!");
// Weight of the R^t
PA.W.SetSize(PA.R->Width(), Device::GetDeviceMemoryType());
PA.W.UseDevice(true);
PA.O.SetSize(dim*ne*nq, Device::GetDeviceMemoryType());
PA.O.UseDevice(true);
PA.O = 1.0;
PA.R->MultTranspose(PA.O, PA.W);
// Scalar vector of '1'
PA.O.SetSize(ne*nq, Device::GetDeviceMemoryType());
PA.O = 1.0;
// TargetConstructor TargetType setup
PA.Jtr.SetSize(dim, dim, PA.ne*PA.nq);//, temp_type);
ComputeElementTargetsPA();
// Coeff0 PA.C0
PA.C0.UseDevice(true);
if (coeff0 == nullptr)
{
PA.C0.SetSize(1, Device::GetMemoryType());
PA.C0.HostWrite();
PA.C0(0) = 0.0;
}
else if (ConstantCoefficient* cQ =
dynamic_cast<ConstantCoefficient*>(coeff0))
{
PA.C0.SetSize(1, Device::GetMemoryType());
PA.C0.HostWrite();
PA.C0(0) = cQ->constant;
}
else
{
PA.C0.SetSize(PA.nq * PA.ne, Device::GetMemoryType());
auto C0 = Reshape(PA.C0.HostWrite(), PA.nq, PA.ne);
for (int e = 0; e < ne; ++e)
{
ElementTransformation& T = *fes.GetElementTransformation(e);
for (int q = 0; q < nq; ++q)
{
C0(q,e) = coeff0->Eval(T, ir->IntPoint(q));
}
}
}
if (coeff0)
{
MFEM_VERIFY(nodes0, "nodes0 has not been set!");
EnableLimitingPA(*nodes0);
}
}
void TMOP_Integrator::AssembleGradientDiagonalPA(const Vector &xe,
Vector &de) const
{
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
if (!PA.setup_Grad) { SetupGradPA(xe); }
if (PA.dim == 2)
{
AssembleDiagonalPA_2D(de);
if (coeff0) { AssembleDiagonalPA_C0_2D(de); }
}
else if (PA.dim == 3)
{
AssembleDiagonalPA_3D(de);
if (coeff0) { AssembleDiagonalPA_C0_3D(de); }
}
else
{
MFEM_ABORT("3D diagonal computation is WIP.");
}
}
void TMOP_Integrator::AddMultPA(const Vector &xe, Vector &ye) const
{
if (!PA.setup_Jtr) { ComputeElementTargetsPA(); }
if (PA.dim == 2)
{
AddMultPA_2D(xe,ye);
if (coeff0) { AddMultPA_C0_2D(xe,ye); }
}
if (PA.dim == 3)
{
AddMultPA_3D(xe,ye);
if (coeff0) { AddMultPA_C0_3D(xe,ye); }
}
}
void TMOP_Integrator::AddMultGradPA(const Vector &xe,
const Vector &re, Vector &ce) const
{
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
if (!PA.setup_Grad) { SetupGradPA(xe); }
if (PA.dim == 2)
{
AddMultGradPA_2D(re,ce);
if (coeff0) { AddMultGradPA_C0_2D(xe,re,ce); }
}
if (PA.dim == 3)
{
AddMultGradPA_3D(xe,re,ce);
if (coeff0) { AddMultGradPA_C0_3D(xe,re,ce); }
}
}
double TMOP_Integrator::GetGridFunctionEnergyPA(const Vector &xe) const
{
double energy = 0.0;
ComputeElementTargetsPA(xe);
if (PA.dim == 2)
{
energy = GetGridFunctionEnergyPA_2D(xe);
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_2D(xe); }
}
if (PA.dim == 3)
{
energy = GetGridFunctionEnergyPA_3D(xe);
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_3D(xe); }
}
return energy;
}
} // namespace mfem
-143
View File
@@ -1,143 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_TMOP_PA_HPP
#define MFEM_TMOP_PA_HPP
#include "../config/config.hpp"
#include "../linalg/dtensor.hpp"
#include "../fem/kernels.hpp"
#include <unordered_map>
namespace mfem
{
namespace kernels
{
/// Generic emplace
template<typename K, const int N,
typename Key_t = typename K::Key_t,
typename Kernel_t = typename K::Kernel_t>
void emplace(std::unordered_map<Key_t, Kernel_t> &map)
{
constexpr Key_t key = K::template GetKey<N>();
constexpr Kernel_t value = K::template GetValue<key>();
map.emplace(key, value);
}
/// Instances
template<class K, typename T, T... idx>
struct instances
{
static void Fill(std::unordered_map<typename K::Key_t,
typename K::Kernel_t> &map)
{
using unused = int[];
(void) unused {0, (emplace<K,idx>(map), 0)... };
}
};
/// Cat instances
template<class K, typename Offset, typename Lhs, typename Rhs> struct cat;
template<class K, typename T, T Offset, T... Lhs, T... Rhs>
struct cat<K, std::integral_constant<T, Offset>,
instances<K, T, Lhs...>,
instances<K, T, Rhs...> >
{ using type = instances<K, T, Lhs..., (Offset + Rhs)...>; };
/// Sequence, empty and one element terminal cases
template<class K, typename T, typename N>
struct sequence
{
using Lhs = std::integral_constant<T, N::value/2>;
using Rhs = std::integral_constant<T, N::value-Lhs::value>;
using type = typename cat<K, Lhs,
typename sequence<K, T, Lhs>::type,
typename sequence<K, T, Rhs>::type>::type;
};
template<class K, typename T>
struct sequence<K, T, std::integral_constant<T,0> >
{ using type = instances<K,T>; };
template<class K, typename T>
struct sequence<K, T, std::integral_constant<T,1> >
{ using type = instances<K,T,0>; };
/// Make_sequence
template<class Instance, typename T = typename Instance::Key_t>
using make_sequence =
typename sequence<Instance, T, std::integral_constant<T,Instance::N> >::type;
/// Instantiator class
template<class Instance,
typename Key_t = typename Instance::Key_t,
typename Return_t = typename Instance::Return_t,
typename Kernel_t = typename Instance::Kernel_t>
class Instantiator
{
private:
using map_t = std::unordered_map<Key_t, Kernel_t>;
map_t map;
public:
Instantiator() { make_sequence<Instance>().Fill(map); }
bool Find(const Key_t id)
{
return (map.find(id) != map.end()) ? true : false;
}
Kernel_t At(const Key_t id) { return map.at(id); }
};
/// MFEM_REGISTER_TMOP_KERNELS macro:
/// - forward declaration of the kernel
/// - kernel pointer declaration
/// - struct K##name##_T definition
/// - Instantiator definition
/// - re-use kernel return type and name before its body
#define MFEM_REGISTER_TMOP_KERNELS(return_t, kernel, ...) \
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0> \
return_t kernel(__VA_ARGS__);\
typedef return_t (*kernel##_p)(__VA_ARGS__);\
struct K##kernel##_T {\
static const int N = 14;\
using Key_t = std::size_t;\
using Kernel_t = kernel##_p;\
using Return_t = return_t;\
template<Key_t I> static constexpr Key_t GetKey() noexcept { return \
I==0 ? 0x22 : I==1 ? 0x23 : I==2 ? 0x24 : I==3 ? 0x25 : I==4 ? 0x26 :\
I==5 ? 0x33 : I==6 ? 0x34 : I==7 ? 0x35 : I==8 ? 0x36 :\
I==9 ? 0x44 : I==10 ? 0x45 : I==11 ? 0x46 :\
I==12 ? 0x55 : I==13 ? 0x56 : 0; }\
template<Key_t ID> static constexpr Kernel_t GetValue() noexcept\
{ return &kernel<(ID>>4)&0xF, ID&0xF>; }\
};\
static kernels::Instantiator<K##kernel##_T> K##kernel;\
template<int T_D1D, int T_Q1D, int T_MAX> return_t kernel(__VA_ARGS__)
/// MFEM_LAUNCH_TMOP_KERNEL macro
#define MFEM_LAUNCH_TMOP_KERNEL(kernel, id, ...)\
if (K##kernel.Find(id)) { return K##kernel.At(id)(__VA_ARGS__,0,0); }\
else {\
constexpr int T_MAX = 4;\
MFEM_VERIFY(D1D <= MAX_D1D && Q1D <= MAX_Q1D, "Max size error!");\
return kernel<0,0,T_MAX>(__VA_ARGS__,D1D,Q1D); }
} // namespace kernels
} // namespace mfem
#endif // MFEM_TMOP_PA_HPP
-161
View File
@@ -1,161 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
/* // Original i-j assembly (old invariants code).
for (int e = 0; e < NE; e++)
{
for (int q = 0; q < nqp; q++)
{
el.CalcDShape(ip, DSh);
Mult(DSh, Jrt, DS);
for (int i = 0; i < dof; i++)
{
for (int j = 0; j < dof; j++)
{
for (int r = 0; r < dim; r++)
{
for (int c = 0; c < dim; c++)
{
for (int rr = 0; rr < dim; rr++)
{
for (int cc = 0; cc < dim; cc++)
{
const double H = h(r, c, rr, cc);
A(e, i + r*dof, j + rr*dof) +=
weight_q * DS(i, c) * DS(j, cc) * H;
}
}
}
}
}
}
}
}*/
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_2D,
const int NE,
const Array<double> &b,
const Array<double> &g,
const DenseTensor &j,
const Vector &h,
Vector &diagonal,
const int d1d,
const int q1d)
{
constexpr int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto G = Reshape(g.Read(), Q1D, D1D);
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, NE);
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
{
constexpr int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double qd[DIM*DIM*MQ1*MD1];
DeviceTensor<4,double> QD(qd, DIM, DIM, MQ1, MD1);
for (int v = 0; v < DIM; v++)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD(0,0,qx,dy) = 0.0;
QD(0,1,qx,dy) = 0.0;
QD(1,0,qx,dy) = 0.0;
QD(1,1,qx,dy) = 0.0;
MFEM_UNROLL(MQ1);
for (int qy = 0; qy < Q1D; ++qy)
{
const double *Jtr = &J(0,0,qx,qy,e);
// Jrt = Jtr^{-1}
double j[4];
ConstDeviceMatrix Jrt(j,2,2);
kernels::CalcInverse<2>(Jtr, j);
const double gg = G(qy,dy) * G(qy,dy);
const double gb = G(qy,dy) * B(qy,dy);
const double bb = B(qy,dy) * B(qy,dy);
const double bgb[4] = { bb, gb, gb, gg };
ConstDeviceMatrix BG(bgb,2,2);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
const double Jij = Jrt(i,i) * Jrt(j,j);
const double alpha = Jij * BG(i,j);
QD(i,j,qx,dy) += alpha * H(v,i,v,j,qx,qy,e);
}
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double d = 0.0;
MFEM_UNROLL(MQ1);
for (int qx = 0; qx < Q1D; ++qx)
{
const double gg = G(qx,dx) * G(qx,dx);
const double gb = G(qx,dx) * B(qx,dx);
const double bb = B(qx,dx) * B(qx,dx);
d += gg * QD(0,0,qx,dy);
d += gb * QD(0,1,qx,dy);
d += gb * QD(1,0,qx,dy);
d += bb * QD(1,1,qx,dy);
}
D(dx,dy,v,e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &D) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const DenseTensor &J = PA.Jtr;
const Array<double> &B = PA.maps->B;
const Array<double> &G = PA.maps->G;
const Vector &H = PA.H;
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_2D,id,N,B,G,J,H,D);
}
} // namespace mfem
-96
View File
@@ -1,96 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_2D,
const int NE,
const Array<double> &b,
const Vector &h0,
Vector &diagonal,
const int d1d,
const int q1d)
{
constexpr int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, NE);
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
{
constexpr int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double qd[MQ1*MD1];
DeviceTensor<2,double> QD(qd, MQ1, MD1);
for (int v = 0; v < DIM; v++)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QD(qx,dy) = 0.0;
MFEM_UNROLL(MQ1);
for (int qy = 0; qy < Q1D; ++qy)
{
const double bb = B(qy,dy) * B(qy,dy);
QD(qx,dy) += bb * H0(v,v,qx,qy,e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double d = 0.0;
MFEM_UNROLL(MQ1);
for (int qx = 0; qx < Q1D; ++qx)
{
const double bb = B(qx,dx) * B(qx,dx);
d += bb * QD(qx,dy);
}
D(dx,dy,v,e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &D) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const Array<double> &B = PA.maps->B;
const Vector &H0 = PA.H0;
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_2D,id,N,B,H0,D);
}
} // namespace mfem
-128
View File
@@ -1,128 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_2D,
const int NE,
const Array<double> &b_,
const Array<double> &g_,
const DenseTensor &j_,
const Vector &h_,
const Vector &x_,
Vector &y_,
const int d1d,
const int q1d)
{
constexpr int DIM = 2;
constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto g = Reshape(g_.Read(), Q1D, D1D);
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
const auto H = Reshape(h_.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, DIM, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
MFEM_SHARED double BG[2][MQ1*MD1];
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,b,g,BG);
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const double *Jtr = &J(0,0,qx,qy,e);
// Jrt = Jtr^{-1}
double Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X^T.DSh
double Jpr[4];
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,Jpr);
// Jpt = Jpr . Jrt
double Jpt[4];
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
// B = Jpt : H
double B[4];
DeviceMatrix M(B,2,2);
ConstDeviceMatrix J(Jpt,2,2);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
M(i,j) = 0.0;
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
M(i,j) += H(r,c,i,j,qx,qy,e) * J(r,c);
}
}
}
}
// C = Jrt . B
double C[4];
kernels::MultABt(2,2,2, Jrt, B, C);
// Overwrite QQ = Jrt . (Jpt : H)^t
kernels::PushGrad<MQ1,NBZ>(qx,qy, C, QQ);
}
}
MFEM_SYNC_THREAD;
kernels::LoadBGt<MD1,MQ1>(D1D,Q1D,b,g,BG);
kernels::GradYt<MD1,MQ1,NBZ>(D1D,Q1D,BG,QQ,DQ);
kernels::GradXt<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,Y,e);
});
}
void TMOP_Integrator::AddMultGradPA_2D(const Vector &R, Vector &C) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const DenseTensor &J = PA.Jtr;
const Array<double> &B = PA.maps->B;
const Array<double> &G = PA.maps->G;
const Vector &H = PA.H;
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_2D,id,N,B,G,J,H,R,C);
}
} // namespace mfem
-107
View File
@@ -1,107 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "linearform.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_C0_2D,
const int NE,
const Array<double> &b_,
const Vector &h0_,
const Vector &r_,
Vector &c_,
const int d1d,
const int q1d)
{
constexpr int DIM = 2;
constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto H0 = Reshape(h0_.Read(), DIM, DIM, Q1D, Q1D, NE);
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto R = Reshape(r_.Read(), D1D, D1D, DIM, NE);
auto Y = Reshape(c_.ReadWrite(), D1D, D1D, DIM, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
constexpr int DIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
MFEM_SHARED double B[MQ1*MD1];
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
MFEM_SHARED double DQ[2][NBZ][MD1*MQ1];
MFEM_SHARED double QQ[2][NBZ][MQ1*MQ1];
kernels::LoadX<MD1,NBZ>(e,D1D,R,XY);
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,XY,DQ);
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
// Xh = X^T . Sh
double Xh[2];
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,Xh);
double B[4];
DeviceMatrix H(B,2,2);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
H(i,j) = H0(i,j,qx,qy,e);
}
}
// p2 = B . Xh
double p2[2];
kernels::Mult(2,2,B,Xh,p2);
kernels::PushEval<MQ1,NBZ>(qx,qy,p2,QQ);
}
}
MFEM_SYNC_THREAD;
kernels::LoadBt<MD1,MQ1>(D1D,Q1D,b,B);
kernels::EvalXt<MD1,MQ1,NBZ>(D1D,Q1D,B,QQ,DQ);
kernels::EvalYt<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,Y,e);
});
}
void TMOP_Integrator::AddMultGradPA_C0_2D(const Vector &X, const Vector &R,
Vector &C) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const Array<double> &B = PA.maps->B;
const Vector &H0 = PA.H0;
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_C0_2D,id,N,B,H0,R,C);
}
} // namespace mfem
-247
View File
@@ -1,247 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
#include "../linalg/dinvariants.hpp"
namespace mfem
{
using Args = kernels::InvariantsEvaluator2D::Buffers;
// weight * ddI1
static MFEM_HOST_DEVICE inline
void EvalH_001(const int e, const int qx, const int qy,
const double weight, const double *Jpt,
DeviceTensor<7,double> H)
{
constexpr int DIM = 2;
double ddI1[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const double h = ddi1(r,c);
H(r,c,i,j,qx,qy,e) = weight * h;
}
}
}
}
}
// 0.5 * weight * dI1b
static MFEM_HOST_DEVICE inline
void EvalH_002(const int e, const int qx, const int qy,
const double weight, const double *Jpt,
DeviceTensor<7,double> H)
{
constexpr int DIM = 2;
double ddI1[4], ddI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args()
.J(Jpt)
.ddI1(ddI1)
.ddI1b(ddI1b)
.dI2b(dI2b));
const double w = 0.5 * weight;
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i,j),DIM,DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const double h = ddi1b(r,c);
H(r,c,i,j,qx,qy,e) = w * h;
}
}
}
}
}
static MFEM_HOST_DEVICE inline
void EvalH_007(const int e, const int qx, const int qy,
const double weight, const double *Jpt,
DeviceTensor<7,double> H)
{
constexpr int DIM = 2;
double ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args()
.J(Jpt)
.ddI1(ddI1)
.ddI2(ddI2)
.dI1(dI1)
.dI2(dI2)
.dI2b(dI2b));
const double c1 = 1./ie.Get_I2();
const double c2 = weight*c1*c1;
const double c3 = ie.Get_I1()*c2;
ConstDeviceMatrix di1(ie.Get_dI1(),DIM,DIM);
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r,c,i,j,qx,qy,e) =
weight * (1.0 + c1) * ddi1(r,c)
- c3 * ddi2(r,c)
- c2 * ( di1(i,j) * di2(r,c) + di2(i,j) * di1(r,c) )
+ 2.0 * c1 * c3 * di2(r,c) * di2(i,j);
}
}
}
}
}
static MFEM_HOST_DEVICE inline
void EvalH_077(const int e, const int qx, const int qy,
const double weight, const double *Jpt,
DeviceTensor<7,double> H)
{
constexpr int DIM = 2;
double dI2[4], dI2b[4], ddI2[4];
kernels::InvariantsEvaluator2D ie(Args()
.J(Jpt)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2));
const double I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r,c,i,j,qx,qy,e) =
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r,c)
+ weight * (I2inv_sq / I2) * di2(r,c) * di2(i,j);
}
}
}
}
}
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_2D,
const Vector &x_,
const double metric_normal,
const int mid,
const int NE,
const Array<double> &w_,
const Array<double> &b_,
const Array<double> &g_,
const DenseTensor &j_,
Vector &h_,
const int d1d,
const int q1d)
{
MFEM_VERIFY(mid == 1 || mid == 2 || mid == 7 || mid == 77,
"Metric not yet implemented!");
constexpr int DIM = 2;
constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto W = Reshape(w_.Read(), Q1D, Q1D);
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto g = Reshape(g_.Read(), Q1D, D1D);
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
auto H = Reshape(h_.Write(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
MFEM_SHARED double s_BG[2][MQ1*MD1];
MFEM_SHARED double s_X[2][NBZ][MD1*MD1];
MFEM_SHARED double s_DQ[4][NBZ][MD1*MQ1];
MFEM_SHARED double s_QQ[4][NBZ][MQ1*MQ1];
kernels::LoadX<MD1,NBZ>(e,D1D,X,s_X);
kernels::LoadBG<MD1,MQ1>(D1D, Q1D, b, g, s_BG);
kernels::GradX<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_X, s_DQ);
kernels::GradY<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_DQ, s_QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const double *Jtr = &J(0,0,qx,qy,e);
const double detJtr = kernels::Det<2>(Jtr);
const double weight = metric_normal * W(qx,qy) * detJtr;
// Jrt = Jtr^{-1}
double Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X^t.DSh
double Jpr[4];
kernels::PullGrad<MQ1,NBZ>(qx,qy,s_QQ,Jpr);
// Jpt = Jpr.Jrt
double Jpt[4];
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
// metric->AssembleH
if (mid == 1) { EvalH_001(e,qx,qy,weight,Jpt,H); }
if (mid == 2) { EvalH_002(e,qx,qy,weight,Jpt,H); }
if (mid == 7) { EvalH_007(e,qx,qy,weight,Jpt,H); }
if (mid == 77) { EvalH_077(e,qx,qy,weight,Jpt,H); }
} // qx
} // qy
});
}
void TMOP_Integrator::AssembleGradPA_2D(const Vector &X) const
{
const int N = PA.ne;
const int M = metric->Id();
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const double mn = metric_normal;
const DenseTensor &J = PA.Jtr;
const Array<double> &W = PA.ir->GetWeights();
const Array<double> &B = PA.maps->B;
const Array<double> &G = PA.maps->G;
Vector &H = PA.H;
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_2D,id,X,mn,M,N,W,B,G,J,H);
}
} // namespace mfem
-125
View File
@@ -1,125 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "linearform.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_C0_2D,
const double lim_normal,
const Vector &lim_dist,
const Vector &c0_,
const int NE,
const DenseTensor &j_,
const Array<double> &w_,
const Array<double> &b_,
const Array<double> &bld_,
Vector &h0_,
const int d1d,
const int q1d)
{
constexpr int DIM = 2;
constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const bool const_c0 = c0_.Size() == 1;
const auto C0 = const_c0 ?
Reshape(c0_.Read(), 1, 1, 1) :
Reshape(c0_.Read(), Q1D, Q1D, NE);
const auto LD = Reshape(lim_dist.Read(), D1D, D1D, NE);
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
const auto W = Reshape(w_.Read(), Q1D, Q1D);
const auto b = Reshape(b_.Read(), Q1D, D1D);
const auto bld = Reshape(bld_.Read(), Q1D, D1D);
auto H0 = Reshape(h0_.Write(), DIM, DIM, Q1D, Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int NBZ = 1;
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
MFEM_SHARED double B[MQ1*MD1];
MFEM_SHARED double BLD[MQ1*MD1];
MFEM_SHARED double XY[NBZ][MD1*MD1];
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
kernels::LoadX<MD1,NBZ>(e,D1D,LD,XY);
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
kernels::LoadB<MD1,MQ1>(D1D,Q1D,bld,BLD);
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,BLD,XY,DQ);
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,BLD,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
const double *Jtr = &J(0,0,qx,qy,e);
const double detJtr = kernels::Det<2>(Jtr);
const double weight = W(qx,qy) * detJtr;
const double coeff0 = const_c0 ? C0(0,0,0) : C0(qx,qy,e);
const double weight_m = weight * lim_normal * coeff0;
double D;
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,D);
const double dist = D; // GetValues, default comp set to 0
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
// d2.Diag(1.0 / (dist * dist), x.Size());
const double c = 1.0 / (dist * dist);
double grad_grad[4];
kernels::Diag<2>(c, grad_grad);
ConstDeviceMatrix gg(grad_grad,DIM,DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
H0(i,j,qx,qy,e) = weight_m * gg(i,j);
}
}
}
}
});
}
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &X) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const double ln = lim_normal;
const Vector &LD = PA.LD;
const DenseTensor &J = PA.Jtr;
const Array<double> &W = PA.ir->GetWeights();
const Array<double> &B = PA.maps->B;
const Array<double> &BLD = PA.maps_lim->B;
const Vector &C0 = PA.C0;
Vector &H0 = PA.H0;
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_C0_2D,id,ln,LD,C0,N,J,W,B,BLD,H0);
}
} // namespace mfem
-150
View File
@@ -1,150 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_3D,
const int NE,
const Array<double> &b,
const Array<double> &g,
const DenseTensor &j,
const Vector &h,
Vector &diagonal,
const int d1d,
const int q1d)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto G = Reshape(g.Read(), Q1D, D1D);
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, Q1D, NE);
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
MFEM_SHARED double qdd[MQ1*MD1*MD1];
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
for (int v = 0; v < DIM; ++v)
{
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
// first tensor contraction, along z direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD(qx,qy,dz) = 0.0;
MFEM_UNROLL(MQ1);
for (int qz = 0; qz < Q1D; ++qz)
{
const double *Jtr = &J(0,0,qx,qy,qz,e);
double jrt[9];
ConstDeviceMatrix Jrt(jrt,3,3);
kernels::CalcInverse<3>(Jtr, jrt);
const double Bz = B(qz,dz);
const double Gz = G(qz,dz);
const double L = i==2 ? Gz : Bz;
const double R = j==2 ? Gz : Bz;
const double Jij = Jrt(i,i) * Jrt(j,j);
const double h = H(v,i,v,j,qx,qy,qz,e);
QQD(qx,qy,dz) += L * Jij * h * R;
}
}
}
}
MFEM_SYNC_THREAD;
// second tensor contraction, along y direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD(qx,dy,dz) = 0.0;
MFEM_UNROLL(MQ1);
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy,dy);
const double Gy = G(qy,dy);
const double L = i==1 ? Gy : By;
const double R = j==1 ? Gy : By;
QDD(qx,dy,dz) += L * QQD(qx,qy,dz) * R;
}
}
}
}
MFEM_SYNC_THREAD;
// third tensor contraction, along x direction
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double d = 0.0;
MFEM_UNROLL(MQ1);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx,dx);
const double Gx = G(qx,dx);
const double L = i==0 ? Gx : Bx;
const double R = j==0 ? Gx : Bx;
d += L * QDD(qx,dy,dz) * R;
}
D(dx,dy,dz,v,e) += d;
}
}
}
MFEM_SYNC_THREAD;
}
}
}
});
}
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &D) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const DenseTensor &J = PA.Jtr;
const Array<double> &B = PA.maps->B;
const Array<double> &G = PA.maps->G;
const Vector &H = PA.H;
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_3D,id,N,B,G,J,H,D);
}
} // namespace mfem
-123
View File
@@ -1,123 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "tmop.hpp"
#include "tmop_pa.hpp"
#include "../general/forall.hpp"
#include "../linalg/kernels.hpp"
namespace mfem
{
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_3D,
const int NE,
const Array<double> &b,
const Vector &h0,
Vector &diagonal,
const int d1d,
const int q1d)
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
{
constexpr int DIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
MFEM_SHARED double qdd[MQ1*MD1*MD1];
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
for (int v = 0; v < DIM; ++v)
{
// first tensor contraction, along z direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
QQD(qx,qy,dz) = 0.0;
MFEM_UNROLL(MQ1);
for (int qz = 0; qz < Q1D; ++qz)
{
const double Bz = B(qz,dz);
QQD(qx,qy,dz) += Bz * H0(v,v,qx,qy,qz,e) * Bz;
}
}
}
}
MFEM_SYNC_THREAD;
// second tensor contraction, along y direction
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
QDD(qx,dy,dz) = 0.0;
MFEM_UNROLL(MQ1);
for (int qy = 0; qy < Q1D; ++qy)
{
const double By = B(qy,dy);
QDD(qx,dy,dz) += By * QQD(qx,qy,dz) * By;
}
}
}
}
MFEM_SYNC_THREAD;
// third tensor contraction, along x direction
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
double d = 0.0;
MFEM_UNROLL(MQ1);
for (int qx = 0; qx < Q1D; ++qx)
{
const double Bx = B(qx,dx);
d += Bx * QDD(qx,dy,dz) * Bx;
}
D(dx,dy,dz, v, e) += d;
}
}
}
MFEM_SYNC_THREAD;
}
});
}
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &D) const
{
const int N = PA.ne;
const int D1D = PA.maps->ndof;
const int Q1D = PA.maps->nqpt;
const int id = (D1D << 4 ) | Q1D;
const Array<double> &B = PA.maps->B;
const Vector &H0 = PA.H0;
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_3D,id,N,B,H0,D);
}
} // namespace mfem

Some files were not shown because too many files have changed in this diff Show More