Compare commits

..
Author SHA1 Message Date
camierjs 5457706a29 Fix merge with global API 2026-05-01 10:28:29 -07:00
camierjs fd195d53a5 Merge branch 'dfem-multiple-outputs' into dfem-kernels 2026-05-01 10:03:48 -07:00
camierjs ced299b0b3 Remove vscode files 2026-05-01 09:49:26 -07:00
camierjs 667022093d Merge branch 'master' into origin-dfem-kernels 2026-05-01 09:47:16 -07:00
camierjs 5a3e9a93c2 Merge branch 'master' into dfem-multiple-outputs 2026-05-01 09:30:41 -07:00
Julian Andrej b2bf589c87 jit playground updates 2026-04-29 13:09:32 -07:00
Julian Andrej ac48dfcfa5 update 2026-04-27 16:02:33 -07:00
Julian Andrej 5e98b82b26 skeleton 2026-04-20 14:52:41 -07:00
Julian Andrej dfc582149d move files 2026-04-20 10:27:29 -07:00
Julian Andrej 79680d9bc9 reorganizing dfem backends 2026-04-20 10:21:39 -07:00
camierjs f261928799 Merge branch 'master' into camierjs-dfem-kernels 2026-04-14 08:10:38 -07:00
camierjs f4baeb41ad Added MMA(magic 4|5) 2026-04-12 11:08:24 -07:00
camierjs fd73f3895f Ini PADiffMmaIntegrator 2026-04-12 09:24:21 -07:00
camierjs 5820d70a50 remove unused variable 2026-04-11 14:33:58 -07:00
camierjs d923432284 Use T_Q1D for offset computation 2026-04-11 14:29:20 -07:00
camierjs cffd9c618f Reorder kernels & add fallbacks 2026-04-11 14:20:43 -07:00
camierjs 6ccc490e18 Specialization only on Q1D for dFEM 2026-04-11 13:51:34 -07:00
camierjs 80f973d708 Swap DIM as fastest indice in shared mem 2026-04-11 13:35:14 -07:00
camierjs ef13f6e9d8 Specialization only on Q1D 2026-04-11 13:24:16 -07:00
camierjs 48cb7bcc2e Cleanup 2026-04-11 13:18:21 -07:00
camierjs 621a9ed1d6 Low @ 4.43378k/s 2026-04-11 13:01:34 -07:00
camierjs 5ca23fd56f Fix PADiffLowMult 2026-04-11 11:31:35 -07:00
camierjs 1b9d97ec28 Fix CPU regs3d_t layout 2026-04-11 11:14:17 -07:00
camierjs 4add547b04 cleanup 2026-04-11 08:32:50 -07:00
camierjs 9c9b5cee2f dFEM new kernels with tensors from raw pointers @ 4.2k/s 2026-04-11 08:21:15 -07:00
camierjs 80b5ecfb7b Add back PA ∂fem std kernels 2026-04-10 17:57:43 -07:00
camierjs 074a716634 Re-enable specialized kernels and add #7 w/o specilization 2026-04-10 17:14:12 -07:00
camierjs ad9682118d wip QF apply LOW kernels 2026-04-10 17:06:19 -07:00
camierjs 25cba3afa5 Cleanup, inc kernels3d header file 2026-04-10 14:58:55 -07:00
camierjs 76f061dd9c Low running on GPU 2026-04-10 14:16:04 -07:00
camierjs 22f25269a5 wip low kernels 2026-04-10 13:19:47 -07:00
camierjs 684d8fc64b WIP LOW reg Grad 3D 2026-04-10 11:34:59 -07:00
camierjs ded0173a92 wip MFEM_SHARED full d0_regs3d_t 2026-04-09 13:40:57 -07:00
camierjs 7aca674961 Added #6 PADiffLowIntegrator 2026-04-09 11:58:10 -07:00
camierjs be6f6299aa Bring smem version 2026-04-08 16:57:34 -07:00
camierjs c81fabcc54 BP3/1/6/160 @ 4k/s 2026-04-08 15:50:23 -07:00
camierjs f963bd897b wip forall_2D_batch, constant memory 2026-04-08 15:13:04 -07:00
camierjs dcd5bee0f6 wip forall_2D_batch tries 2026-04-08 13:02:14 -07:00
camierjs 1dd20334e2 Add forall_kernel_static_smem_launch_bounds 2026-04-07 16:35:07 -07:00
camierjs 8889988956 Tuos runs w/ new kernels 2026-04-07 10:21:12 -07:00
camierjs 06b4a68c7d dFEM new kernels hc vdim, same B/G outputs 2026-04-07 09:52:07 -07:00
camierjs 17e15d08cd Merge branch 'main' of github.com:camierjs/mfem-dfem-kernels into main 2026-04-07 08:13:53 -07:00
camierjs c05ce7dacd R Identity 2026-04-07 08:13:51 -07:00
camierjs c80f7fb1f1 json update 2026-04-07 08:12:32 -07:00
camierjs 0eac62aa3b wip restriction D2D 2026-04-07 07:07:58 -07:00
camierjs fbe07d97ea All but MF new kernels 2026-04-07 05:58:47 -07:00
camierjs e4ce8375f3 Reuse qdata for PA ∂fem new kernels 2026-04-06 18:06:09 -07:00
camierjs da959ef7e9 Add back mi300a constant B & G 2026-04-03 20:19:32 -07:00
camierjs aba87d9e14 Add user cmake tuo 2026-04-03 19:27:40 -07:00
camierjs f0efcf4253 nvcc compilations 2026-04-03 19:18:04 -07:00
camierjs 7aee5f56ba Rename user cmake file 2026-04-03 18:22:16 -07:00
camierjs 9bd3409458 Add darwin/matrix user cmake files 2026-04-03 18:21:46 -07:00
camierjs cca1678ccb Add vscode files 2026-04-03 09:32:34 -07:00
camierjs 97ca2f9ecc Fix dFEM new action for exact CG iterations 2026-04-03 09:16:37 -07:00
camierjs a00f222761 dFEM kernels CPU runs 2026-04-02 17:48:30 -07:00
camierjs 99ebc58be4 wip merge fixes 2026-04-02 11:47:40 -07:00
camierjs cf213ea6b6 Merge branch 'master' into camierjs-dfem-kernels 2026-04-02 10:21:43 -07:00
camierjs 031be712a1 Warnings fix 2026-04-02 10:13:53 -07:00
camierjs 134bc32d93 Merge branch 'origin-dfem-kernels' into camierjs-dfem-kernels 2026-04-02 10:05:30 -07:00
camierjs e640a3e3fb Cleanup and run with new traces 2026-04-02 09:49:26 -07:00
Julian Andrej faba224c26 jit playground 2026-04-02 08:34:33 -07:00
Giorgis Georgakoudis 485121d3ad Use proteus::jit_arg instrumentation 2026-03-30 16:55:34 -07:00
Giorgis Georgakoudis af527e27d1 Update top-level CMakeLists.txt for proteus
- Add target-based path for libProteusPass
- Link with libproteus
2026-03-30 16:51:52 -07:00
Julian Andrej 3e680af733 disable derivatives temporarily 2026-03-30 12:41:16 -07:00
Julian Andrej 4212310405 add proteus 2026-03-30 12:20:44 -07:00
Julian Andrej 41aed0e916 cosmetic changes 2026-03-12 08:42:35 -07:00
Julian Andrej 9bf156adf2 bugfix 2026-03-10 09:18:45 -07:00
Julian Andrej 9f0fcd6b10 custom layouts 2026-03-10 08:10:38 -07:00
Julian Andrej 2350a5e9eb typo 2026-03-05 10:52:39 -08:00
Julian Andrej 001c686a19 make rank 0 tensor compatible with real_t 2026-03-05 10:50:33 -08:00
Julian Andrej da9fc85862 support MultiVector 2026-03-04 13:32:19 -08:00
Julian Andrej 996553be3d simplify assert 2026-03-04 12:53:26 -08:00
Julian Andrej ff6715b8b1 Merge branch 'multi-vector-dev' into dfem-multiple-outputs 2026-03-04 12:46:47 -08:00
Julian Andrej 54acbdd395 consistency checks 2026-03-04 10:35:30 -08:00
Julian Andrej 76d4f1942b refactor how bases are created 2026-03-04 07:36:19 -08:00
Julian Andrej 939310203d updates 2026-03-03 15:02:16 -08:00
Julian Andrej 979f08b3eb allow Q-function arguments to be non-const references 2026-03-02 09:14:15 -08:00
Julian Andrej 60a04e4e4f qdata L to Q 2026-02-23 17:22:52 -08:00
Julian Andrej 8d95a6e5ca qdata 2026-02-23 16:51:58 -08:00
Julian Andrej 10cb466fb2 more stuff 2026-02-23 14:14:41 -08:00
Julian Andrej 5be9de7e95 bugfixes 2026-02-23 09:15:17 -08:00
Julian Andrej a13a4f4d8b more 2026-02-20 14:14:10 -08:00
Julian Andrej bf9b6f4d83 multiple outputs with derivatives 2026-02-19 12:51:30 -08:00
Julian Andrej 52b8703b78 bugs 2026-02-06 16:26:43 -08:00
Julian Andrej 69e7820d01 phew 2026-02-06 15:05:03 -08:00
Julian Andrej dbedeecece more refactor 2026-02-05 09:58:24 -08:00
Julian Andrej e8847b80a2 refactor 2026-02-04 13:25:00 -08:00
Julian Andrej 0fe2aece0b enable multiple outputs 2026-01-26 14:58:58 -08:00
camierjs b639394d56 Cleanup 2025-07-10 08:26:00 -07:00
camierjs 12c2d71bd2 Merge branch 'master' into dfem-kernels 2025-07-10 08:25:18 -07:00
camierjs c6f7e9f635 Switched VDIM/DIM dimensions runs 2025-07-07 13:56:11 -07:00
camierjs 7151713d9c Try StiffnessMult with VDIM/DIM layout 2025-07-07 10:10:06 -07:00
camierjs a5379de077 Use constants, simplify & cleanup 2025-07-06 16:29:53 -07:00
camierjs 85d24f1354 Merge branch 'master' into dfem-kernels 2025-07-06 14:45:43 -07:00
camierjs b60a76f0db Cleanup 2025-07-06 14:45:06 -07:00
camierjs e0dc7659fb With specializations 2025-07-06 11:56:22 -07:00
camierjs b7b4268138 use_kernels_specialization 2025-07-05 21:11:17 -07:00
camierjs e89a61399c GPU runs 2025-07-05 15:57:31 -07:00
camierjs 6c90686880 Forced inlines w/o changes 2025-07-05 15:12:09 -07:00
camierjs 331a66c027 Cleanup 2025-07-05 14:51:59 -07:00
camierjs 6d509fa5c3 Cleanup 2025-07-05 14:40:23 -07:00
camierjs b62cf39359 Simplify 2025-07-05 14:34:58 -07:00
camierjs 49f44e65c0 Cleanup & Simplify 2025-07-05 12:04:54 -07:00
camierjs 32697fea42 cleanup 2025-07-05 11:37:11 -07:00
camierjs c80a091f56 w/o unpack_shmem 2025-07-05 10:56:43 -07:00
camierjs bbc6708976 with r2 2025-07-05 10:46:52 -07:00
camierjs 6a01f6551a Action 2025-07-05 10:17:05 -07:00
camierjs 0c663a8aa2 with process_qf_result 2025-07-05 08:28:27 -07:00
camierjs be224ed94a wip apply_kernel 2025-07-05 08:17:17 -07:00
camierjs c09da71078 wip back action_callback_new 2025-07-04 18:07:15 -07:00
camierjs cf5f0126ce Avoid double mdofs in first iteration 2025-07-04 17:15:46 -07:00
camierjs a23e8907d7 removed fqp and use directly r0 2025-07-04 17:03:04 -07:00
camierjs 4ca2805303 map_quadrature_data_to_fields 2025-07-04 14:14:08 -07:00
camierjs 1b775faa43 is_gradient_fop 2025-07-04 13:38:00 -07:00
camierjs 0edefaeae5 action_callback_new cleanup 2025-07-04 12:50:59 -07:00
camierjs a12132ccb6 MFEM_NEW_KERNELS & action_callback_new 2025-07-04 11:30:30 -07:00
camierjs 9f044d89b5 bench_dfem run with assert Grad diff 2025-07-04 09:51:07 -07:00
camierjs 191e3df84b LoadDofs3d, Grad3d 2025-07-04 09:45:34 -07:00
camierjs 6b6f8afdac wip sync 2025-07-04 09:11:58 -07:00
camierjs 25e333dbf4 Merge branch 'dfem-bench' 2025-07-04 08:12:03 -07:00
camierjs 856d13e9ff Roctx init 2025-07-04 08:00:50 -07:00
camierjs eb8f7f433c Run tests/unit/dfem/test_diffusion_q1d 2025-07-02 17:00:41 -07:00
camierjs 6a693a818f wip merge fix 2025-07-02 15:05:47 -07:00
camierjs fa7d81095a Merge branch 'master' into dfem-kernels 2025-07-02 15:05:30 -07:00
camierjs 16260082f6 wip interpolate 2025-07-02 14:52:18 -07:00
camierjs 7763785ed7 Use MFEM_FOREACH_THREAD_DIRECT 2025-07-02 10:29:40 -07:00
camierjs 2baa889917 Merge branch 'master' into dfem-bench 2025-07-02 08:38:17 -07:00
camierjs 2ed1a9eaad BP3/1/6/25 @ 40 MDof/s 2025-06-30 18:02:56 -07:00
camierjs 1545f03a94 Merge branch 'master' into dfem-bench 2025-06-30 16:19:53 -07:00
camierjs 59a5c9fc79 Sync with fem/kernels.hpp, still performance wip 2025-06-24 11:43:05 -07:00
camierjs c389a3c434 Use latest dFEM for benchmark 2025-06-24 11:27:34 -07:00
camierjs ec96a85f86 Merge branch 'master' into dfem-bench 2025-06-24 11:27:15 -07:00
camierjs dd99371cda dFEM diffusion test Identity vs. None fix 2025-05-19 16:48:56 -07:00
camierjs 90aa6fc544 Merge branch 'dfem-phase1-dev' into dfem-kernels 2025-05-19 16:44:02 -07:00
camierjs 7b4fcc3e52 Add qp wip header/test 2025-05-19 16:42:15 -07:00
camierjs 81b6b7eeb2 Merge branch 'master'/'dfem-phase-1' into dfem-bench 2025-05-19 16:02:46 -07:00
camierjs 4b5974f600 Fix CMake and dFEM bench 2025-05-19 16:01:16 -07:00
camierjs a6926f4ce6 Merge branch 'dfem-phase1-dev' 2025-05-19 15:54:58 -07:00
camierjs b6e972af79 Merge branch 'master' 2025-05-19 15:49:54 -07:00
Julian Andrej e76ec19775 restructure 2025-05-19 14:46:20 -07:00
Julian Andrej d797322fea path 2025-05-19 08:22:36 -07:00
Julian AndrejandJohn Camier 5a5e34a744 Update fem/dfem/doperator.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-05-19 08:11:13 -07:00
Julian AndrejandJohn Camier 93db7052ff Update fem/dfem/util.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-05-19 08:10:44 -07:00
Julian AndrejandJohn Camier f7170af7bd Update fem/dfem/tuple.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-05-19 08:10:12 -07:00
Julian AndrejandJohn Camier b78eef3eaa Update fem/dfem/tuple.hpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-05-19 08:10:00 -07:00
Julian Andrej d5decea85c Revert "change default location for enzyme and add instructions"
This reverts commit dea3ae3317.
2025-05-16 12:59:14 -07:00
Julian Andrej dea3ae3317 change default location for enzyme and add instructions 2025-05-16 12:55:04 -07:00
Julian Andrej 33c1e50235 astyle 2025-05-16 12:39:47 -07:00
Julian Andrej 5718ad1b53 cuda compat 2025-05-16 12:38:08 -07:00
Julian AndrejandAndrew Ho 4f3671e253 Update fem/dfem/util.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:58:50 -07:00
Julian AndrejandAndrew Ho 4e08bb1b66 Update fem/dfem/util.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:58:42 -07:00
Julian AndrejandAndrew Ho 69c5016b63 Update fem/dfem/util.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:58:20 -07:00
Julian AndrejandAndrew Ho ce1bf58dc0 Update fem/dfem/doperator.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:58:13 -07:00
Julian AndrejandAndrew Ho 5eb00c9ee6 Update fem/dfem/doperator.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:58:05 -07:00
Julian AndrejandAndrew Ho 1f3b6b95aa Update fem/dfem/util.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:57:58 -07:00
Julian AndrejandAndrew Ho 118db41049 Update fem/dfem/util.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:57:49 -07:00
Julian AndrejandAndrew Ho 8390c3e50b Update fem/dfem/doperator.hpp
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:57:40 -07:00
Julian Andrej 168b5179e6 remove findenzyme module 2025-05-16 11:57:11 -07:00
Julian AndrejandAndrew Ho 7697f6d400 Update CMakeLists.txt
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-05-16 11:56:11 -07:00
Julian AndrejandJan Nikl 235ebce5d5 Update examples/dfem/minimal_surface.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-05-16 07:58:47 -07:00
Julian AndrejandJan Nikl e8a09d6499 Update examples/dfem/minimal_surface.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-05-16 07:57:47 -07:00
Julian AndrejandJan Nikl edc67827d8 Update examples/dfem/minimal_surface.cpp
Co-authored-by: Jan Nikl <nikl1@llnl.gov>
2025-05-16 07:57:34 -07:00
Julian Andrej 9e5cdef2ef add host device 2025-05-14 17:39:40 -07:00
Andrew Ho c2f4a5e248 Updated makefile to work with clang as the cuda compiler 2025-05-14 11:25:59 -07:00
Julian Andrej 72d811b289 device support for fdjacobian 2025-05-14 09:57:18 -07:00
Julian Andrej 43731aa990 memory type for temporary 2025-05-14 09:46:39 -07:00
Julian Andrej 45f59fff3a device memory locations 2025-05-14 09:23:58 -07:00
Julian Andrej 58a4cfa132 cuda compat 2025-05-14 07:43:13 -07:00
Julian Andrej 333dd3f2fd rename ParametricSpace -> ParameterSpace 2025-05-13 13:29:38 -07:00
Julian Andrej c4f7dd77b1 bugs 2025-05-13 13:24:18 -07:00
Julian Andrej f442b83573 whitespace 2025-05-13 11:33:54 -07:00
Julian Andrej 768aaae25d docs 2025-05-13 11:27:36 -07:00
Julian Andrej eab997c557 typo 2025-05-13 11:26:09 -07:00
Julian Andrej 9d73dc487d docs 2025-05-13 11:24:36 -07:00
Julian Andrej 2575ac61ba more comments 2025-05-13 09:08:54 -07:00
Julian Andrej 6130144da1 comments 2025-05-13 08:46:15 -07:00
camierjs 68db31da44 SetMaxOf comments 2025-05-12 18:15:09 -07:00
Julian Andrej 1acbce733c cmake 2025-05-09 11:46:10 -07:00
Julian Andrej b44316049b cmake 2025-05-09 10:41:11 -07:00
Julian Andrej 2e133e8ecb remove serial tests from cmake 2025-05-09 10:36:37 -07:00
Julian Andrej dfb2f4d7f2 typos 2025-05-09 08:41:17 -07:00
Julian Andrej 10e9e4215f cmake 2025-05-09 08:38:46 -07:00
Julian Andrej 8125a211d3 Merge branch 'master' into dfem-phase1-dev 2025-05-08 09:40:48 -07:00
Julian Andrej 818b8db433 switch example to CG 2025-05-07 17:19:36 -07:00
Julian Andrej ad4626edfc leftover comment 2025-05-07 15:51:29 -07:00
Julian Andrej 8d7e8933cf mesh 2025-05-07 15:50:43 -07:00
Julian Andrej 3ad21a409f precision 2025-05-07 15:31:39 -07:00
Julian Andrej b16b550150 corrections 2025-05-07 15:08:11 -07:00
Julian Andrej 102dc8bd02 ifdef 2025-05-07 14:20:41 -07:00
Julian Andrej e306ba0c85 ifdef 2025-05-07 13:44:30 -07:00
Julian Andrej 4b88ad2b0a more minsurface 2025-05-07 13:19:28 -07:00
Julian Andrej d0fb4e342e example draft 2025-05-06 21:06:56 -07:00
Julian Andrej b53d0529db bug 2025-05-06 17:41:48 -07:00
Julian Andrej 8c7988b525 changes 2025-05-06 17:41:22 -07:00
Julian Andrej dfffe4b5e8 rename fops 2025-05-06 09:12:33 -07:00
Julian Andrej 538aa11904 rename fops 2025-05-06 08:56:20 -07:00
Julian Andrej 6fa978af9a rename fops 2025-05-06 08:53:31 -07:00
Julian Andrej 3f81af72f6 rename fops 2025-05-06 08:51:06 -07:00
Julian Andrej 97f1cf08fb docs 2025-05-05 13:45:03 -07:00
camierjs e047cec18a All 3 MQ1Settings working 2025-05-03 14:30:34 -07:00
camierjs bdcf59d109 make_qf_map 2025-05-03 13:48:03 -07:00
Tzanio Kolev d3f1379dc8 Merge branch 'master' into dfem-phase1-dev 2025-05-03 13:46:43 -07:00
camierjs b876d32452 Pre cleanup MQ1 on qfunction 2025-05-03 13:06:34 -07:00
camierjs 50a6be3d58 wip runtime_get 2025-05-03 10:49:25 -07:00
camierjs 5251db2278 All interpolate gradient tests 2025-05-02 17:31:40 -07:00
camierjs 92fca7cf01 Interpolate all Gradient but toroid mesh 2025-05-02 17:27:54 -07:00
camierjs e22f5bc048 Interpolate Gradient AlmostEq 2025-05-02 17:17:22 -07:00
camierjs f971d1e0bb Merge branch 'dfem-phase1-dev' 2025-05-02 15:29:05 -07:00
camierjs 752917acaa Rename Diffusion PA kernels
dFEM DOperator debug traces
2025-05-02 15:27:59 -07:00
Julian Andrej ea6fb52698 bug 2025-05-02 13:09:47 -07:00
Julian Andrej 07a87e369c style 2025-05-02 12:01:08 -07:00
camierjs 56c46e6da5 Remove MFEM_FOREACH_THREAD1 2025-05-02 11:23:27 -07:00
camierjs 2db4ca1300 Avoid MFEM recompilation with dFEM changes 2025-05-02 11:19:48 -07:00
camierjs 53bc415268 Squashed commit of the following:
commit be537728df
Merge: 8cc9eec53 4e5b98b10
Author: camierjs <camierjs@gmail.com>
Date:   Fri May 2 10:37:34 2025 -0700

    Merge branch 'dfem-phase1-dev' into dfem-bench

commit 4e5b98b10f
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 10:05:20 2025 -0700

    doc

commit d4acd906bf
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 09:01:51 2025 -0700

    update brew before enzyme install

commit d751ce66a3
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:43:46 2025 -0700

    ci

commit 3f0abd4dfd
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:40:45 2025 -0700

    ci

commit 44a423d804
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:39:58 2025 -0700

    ci

commit 3e61e0490e
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:38:47 2025 -0700

    ci

commit def4919592
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:33:33 2025 -0700

    ci config

commit 2d147d70e0
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:33:29 2025 -0700

    reintroduce tests

commit e29e64dffe
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Fri May 2 08:04:44 2025 -0700

    reintroduce macos fp64 ci target

commit a7ec259bd5
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Thu May 1 16:46:22 2025 -0700

    reintroduce macos fp64 ci target

commit 3e93e19767
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Thu May 1 14:41:25 2025 -0700

    enzyme bug notes

commit 532b065596
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Thu May 1 14:41:17 2025 -0700

    consistency

commit 82c1e2315b
Author: Julian Andrej <andrej1@llnl.gov>
Date:   Thu May 1 13:19:25 2025 -0700

    modernize

commit 8cc9eec535
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 11:06:03 2025 -0700

    Remove unused code

commit dece65be31
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 10:59:50 2025 -0700

    Header and style

commit 3e6d29b3dd
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 10:53:12 2025 -0700

    Meld toward dfem

commit 487135b497
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 10:48:02 2025 -0700

    Meld back toward dfem dev

commit 91f648aa95
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 10:36:43 2025 -0700

    Remove examples leftovers

commit 999931ded2
Author: camierjs <camierjs@gmail.com>
Date:   Thu May 1 10:36:18 2025 -0700

    Sync dfem bench
2025-05-02 10:40:23 -07:00
camierjs be537728df Merge branch 'dfem-phase1-dev' into dfem-bench 2025-05-02 10:37:34 -07:00
Julian Andrej 4e5b98b10f doc 2025-05-02 10:05:20 -07:00
Julian Andrej d4acd906bf update brew before enzyme install 2025-05-02 09:01:51 -07:00
Julian Andrej d751ce66a3 ci 2025-05-02 08:43:46 -07:00
Julian Andrej 3f0abd4dfd ci 2025-05-02 08:40:45 -07:00
Julian Andrej 44a423d804 ci 2025-05-02 08:39:58 -07:00
Julian Andrej 3e61e0490e ci 2025-05-02 08:38:47 -07:00
Julian Andrej def4919592 ci config 2025-05-02 08:33:33 -07:00
Julian Andrej 2d147d70e0 reintroduce tests 2025-05-02 08:33:29 -07:00
Julian Andrej e29e64dffe reintroduce macos fp64 ci target 2025-05-02 08:04:44 -07:00
Julian Andrej a7ec259bd5 reintroduce macos fp64 ci target 2025-05-01 16:46:22 -07:00
Julian Andrej 3e93e19767 enzyme bug notes 2025-05-01 14:41:25 -07:00
Julian Andrej 532b065596 consistency 2025-05-01 14:41:17 -07:00
Julian Andrej 82c1e2315b modernize 2025-05-01 13:19:25 -07:00
camierjs 8cc9eec535 Remove unused code 2025-05-01 11:06:03 -07:00
camierjs dece65be31 Header and style 2025-05-01 10:59:50 -07:00
camierjs 3e6d29b3dd Meld toward dfem 2025-05-01 10:53:12 -07:00
camierjs 487135b497 Meld back toward dfem dev 2025-05-01 10:48:02 -07:00
camierjs 91f648aa95 Remove examples leftovers 2025-05-01 10:36:43 -07:00
camierjs 999931ded2 Sync dfem bench 2025-05-01 10:36:18 -07:00
camierjs 15dbcae725 Add version info 2025-05-01 10:26:49 -07:00
camierjs 01efb623da Update kernels pa to regs use 2025-05-01 10:08:01 -07:00
camierjs f854c5262d Move kernels pa to dfem regs 2025-05-01 10:07:46 -07:00
camierjs a91b754aaa Update dfem examples 2025-05-01 10:06:06 -07:00
camierjs b1623ff3d4 Sync dfem examples with latest changes 2025-05-01 10:05:54 -07:00
camierjs c2426ca45a Merge branch 'dfem-phase1-dev' 2025-05-01 09:35:22 -07:00
camierjs 276f419a3d Merge branch 'dfem-phase1-dev' of github.com:mfem/mfem into dfem-phase1-dev 2025-05-01 09:31:53 -07:00
Julian Andrej c91b8bea01 prevent possible indexing error 2025-05-01 08:42:56 -07:00
Veselin Dobrev 7bdceca6ce Windows CI debug 2025-05-01 04:48:28 -07:00
Veselin Dobrev 6f9a263435 Disable Ninja on windows -- it does not detect MSVC.
Add a debug action step to print the environment under windows.
2025-05-01 03:49:52 -07:00
Veselin Dobrev 28a7865ed1 Fix MSVC build issue.
Use the Ninja CMake generator on Windows to try to speedup the build.
2025-05-01 00:44:37 -07:00
Julian Andrej 6e7335ac52 Revert "test more captures"
This reverts commit 8115383dec.
2025-04-30 16:49:35 -07:00
Julian Andrej 8115383dec test more captures 2025-04-30 16:34:17 -07:00
Julian Andrej 3d1b017a60 Revert "test capture"
This reverts commit bf14e5b018.
2025-04-30 16:29:16 -07:00
Julian Andrej bf14e5b018 test capture 2025-04-30 16:12:23 -07:00
Julian Andrej fb3517453f correctness 2025-04-30 15:48:35 -07:00
Julian Andrej bfca6beb28 Revert "hints for mscv"
This reverts commit 78a60cc1d9.
2025-04-30 14:30:02 -07:00
Julian Andrej f51e46d3d8 changelog 2025-04-30 14:02:28 -07:00
Julian Andrej 78a60cc1d9 hints for mscv 2025-04-30 14:02:24 -07:00
Julian Andrej 935d3a9e42 namespaces 2025-04-30 10:30:44 -07:00
Julian Andrej 35866f8485 namespaces 2025-04-30 09:40:45 -07:00
Julian Andrej 6b4b644355 namespaces 2025-04-30 09:38:52 -07:00
Julian Andrej b9ec58e7a1 guards 2025-04-30 09:19:17 -07:00
Julian Andrej 4644aed322 native ad test 2025-04-30 09:18:22 -07:00
Julian Andrej 80da896859 namespaces 2025-04-30 09:18:16 -07:00
Julian Andrej b96dcb4401 namespaces 2025-04-30 08:54:57 -07:00
Julian Andrej 5054f1784d again 2025-04-29 14:13:51 -07:00
Julian Andrej 788c0efda0 sync input values 2025-04-29 14:10:42 -07:00
Julian Andrej 4d49d42702 typo 2025-04-29 13:33:16 -07:00
Julian Andrej f5192230e0 more msvc handholding 2025-04-29 11:22:24 -07:00
Tzanio Kolev 400e3eca7d Merge branch 'master' into dfem-phase1-dev 2025-04-29 09:55:18 -07:00
Julian Andrej b90c8d80fe remove problematic constexpr 2025-04-29 08:50:20 -07:00
Veselin Dobrev a0491f6bfc Fix some msvc warnings which also fixed some compilation errors 2025-04-29 01:43:41 -07:00
Julian Andrej a8df54cf5d please msvc 2025-04-28 20:39:29 -07:00
Julian Andrej 9c4e43ee12 revert 2025-04-28 19:29:23 -07:00
Julian Andrej 7a1887c525 oops 2025-04-28 17:58:40 -07:00
Julian Andrej 907783f9ca testing 2025-04-28 17:56:23 -07:00
Julian Andrej f4f68fa021 size 2025-04-28 17:08:04 -07:00
Julian Andrej b76e9e80a7 real annoying real_t 2025-04-28 17:03:59 -07:00
Julian Andrej b8f677b2fe shadows 2025-04-28 16:59:44 -07:00
Julian Andrej 6e42fbae4d guards 2025-04-28 16:51:34 -07:00
Julian Andrej b8c0008061 include orders etc 2025-04-28 16:36:44 -07:00
Julian Andrej cdce090c2a cmake 2025-04-28 15:48:25 -07:00
Julian Andrej e246c0852b c++17 2025-04-28 15:40:50 -07:00
Julian Andrej 4db86286ee unguard test 2025-04-28 14:05:55 -07:00
Julian Andrej 9308946715 guards 2025-04-28 14:05:44 -07:00
Julian Andrej d28eca6b7f renaming 2025-04-28 14:05:34 -07:00
Julian Andrej 8e26105232 temporary disable offended unit tests 2025-04-28 11:53:12 -07:00
Julian Andrej c674f9f7ad defuse test 2025-04-24 15:29:09 -07:00
Julian Andrej 537d30120a Merge branch 'master' into dfem-phase1-dev 2025-04-24 14:35:23 -07:00
Julian Andrej 519267e1cb paths 2025-04-24 14:02:09 -07:00
Julian Andrej 2c495fb70d shadow warnings 2025-04-24 13:36:05 -07:00
Julian Andrej 401d1aec7b ci 2025-04-24 13:02:10 -07:00
Julian Andrej 8299b1c036 ci 2025-04-24 12:52:18 -07:00
Julian Andrej 9dd1e4dbdb ci 2025-04-24 11:48:17 -07:00
Julian Andrej 47a3534eff ci 2025-04-24 11:44:08 -07:00
Julian Andrej 2f39ff66f3 ci 2025-04-24 11:34:36 -07:00
Julian Andrej ddca183704 ci 2025-04-24 11:20:31 -07:00
Julian Andrej 078ce6130c ci 2025-04-24 11:17:54 -07:00
Julian Andrej 6d15c2a156 ci 2025-04-24 11:11:54 -07:00
Julian Andrej 7d705c0677 ci 2025-04-24 11:09:42 -07:00
Julian Andrej c027328b91 ci 2025-04-24 11:07:27 -07:00
Julian Andrej 65cb67e1c1 ci 2025-04-24 11:04:54 -07:00
Julian Andrej 494f27c14c ci 2025-04-24 11:00:37 -07:00
Julian Andrej 3c02b72084 ci 2025-04-24 10:51:20 -07:00
Julian Andrej 75e2be35ba ci 2025-04-24 10:46:12 -07:00
Julian Andrej b0f9cbfd26 ci 2025-04-24 10:42:24 -07:00
Julian Andrej 6afea18cde ci 2025-04-24 10:39:14 -07:00
Julian Andrej 2e69ff4b97 ci 2025-04-24 10:33:08 -07:00
Julian Andrej 6c70fe9334 ci 2025-04-24 10:27:46 -07:00
Julian Andrej 4ccbd4581e ci 2025-04-24 10:19:07 -07:00
Julian Andrej 6e262f6c3f ci 2025-04-24 10:13:37 -07:00
Julian Andrej 52e10475a5 ci 2025-04-24 10:10:20 -07:00
Julian Andrej c0299a5a4b ci 2025-04-24 10:05:24 -07:00
Julian Andrej 06eecb0dce yaml lint and first enzyme ci entries 2025-04-24 10:01:07 -07:00
Julian Andrej 96261a7742 c++17 and experimental namespace 2025-04-23 18:09:12 -07:00
Julian Andrej d7c479fa1e documentation 2025-04-21 09:21:07 -07:00
Julian Andrej 8b01d8f13b std::cout -> mfem::out 2025-04-16 10:53:23 -07:00
Julian Andrej 710da275c8 add dfem folder to makefile 2025-04-16 09:02:57 -07:00
Julian Andrej fd481eb725 correct include orders 2025-04-16 09:02:43 -07:00
Julian Andrej b5bbdbbed5 vectorfe leftover 2025-04-16 09:02:31 -07:00
Julian Andrej 6a26200314 remove vectorfe crumbs 2025-04-15 11:31:14 -07:00
Julian Andrej a485121526 msvc ambiguity enable_if 2025-04-14 14:40:52 -07:00
Julian Andrej 1f9e1cf175 brackets 2025-04-14 14:03:28 -07:00
Julian Andrej ec402882da move guard 2025-04-14 13:58:10 -07:00
Julian Andrej e7633e0e2c guard tests 2025-04-14 13:49:46 -07:00
Julian Andrej 30aeb465b7 includes 2025-04-14 13:27:15 -07:00
Julian Andrej ff4993fc51 array include 2025-04-14 13:13:44 -07:00
Julian Andrej 0a42ea8021 copyright dates 2025-04-14 13:13:33 -07:00
Julian Andrej b8d024b59b remove example subdirectory 2025-04-14 10:51:57 -07:00
Julian Andrejandcamierjs 9e1ccf4543 phase 1 skeleton
Co-authored-by: camierjs <camierjs@gmail.com>
2025-04-14 09:43:43 -07:00
camierjs 075ebb255d Do one first benchmark 2025-04-09 11:35:40 -07:00
camierjs 3eb6a5b3b2 Merge branch 'dfem-phase1-dev' 2025-04-03 14:01:28 -07:00
Julian Andrej 8ba1f17f72 add nonlinear solver options to command line arguments 2025-04-03 11:01:31 -07:00
Julian Andrej e5f5a79e66 attempt to fix parametric function transfers 2025-04-03 08:19:53 -07:00
Julian Andrej 43f1b19767 switch to 2d by default 2025-04-03 08:19:32 -07:00
Julian Andrej 7bebe4528f stop printing dependency maps 2025-04-03 08:19:16 -07:00
camierjs da63657cdd GCC warning fixes 2025-04-02 18:40:57 -07:00
camierjs 2b1d271888 Merge branch 'dfem-phase1-dev' 2025-04-02 18:34:46 -07:00
camierjs 47fb8a4fda No auto for gcc 2025-04-02 18:34:30 -07:00
camierjs ee7d9726df Warnings & fixes 2025-04-02 18:34:08 -07:00
camierjs 44b560a916 Merge branch 'dfem-phase1-dev' 2025-04-02 17:56:37 -07:00
Julian Andrej 5657f6ebe8 Merge branch 'dfem-phase1-dev' of github.com:mfem/mfem into dfem-phase1-dev 2025-04-02 17:44:22 -07:00
Julian Andrej 19543b6b16 more device stuff 2025-04-02 17:41:57 -07:00
camierjs 94a832a0c6 Merge branch 'dfem-phase1-dev' 2025-04-02 17:21:09 -07:00
camierjs b56e994ecd Copyright header, includes trim & warning fixes 2025-04-02 17:20:29 -07:00
camierjs 8be11cdfdb Remove duplicate inline 2025-04-02 16:49:20 -07:00
camierjs d71a9602b5 Merge branch 'dfem-phase1-dev' 2025-04-02 16:41:06 -07:00
camierjs 1108bb7e85 Use SetMaxOf inside kernel 2025-04-02 16:40:38 -07:00
Julian Andrej ae8e5aa88d some device stuff 2025-04-02 16:21:18 -07:00
camierjs 17f4acf6b1 Merge branch 'main' of github.com:camierjs/mfem-dfem-bench into main 2025-04-02 16:03:32 -07:00
camierjs 2ce3f3037c Cleanup 2025-04-02 16:03:30 -07:00
camierjs 29189a6d4a Merge branch 'dfem-phase1-dev' 2025-04-02 16:02:10 -07:00
Julian Andrej 08f3c86b8a make attributes device compatible 2025-04-02 15:46:19 -07:00
camierjs c6eb171b5b Back to foreach treads 2025-04-02 14:33:42 -07:00
camierjs d26695cd2a Use latest AddDomainIntegrator API 2025-04-02 12:18:21 -07:00
camierjs 01ab390b06 Merge branch 'dfem-phase1-dev' 2025-04-02 12:09:57 -07:00
camierjs 43c42295d3 Few changes with clang 20.1 2025-04-02 12:09:36 -07:00
camierjs 52bc915120 Few fixes to run on device and removed warnings 2025-04-02 12:07:57 -07:00
camierjs cd9cabb955 Cleanup all hipGetLastError 2025-04-02 09:32:36 -07:00
Julian Andrej e66a61c198 add build instructions 2025-03-31 17:23:33 -07:00
Julian Andrej f8b3c78b19 tensor additions 2025-03-31 14:28:29 -07:00
Julian Andrej 4749746171 add laghos 2025-03-31 14:28:10 -07:00
camierjs 1ddd01c2a0 All dfem BP3 versions 2025-03-31 13:47:06 -07:00
camierjs 87ec3850b5 Update Diffusion class 2025-03-30 13:08:31 -07:00
camierjs 1b25a61c9e Re-order kpc benchmarks 2025-03-30 11:52:41 -07:00
camierjs 7bee8e8161 tests/benchmarks/bench_dfem 2025-03-30 11:40:41 -07:00
camierjs a545ff8264 Bring StiffnessIntegrator in bench dfem 2025-03-30 10:29:17 -07:00
camierjs 5352234aef Use SetMaxOf 2025-03-30 10:06:54 -07:00
camierjs c3732f9d86 dfem diffusion3d D1D Q1D tests 2025-03-30 09:46:08 -07:00
camierjs b95f3809fe ParametricSpace d1d/q1d 2025-03-28 17:24:29 -07:00
camierjs e6a28b7753 Merge branch 'dfem-phase1-dev' 2025-03-28 15:09:46 -07:00
camierjs 62adea8b46 WIP dfem diffusion 2025-03-28 15:09:22 -07:00
camierjs 7d11db33c0 Add dfem diffusion multi-version example and bench dfem setup 2025-03-28 12:09:52 -07:00
Julian Andrej 11fce4235b revert width determination 2025-03-28 08:16:26 -07:00
camierjs f500b4875f dfem bench check 2025-03-27 10:48:46 -07:00
camierjs ba212c583e bench dfem init with nvtx 2025-03-27 10:34:12 -07:00
Julian Andrej fd341e07da example 2025-03-21 15:54:58 -07:00
Julian Andrej d59e2a229c phase 1 skeleton 2025-03-21 15:54:18 -07:00
250 changed files with 12672 additions and 25420 deletions
@@ -94,16 +94,6 @@ inputs:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
# Unfortunately, "uses:" fields cannot have references to variables like
# ${{env.MFEM_ACTIONS_VERSION}}, so the branch/tag name has to be hard coded.
# Therefore, in the future, when updating the version of the
# mfem/github-actions to use, we'll have to replace:
# - all definitions of MFEM_ACTIONS_VERSION and
# - all "uses:" fields that refer to mfem/github-actions.
MFEM_ACTIONS_VERSION:
description: Version (branch or tag) of the mfem/github-actions to use.
default: v2.7
runs:
using: 'composite'
steps:
@@ -128,7 +118,6 @@ runs:
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
echo MFEM_ACTIONS_VERSION=${{inputs.MFEM_ACTIONS_VERSION}} >> $GITHUB_ENV
shell: bash
- name: Env (dir)
+2 -2
View File
@@ -53,7 +53,7 @@ runs:
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- uses: mfem/github-actions/build-mfem@v2.7
- uses: mfem/github-actions/build-mfem@v2.5
if: ${{steps.debug.outputs.cache-hit != 'true'}}
env:
CXXFLAGS: ${{env.CXXFLAGS}}
@@ -82,7 +82,7 @@ runs:
run: find . -type f -name '*.o' -delete
shell: bash
- uses: actions/upload-artifact@v7
- uses: actions/upload-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
-6
View File
@@ -12,11 +12,6 @@
name: 'Install MPI'
description: 'Installs MPI and set up its environment variables'
inputs:
NO_FLAGS:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
runs:
using: 'composite'
steps:
@@ -32,7 +27,6 @@ runs:
shell: bash
- name: Env (bis)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
+1 -1
View File
@@ -49,7 +49,7 @@ runs:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- uses: actions/download-artifact@v8
- uses: actions/download-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
+2 -2
View File
@@ -37,14 +37,14 @@ runs:
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{env.MFEM_ACTIONS_VERSION}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- uses: actions/cache/restore@v5 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Hypre/Metis links
if: ${{inputs.par == 'true'}}
-42
View File
@@ -1,42 +0,0 @@
# MFEM Pull Request Review Agent Guide
## Purpose and scope
Review MFEM PRs for correctness, maintainability, performance, portability, test coverage, and MFEM consistency. Use the diff and PR context; reference source files, tests, and CI results when available. Follow `CONTRIBUTING.md`, especially Developer Guidelines, PR rules, checklist, and testing.
## Critical review pillars
- Correctness and numerical behavior
- API and user-facing impact
- Performance implications
- Maintainability and portability
## Review workflow
1. Read the PR description, linked issues, and intended behavior.
2. Inspect the diff before commenting.
3. Identify affected MFEM components, examples, tests, build or docs changes, and downstream APIs.
4. Analyze the code against the critical review pillars.
5. Compare the change against nearby code and MFEM patterns; flag unmotivated deviations.
6. Check whether tests and documentation were updated appropriately.
7. Review CI results and suggest actions.
8. Produce a structured review with prioritized findings.
9. Always limit conclusions to available evidence.
## MFEM-specific review checklist
- Component-aware scope: identify the touched subsystem (FEM, solvers, preconditioners, linear algebra, mesh, examples, miniapps, build, or docs) and assess its impact against the review pillars.
- Numerical and algorithmic behavior: assess issues in convergence, stability, tolerances, precision, iteration limits, and failure handling. If clear opportunities exist to improve the algorithmic approach, call them out with expected impact.
- API and user-facing impact: assess backward compatibility, user-visible behavior and default changes, migration impact, deprecations, and whether documentation clearly explains user-facing API changes.
- Data structure and memory semantics: assess ownership, lifetime, aliasing, container behavior, and device-host synchronization.
- Parallel and serial behavior: assess whether the change preserves equivalent semantics in serial and parallel modes where applicable; if logic is currently mode-specific, check whether extension to the other mode is straightforward (clear abstractions, no hard-wired assumptions), document constraints, and call out expected behavior differences explicitly.
- Backend and portability impact: assess likely cross-backend risks in CPU, CUDA, HIP, OCCA, RAJA, partial assembly, fallback paths, compiler compatibility, and platform assumptions.
- Build, dependency, and configuration impact: assess CMake or make changes, optional dependency behavior, and feature-flag interactions.
- Tests and docs alignment: check available regression or unit coverage evidence for changed behavior, and ensure docs are updated for new flags, APIs, options, or behavior changes.
- MFEM developer-guideline fit: keep code lean, simple, general, logically separated, and portable; suggest C++17 improvements when they clearly improve safety, clarity, or maintainability.
- New source files, examples, or miniapps: if a PR adds source/header files, verify they are properly wired into the relevant `makefile` and `CMakeLists.txt`, referenced in docs where applicable (including `doc/CodeDocumentation.dox`), and added to top-level `.gitignore` only when generated artifacts require it.
- Changelog: verify `CHANGELOG` is updated if the PR introduces significant new features or user-facing changes.
- MFEM conventions: use `real_t`; use `mfem::out`/`mfem::err` instead of `std::cout`/`std::cerr` in library code; flag large/binary files; if AI assistance is apparent but undisclosed, suggest following `CONTRIBUTING.md`.
- Edge cases: if the PR touches complex or error-prone areas, suggest additional tests for edge cases, failure modes, and parallel behavior.
## Commenting guidelines
- Keep comments concise, actionable, and grounded in the diff.
- Focus on correctness, behavior changes, and user impact over style nits.
- Be professional, concise, collaborative, technically precise, and avoid unsupported assumptions.
+7 -3
View File
@@ -13,7 +13,7 @@ Note that some of these scripts use the shared MFEM GitHub Actions from the exte
<https://github.com/mfem/github-actions>
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch (or tag) in the above from which the action is taken.
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
The current CI workflows are:
@@ -29,12 +29,16 @@ Runs a number of static repository-level sanity checks.
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
## `mfem-analysis.yml` (`build-analysis`)
Checks if the code builds and satisfies minimal requirements.
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
## `builds-and-tests.yml`
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
One matrix job runs `tests/scripts/gitignore` after `make test-noclean` to check generated artifacts against `.gitignore`.
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
- `mfem/github-actions/build-hypre`
+23 -25
View File
@@ -40,7 +40,6 @@ env:
METIS_ARCHIVE_MAC: metis-4.0.3-mac.tgz
METIS_TOP_DIR: metis-4.0.3
MFEM_TOP_DIR: mfem
MFEM_ACTIONS_VERSION: v2.7
# Note for future improvements:
#
@@ -111,7 +110,6 @@ jobs:
build-system: make
hypre-target: int64
precision: fp64
gitignore-check: YES
- os: ubuntu-latest
target: opt
codecov: NO
@@ -172,6 +170,20 @@ jobs:
env
shell: bash
# For info on Xcode see:
# - https://github.com/actions/runner-images/issues/12541
# - https://github.com/actions/runner-images/blob/releases/macos-15-arm64/20250811/images/macos/macos-15-arm64-Readme.md#xcode
- name: Xcode version setup (MacOS)
if: matrix.os == 'macos-latest'
run: |
XCODE_PATH="/Applications/Xcode_16.4.app"
echo "> sudo xcode-select -s ${XCODE_PATH}"
sudo xcode-select -s ${XCODE_PATH}
echo "> g++ -v"
g++ -v
echo "> clang++ -v"
clang++ -v
# Only get MPI if defined for the job.
# TODO: It would be nice to have only one step, e.g. with a dedicated
# action, but I (@adrienbernede) don't see how at the moment.
@@ -216,11 +228,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-${{ env.MFEM_ACTIONS_VERSION }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
- name: get hypre
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.7
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -230,7 +242,7 @@ jobs:
- name: get hypre (Windows)
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
uses: mfem/github-actions/build-hypre@v2.7
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
@@ -246,11 +258,11 @@ jobs:
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
- name: install metis
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.7
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{ matrix.os != 'macos-latest' && env.METIS_ARCHIVE || env.METIS_ARCHIVE_MAC }}
dir: ${{ env.METIS_TOP_DIR }}
@@ -292,7 +304,7 @@ jobs:
# MFEM build and test
- name: build
uses: mfem/github-actions/build-mfem@v2.7
uses: mfem/github-actions/build-mfem@v2.5
env:
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
with:
@@ -318,13 +330,7 @@ jobs:
- name: tests
if: matrix.build-system == 'make' && (matrix.target == 'opt' || matrix.os == 'ubuntu-latest')
run: |
cd ${{ env.MFEM_TOP_DIR }}
if [[ "${{ matrix.gitignore-check }}" == "YES" ]]; then
make test-noclean
else
make test
fi
shell: bash
cd ${{ env.MFEM_TOP_DIR }} && make test
- name: cmake checks
if: matrix.build-system == 'cmake' && matrix.target == 'dbg'
@@ -369,16 +375,8 @@ jobs:
# Code coverage (process and upload reports)
- name: codecov
if: matrix.codecov == 'YES'
uses: mfem/github-actions/upload-coverage@v2.7
uses: mfem/github-actions/upload-coverage@v2.5
with:
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
project_dir: ${{ env.MFEM_TOP_DIR }}
directories: "fem general linalg mesh"
env:
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
- name: gitignore
if: matrix.gitignore-check == 'YES'
run: |
cd ${{ env.MFEM_TOP_DIR }}/tests/scripts
./runtest gitignore
-10
View File
@@ -14,19 +14,9 @@ name: "Static Analysis"
on:
push:
branches: ["master", "next"]
paths-ignore: &docs-only-paths
- "**/*.md"
- "doc/**"
- ".binder/**"
- "CITATION.cff"
- "LICENSE"
- "NOTICE"
- "CHANGELOG"
- "INSTALL"
pull_request:
# The branches below must be a subset of the branches above
branches: ["master"]
paths-ignore: *docs-only-paths
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
+100
View File
@@ -0,0 +1,100 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
name: "Build Analysis"
permissions:
actions: write
on:
push:
branches:
- master
- next
pull_request:
workflow_dispatch:
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
env:
HYPRE_ARCHIVE: v2.19.0.tar.gz
HYPRE_TOP_DIR: hypre-2.19.0
METIS_ARCHIVE: metis-4.0.3.tar.gz
METIS_TOP_DIR: metis-4.0.3
COVERAGE_ENV: mfem-coverage
jobs:
gitignore:
runs-on: ubuntu-latest
steps:
- name: checkout MFEM
uses: actions/checkout@v6
with:
path: mfem
- name: Get MPI (Linux)
run: |
sudo apt-get install openmpi-bin libopenmpi-dev
export OMPI_MCA_rmaps_base_oversubscribe=1
- name: Cache Hypre Install
id: hypre-cache
uses: actions/cache@v5
with:
path: ${{ env.HYPRE_TOP_DIR }}
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
- name: Get Hypre
if: steps.hypre-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{ env.HYPRE_ARCHIVE }}
dir: ${{ env.HYPRE_TOP_DIR }}
target: int32
- name: Cache Metis Install
id: metis-cache
uses: actions/cache@v5
with:
path: ${{ env.METIS_TOP_DIR }}
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
- name: Install Metis
if: steps.metis-cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{ env.METIS_ARCHIVE }}
dir: ${{ env.METIS_TOP_DIR }}
# MFEM build and test
- name: build-mfem
uses: mfem/github-actions/build-mfem@v2.5
with:
os: ${{ runner.os }}
target: opt
codecov: NO
mpi: par
build-system: make
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
metis-dir: ${{ env.METIS_TOP_DIR }}
mfem-dir: mfem
- name: test (no clean)
run: |
cd mfem && make test-noclean
- name: gitignore
run: |
cd mfem/tests/scripts
./runtest gitignore
+2 -6
View File
@@ -19,22 +19,18 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.7
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{env.HYPRE_TGZ}}
dir: ${{env.HYPRE_DIR}}
+2 -6
View File
@@ -19,22 +19,18 @@ jobs:
steps:
- uses: actions/checkout@v6
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v5
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
with:
NO_FLAGS: true
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.7
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{env.METIS_TGZ}}
dir: ${{env.METIS_DIR}}
+2 -2
View File
@@ -146,7 +146,7 @@ jobs:
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: find . -type f -name '*.o' -delete
- uses: actions/upload-artifact@v7
- uses: actions/upload-artifact@v4
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit/${{env.unit_tests}}
@@ -172,7 +172,7 @@ jobs:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- uses: actions/download-artifact@v8
- uses: actions/download-artifact@v4
if: ${{steps.restore.outputs.cache-hit != 'true'}}
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
-10
View File
@@ -17,17 +17,7 @@ permissions:
on:
push:
branches: ["master", "next"]
paths-ignore: &docs-only-paths
- "**/*.md"
- "doc/**"
- ".binder/**"
- "CITATION.cff"
- "LICENSE"
- "NOTICE"
- "CHANGELOG"
- "INSTALL"
pull_request:
paths-ignore: *docs-only-paths
workflow_dispatch:
concurrency:
+2 -2
View File
@@ -451,8 +451,8 @@ miniapps/plasma/pic/*.csv
tests/unit/output_meshes
tests/unit/unit_tests
tests/unit/punit_tests
tests/unit/gpu_unit_tests
tests/unit/pgpu_unit_tests
tests/unit/cunit_tests
tests/unit/pcunit_tests
tests/unit/sedov_tests_*
tests/unit/psedov_tests_*
tests/unit/tmop_pa_tests_*
-5
View File
@@ -85,8 +85,3 @@ opt_par_gcc_10_pumi:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +pumi"
opt_par_gcc_10_gslib:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +gslib"
-5
View File
@@ -63,8 +63,3 @@ opt_mpi_cuda_hypre_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
opt_mpi_cuda_gcc_gslib:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda +gslib cuda_arch=90 ^hypre+cuda"
+2 -2
View File
@@ -32,9 +32,9 @@ mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
# run
if [[ "${MACHINE_NAME}" == "dane" ]]; then
srun --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
salloc --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
srun --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
else
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
exit 1
+20 -66
View File
@@ -8,89 +8,43 @@
https://mfem.org
Version 4.9.1 (development)
===========================
- Added policy for AI-assisted contribution to CONTRIBUTING.md.
Version 4.10 (development)
==========================
Discretization improvements
---------------------------
- Improved FindPointsGSLIB surface mesh capability with support for simplices
and an option to specify axis-aligned bounding box padding for near-surface
point queries.
- Replaced legacy simplex quadrature rules with symmetric positive-weight
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
rules guarantee all-positive weights and interior quadrature points,
improving numerical stability. Higher orders fall back to Grundmann-Moller.
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
2015.
Tet rules (d=1-13): Witherden & Vincent (ibid).
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
2022.
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
integrators.
- Replaced legacy simplex quadrature rules with symmetric positive weight rules
for triangles (orders 0-25) and tetrahedra (orders 0-20). These rules
guarantee all-positive weights and interior quadrature points, improving
numerical stability. Higher orders fall back to Grundmann-Moller.
* Triangle rules: Witherden and Vincent, DOI: 10.1016/j.camwa.2015.03.017
* Tet rules (d=1-13): Witherden and Vincent (same as above)
* Tet rules (d=14-20): Chuluunbaatar et al., DOI: 10.1016/j.camwa.2022.08.016
Version 4.9.1 (development)
===========================
- Added support for general 1D Gauss-Jacobi quadrature rules and Stroud conical
quadrature rules on triangles and tetrahedra.
- Improved the GridFunction projection routines. Projections work for Scalar,
Discretization improvements
---------------------------
- Improved the gridfunction projection routines. Projections work for Scalar,
Vector and VectorFE, also NURBS versions. Optionally different types of
projections can be selected, default behavior has not changed.
projections can be selected, default behaviour has not changed.
- Added GridFunction projection methods for trace spaces, i.e., project
coefficients on the mesh skeleton.
- Added methods to estimate function extremum using piecewise linear bounds plus
- Added methods to estimate function extremum using piecewise linear bounds +
recursive subdivision.
- Extend FindPointsGSLIB to support surface meshes.
Meshing improvements
--------------------
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
bounds on the determinant of the mesh transformation Jacobian.
- Added PA support for TMOP's adaptive limiting functionality. Multiple
GridFunctions and Coefficients can be combined to form a composite term.
- Improved support for 1D NURBS meshes with variable order, including using
the patches construct for 1D NURBS meshes.
- Added the option to include material interfaces (faces separating elements
with different element attributes) as additional boundary elements, for
parallel visualization, e.g. with GLVis. This is supported by both the Print
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
Linear and nonlinear solvers
----------------------------
- Added support for trace spaces in PRefinementTransferOperator. This is used in
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
DPG miniapps).
GPU computing
-------------
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
- Added NVIDIA cuDSS library interface. Implementation examples have been
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
details. Supported versions >= 0.6.0.
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
New and updated examples and miniapps
-------------------------------------
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
leverage the ParticleSet capability.
- Added (Complex)PRefinementMultigrid solver option in the DPG miniapps.
Miscellaneous
-------------
- Fixed signed DOF handling in ParGridFunction reading (read constructor) and
saving via SaveAsOne(). Simplified the process of applying the DOF signs by
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
method will return immediately if no sign flips are needed.
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
capability.
Version 4.9, released on Dec 11, 2025
+18 -10
View File
@@ -433,15 +433,6 @@ if (MFEM_USE_STRUMPACK)
endif()
endif()
# cuDSS can only be enabled in CUDA
if (MFEM_USE_CUDSS)
if (MFEM_USE_CUDA)
find_package(CUDSS REQUIRED)
else()
message(FATAL_ERROR " *** cuDSS requires that CUDA be enabled.")
endif()
endif()
# GnuTLS
if (MFEM_USE_GNUTLS)
find_package(_GnuTLS REQUIRED)
@@ -601,6 +592,13 @@ if (MFEM_USE_ENZYME)
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
endif()
if (MFEM_USE_PROTEUS)
enable_language(C)
find_package(proteus REQUIRED PATHS "${PROTEUS_DIR}")
message(STATUS "${PROTEUS_DIR}/include")
include_directories("${PROTEUS_DIR}/include")
endif()
# MFEM_TIMER_TYPE
if (NOT DEFINED MFEM_TIMER_TYPE)
if (APPLE)
@@ -640,7 +638,7 @@ find_package(Threads REQUIRED)
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CUDSS CALIPER CODIPACK
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
ALGOIM ENZYME CUDA::cudart)
@@ -737,6 +735,16 @@ mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
target_compile_features(mfem PUBLIC cxx_std_${CMAKE_CXX_STANDARD})
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES} ${TPL_TARGETS})
if (MFEM_USE_PROTEUS)
add_library(ClangProteusFlags INTERFACE IMPORTED)
set_target_properties(ClangProteusFlags PROPERTIES
INTERFACE_COMPILE_OPTIONS "-fpass-plugin=$<TARGET_FILE:ProteusPass>"
)
target_link_libraries(mfem PUBLIC ClangProteusFlags)
target_link_libraries(mfem PUBLIC proteus)
endif()
if (TPL_TARGETS)
add_dependencies(mfem ${TPL_TARGETS})
endif()
+65 -72
View File
@@ -3,12 +3,12 @@
</p>
<p align="center">
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
</p>
@@ -24,14 +24,6 @@ must be made under this license.
Note also that MFEM has a [Code of Conduct](CODE_OF_CONDUCT.md). By participating
in the MFEM community, you agree to abide by its rules.
## AI Policy
- Use of AI code generation in MFEM is allowed but must be disclosed, e.g. by
selecting the `AI-assisted` label on the PR.
- By submitting a PR, the author acknowledges that they have reviewed and
understand the changes they are proposing.
- PR authors are still responsible for correctness, licensing, and attribution
of all changes.
If you plan on contributing to MFEM, consider reviewing the
[issue tracker](https://github.com/mfem/mfem/issues) first to check if a thread
already exists for your desired feature or the bug you ran into. Use a pull
@@ -84,7 +76,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
follow the [MFEM PR Rules](#mfem-pr-rules).
- When your contribution is fully working and ready to be reviewed, add
the `ready-for-review` label.
- PRs are treated similarly to journal submission, with an "editor" assigning two
- PRs are treated similarly to journal submission with an "editor" assigning two
reviewers to evaluate the changes.
- The reviewers have 3 weeks to evaluate the PR and work with the author to
fix issues and implement improvements.
@@ -125,7 +117,7 @@ The MFEM source code has the following structure:
│ ├── petsc
│ ├── pumi
│ ├── sundials
└── superlu
| └── superlu
├── fem
│ ├── ceed
│ ├── dfem
@@ -137,6 +129,10 @@ The MFEM source code has the following structure:
│ ├── moonolith
│ ├── qinterp
│ └── tmop
│ | ├── assemble
│ | ├── metrics
│ | ├── mult
│ | └── tools
├── general
├── linalg
│ ├── batched
@@ -149,10 +145,11 @@ The MFEM source code has the following structure:
│ ├── common
│ ├── contact
│ ├── dfem
│ ├── diag-smoothers
│ ├── dpg
│ ├── electromagnetics
│ ├── fluids
│ │ ├── navier
│ │ └── schrodinger-flow
│ ├── gslib
│ ├── hdiv-linear-solver
│ ├── hooke
@@ -162,7 +159,6 @@ The MFEM source code has the following structure:
│ ├── nurbs
│ ├── parelag
│ ├── performance
│ ├── plasma
│ ├── shifted
│ ├── solvers
│ ├── spde
@@ -193,15 +189,15 @@ respectively.
- The main finite element classes are:
+ [`FiniteElement`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElementCollection.html)
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
+ [`FiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1FiniteElementSpace.html)
+ [`GridFunction`](https://docs.mfem.org/html/classmfem_1_1GridFunction.html)
+ [`BilinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
- The main linear algebra classes and sources are
+ [`Operator`](https://docs.mfem.org/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1Vector.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
+ [`DenseMatrix`](https://docs.mfem.org/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://docs.mfem.org/html/classmfem_1_1SparseMatrix.html)
+ Sparse [smoothers](https://docs.mfem.org/html/sparsesmoothers_8hpp.html) and linear [solvers](https://docs.mfem.org/html/solvers_8hpp.html)
@@ -213,8 +209,8 @@ shared geometric entities between different tasks. The parallel source files
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
- The main parallel classes are
+ [`ParMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParNCMesh.html)
+ [`ParMesh`](https://docs.mfem.org/html/solvers_8hpp.html)
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
+ [`ParFiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1ParFiniteElementSpace.html)
+ [`ParGridFunction`](https://docs.mfem.org/html/classmfem_1_1ParGridFunction.html)
+ [`ParBilinearForm`](https://docs.mfem.org/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://docs.mfem.org/html/classmfem_1_1ParLinearForm.html)
@@ -224,14 +220,14 @@ have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
#### GPU and general device support
GPU and multi-core CPU support is based on device kernels supporting different
backends (CUDA, HIP, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
device/host memory manager.
- The main device-relevant classes and sources are:
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html), [`hip.hpp`](https://docs.mfem.org/html/hip_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
#### Utilities, building and documentation
- The `general/` directory contains C++ classes that serve as utilities for
@@ -245,8 +241,8 @@ device/host memory manager.
- `examples` and `miniapps` respectively gather simple and more fully-featured
demonstrations of the usage on MFEM. They both rely on `data/` for the
collection of meshes.
- The `tests/` directory contains a unit test suite, additional tests, and
benchmarks.
- The `tests/` directory contains a unit test suite and will later contain more
tests that run example codes.
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
website.
@@ -280,8 +276,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
the top of https://github.com/mfem.
- Consider making your membership public by going to https://github.com/orgs/mfem/people
and clicking on the organization visibility drop box next to your name.
- Project discussions and announcements will be posted at https://github.com/orgs/mfem/discussions,
tagging the `@mfem/everyone` team when appropriate.
- Project discussions and announcements will be posted at
https://github.com/orgs/mfem/teams/everyone.
#### Structure
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
@@ -341,12 +337,11 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Well-designed simple code is frequently more general and powerful.
- Lean code base is easier to understand by new collaborators.
- New features should be added only if they are necessary or generally useful.
- Introduction of language constructs not currently used in MFEM should be
- Introduction of language constructions not currently used in MFEM should be
justified and generally avoided (to maintain portability to various systems
and compilers, including early access hardware).
- We prefer basic C++. Use C++17 features judiciously, prioritizing readability,
consistency with existing MFEM code, and portability to different systems,
compilers and device backends.
- We prefer basic C++ and the C++03 standard, to keep the code readable by
a large audience and to make sure it compiles anywhere.
- *Keep the code general and reasonably efficient*
- The main goal is fast prototyping for research and application development.
@@ -389,7 +384,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- When your branch is ready for other developers to review / comment on
the code, create a pull request towards `mfem:master`.
- Pull requests typically have titles like:
- Pull request typically have titles like:
`Description [new-feature-dev]`
@@ -410,12 +405,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
team will add reviewers as appropriate.
- List outstanding TODO items in the description.
- List outstanding TODO items in the description, see PR #222 for an example.
- When your contribution is fully working and ready to be reviewed, add
or request the `ready-for-review` label.
the `ready-for-review` label.
- PRs are treated similarly to journal submission, with an "editor" assigning
- PRs are treated similarly to journal submission with an "editor" assigning
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
the PR and work with the author to implement improvements and fix issues.
@@ -441,7 +436,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
checks in GitHub Actions enforce MFEM-specific rules which are explained in
the error messages and the `tests/scripts` directory.
- Also note that the tests `branch-history` and `repo-check` found in GitHub
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
Actions can be triggered automatically before each push using git hooks. See
the [git hooks README](config/githooks/README.md) for a detailed explanation.
@@ -498,15 +493,15 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
3. To ensure the quality of the PR by making sure that the code adheres to the [Developer Guidelines](#developer-guidelines), e.g. all methods, data members, and functions have documentation, including data ownership and lifetime, new examples/miniapps have a corresponding PR in mfem/web, major features have `CHANGELOG` entries, etc.
4. To seek help from the editors in case of difficulties.
3. To seek help from the editors in case of difficulties.
5. To complete the review in a timely manner: 3 weeks from assignment.
4. To complete the review in a timely manner: 3 weeks from assignment.
6. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
7. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
8. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
#### Responsibilities of Authors
@@ -532,30 +527,30 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Code builds.
- [ ] Code passes `make style`.
- [ ] Update `CHANGELOG`:
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
- [ ] Update `INSTALL`:
- [ ] Has a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
- [ ] Have the version ranges for any required or optional libraries changed?
- [ ] Does `make` or `cmake` have a new target?
- [ ] Did the requirements or the installation process change? *(rare)*
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] `.github`
- [ ] `.appveyor.yml`
- [ ] Update `.gitignore`:
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
- [ ] Add new patterns (just for the new files above) and re-run the above test.
- [ ] New examples:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] All sample runs at the top of the example source file work.
- [ ] Update `examples/makefile`:
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
- [ ] Add any files generated by it to the `clean` target.
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Update `examples/CMakeLists.txt`:
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
- [ ] List the new example in `doc/CodeDocumentation.dox`.
- [ ] If new examples directory (e.g. `examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
@@ -572,13 +567,13 @@ Before a PR can be merged, it should satisfy the following:
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
- [ ] Consider adding a new test for the new miniapp.
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
- [ ] New capability:
- [ ] All new public, protected, and private classes, methods, data members, and functions have full Doxygen-style documentation in source comments. Documentation should include descriptions of member data, function arguments and return values, template parameters, and prerequisites for calling new functions.
- [ ] Pointer arguments and return values must specify whether ownership is being transferred or lent with the call.
@@ -680,7 +675,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
- [ ] Update URL shortlinks:
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
- [ ] Add the new shortlinks to the MFEM package in `spack`.
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
- [ ] Update website in `mfem/web` repo:
- Update version and shortlinks in `src/index.md` and `src/download.md`.
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
@@ -732,24 +727,22 @@ commit or push, see the [README](config/githooks/README.md) in the `config/githo
directory.
### GitHub Actions smoke tests
### Linux and Mac smoke tests
We use GitHub Actions to drive the default tests on the `master` and `next`
branches. See the `.github/workflows` files and the logs at
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
GitHub Actions testing should be kept lightweight, as there is a time
constraint on jobs. The current workflows cover Linux, macOS, and Windows
configurations.
Testing using GitHub Actions should be kept lightweight, as there is a time
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
- Tests on the `next` branch are currently scheduled to run each night.
### Additional Windows smoke test
We also use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor.yml` file
and the build logs at
### Windows smoke test
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
environment, as well as to test the CMake build. See the `.appveyor` file and the
build logs at
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
CMake is used to generate the MSVC Project files and drive the build. A release
+16 -31
View File
@@ -38,13 +38,14 @@ the option MFEM_USE_METIS.
MFEM also includes support for devices such as GPUs, and programming models such
as CUDA, HIP, OCCA, OpenMP and RAJA.
- Starting with version 4.9, MFEM requires a C++17 compiler.
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
a newer compiler, e.g. GCC version 4.9 or higher.
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
https://developer.nvidia.com/cuda-toolkit
- HIP support requires an AMD GPU and an installation of the ROCm software stack
https://rocm.docs.amd.com
https://rocmdocs.amd.com
- OCCA support requires the OCCA library
https://libocca.org
@@ -82,9 +83,9 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
make parallel -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
CUDA build:
make cuda -j 4
@@ -114,14 +115,14 @@ Serial build:
Parallel build:
(download hypre and METIS 4 from above URLs)
(build METIS 4 in ../metis-4.0 relative to mfem/)
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
(build hypre in ../hypre relative to mfem/)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
make -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
Parallel build with fetching of hypre and METIS:
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
make -j 4
@@ -133,8 +134,7 @@ CUDA build:
HIP build:
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 \
-DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
make -j 4
Example codes (serial/parallel, depending on the build):
@@ -269,7 +269,6 @@ Compilers:
CXX - C++ compiler, serial build
MPICXX - MPI C++ compiler, parallel build
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
HIP_CXX - The HIP compiler, e.g. 'hipcc'
Compiler options:
OPTIM_FLAGS - Options for optimized build
@@ -396,11 +395,6 @@ MFEM_USE_STRUMPACK = YES/NO
classes. When enabled, this option uses the STRUMPACK_* library options, see
below.
MFEM_USE_CUDSS = YES/NO
Enable MFEM functionality based on the cuDSS library. When using cuDSS, CUDA
support must be also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
When enabled, this option uses the CUDSS_* library options, see below.
MFEM_USE_GINKGO = YES/NO
Enable MFEM functionality based on the Ginkgo library, which provides
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
@@ -560,13 +554,13 @@ MFEM_USE_RAJA = YES/NO
MFEM_USE_OCCA = YES/NO
Enables support for the OCCA library in MFEM. OCCA is an open-source library
which aims to make it easy to program different types of devices (e.g. CPU,
GPU, FPGA) by providing a unified API for interacting with JIT-compiled
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
MFEM_USE_GSLIB = YES/NO
Enables MFEM functionality based on the GSLIB library, and specifically its
FindPoints component, which provides robust algorithms to evaluate finite
FindPoints component, which provides a robust algorithms to evaluate finite
element functions in a collection of points in physical space. When enabled,
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
@@ -725,18 +719,9 @@ The specific libraries and their options are:
Options: STRUMPACK_OPT, STRUMPACK_LIB.
Versions: STRUMPACK >= 3.0.0.
- CUDSS (optional), used when MFEM_USE_CUDSS = YES. Note that CUDSS requires
CUDA 12.x toolkit and the cuDSS libraries. The supported communication backend
is OpenMPI 4.x (default), and OpenMPI 4.x or a later version must be pre-built.
The source files in the cuDSS tarball provide guidance for developing custom
MPI implementations.
URL: https://developer.nvidia.com/cudss
https://docs.nvidia.com/cuda/cudss/advanced_features.html#communication-layer-library-in-cudss
Options: CUDSS_OPT, CUDSS_LIB.
Versions: cuDSS >= 0.6.0.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Ginkgo may have additional
requirements and module-specific dependencies; see the webpage below.
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
C++ compiler that supports the C++-17 standard. For additional requirements
and dependencies of specific modules, see the Ginkgo webpage below.
URL: https://ginkgo-project.github.io
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or
Debug).
@@ -808,7 +793,7 @@ The specific libraries and their options are:
Options: CONDUIT_OPT, CONDUIT_LIB.
Versions: Conduit >= 0.3.1.
- ADIOS2 (optional), used when MFEM_USE_ADIOS2 = YES.
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
URL: https://adios2.readthedocs.io/
Versions: ADIOS >= 2.5.0.
@@ -884,7 +869,7 @@ The specific libraries and their options are:
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
Versions: RAJA >= 2022.10.3.
- Moonolith (optional), used when MFEM_USE_MOONOLITH = YES.
- Moonolith (optional), use when MFEM_USE_MOONOLITH = YES.
URL: https://bitbucket.org/zulianp/par_moonolith
Options: MOONOLITH_DIR
Versions: MOONOLITH >= 1.1.0.
@@ -972,7 +957,7 @@ CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
To use a specific generator use the "-G <generator>" option of cmake:
cmake <mfem-source-dir> -G "Xcode"
cmake <mfem-source-dir> -G "Visual Studio 17 2022"
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
cmake <mfem-source-dir> -G "MinGW Makefiles"
With CMake it is possible to build MFEM as a shared library using the standard
@@ -1217,7 +1202,7 @@ larger problems, there are two options:
Specific options for HIP
========================
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
ROCm install, as well as having `$ROCM_PATH/bin` in `PATH`.
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
Specific options for RAJA+HIP+MPI
=================================
-1
View File
@@ -28,7 +28,6 @@ license files. These software products and their licenses are as follows:
* AmgXWrapper (linalg/amgxsolver.{hpp,cpp}) -- MIT license
* Catch++ (tests/unit/catch.hpp) -- Boost 1.0 license
* Gecko (general/gecko.{cpp,hpp}) -- BSD 3-clause license
* gslib (fem/gslib.{cpp,hpp}, mesh/bb_grid_map.{cpp,hpp}) -- BSD 3-clause license
* Picojson (fem/picojson.h) -- Custom 2-clause license
* TinyXML2 (general/tinyxml2.{cpp,h}) -- zlib license
* Zstr (general/zstr.hpp) -- MIT license
-5
View File
@@ -35,7 +35,6 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
set(MFEM_USE_CUDSS @MFEM_USE_CUDSS@)
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
@@ -110,10 +109,6 @@ if (MFEM_USE_RAJA)
find_dependency(RAJA)
endif()
if (MFEM_USE_CUDSS)
find_dependency(cudss)
endif (MFEM_USE_CUDSS)
if (MFEM_USE_UMPIRE)
find_dependency(umpire)
endif()
-9
View File
@@ -108,15 +108,6 @@
// Enable MFEM functionality based on the STRUMPACK library.
#cmakedefine MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
#cmakedefine MFEM_USE_CUDSS
// CUDSS communication layer library path
#cmakedefine MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
#cmakedefine MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable functionality based on the Ginkgo library.
#cmakedefine MFEM_USE_GINKGO
-68
View File
@@ -1,68 +0,0 @@
if (NOT cudss_DIR AND CUDSS_DIR)
set(cudss_DIR ${CUDSS_DIR}/lib/cmake/cudss)
endif()
message(STATUS "Looking for CUDSS ...")
message(STATUS " in CUDSS_DIR = ${CUDSS_DIR}")
message(STATUS " cudss_DIR = ${cudss_DIR}")
find_package(cudss)
set(CUDSS_FOUND ${cudss_FOUND})
set(CUDSS_LIBRARIES "cudss")
if (CUDSS_FOUND)
message(STATUS
"Found CUDSS target: ${CUDSS_LIBRARIES} (version: ${cudss_VERSION})")
else()
set(msg STATUS)
if (CUDSS_FIND_REQUIRED)
set(msg FATAL_ERROR)
endif()
message(${msg}
"CUDSS not found. Please set CUDSS_DIR to the install prefix.")
endif()
if(CUDSS_FOUND AND TARGET cudss)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION)
if(NOT CUDSS_LIBRARY_LOCATION)
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION_RELEASE)
endif()
if(CUDSS_LIBRARY_LOCATION)
get_filename_component(CUDSS_LIBRARY_DIR "${CUDSS_LIBRARY_LOCATION}" DIRECTORY)
else()
message(WARNING "Could not determine the location of the cuDSS library.")
endif()
else()
message(WARNING "cuDSS target not available; cannot determine library directory.")
endif()
# Set the full name of the cuDSS threading library if OpenMP is enabled.
# The threading layer library (libcudss_mtlayer_gomp.so) is located under the
# cuDSS library directory by default.
if (MFEM_USE_OPENMP)
find_file(
CUDSS_THREADING_LIB
NAMES libcudss_mtlayer_gomp.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_THREADING_LIB AND CUDSS_THREADING_LIB)
set(MFEM_CUDSS_THREADING_LIB "${CUDSS_THREADING_LIB}")
endif()
message(STATUS "CUDSS threading layer library: ${MFEM_CUDSS_THREADING_LIB}")
endif()
# Set the full name of the cuDSS communication library if MFEM use OpenMPI.
# The communication layer library (libcudss_commlayer_mpi.so) is located under the
# cuDSS library directory by default.
# The communication layer library is used pre-built communication layers for OpenMPI
# by default.
if (MFEM_USE_MPI)
find_file(
CUDSS_COMM_LIB
NAMES libcudss_commlayer_openmpi.so
PATHS ${CUDSS_LIBRARY_DIR}
NO_DEFAULT_PATH
)
if (NOT DEFINED MFEM_CUDSS_COMM_LIB AND CUDSS_COMM_LIB)
set(MFEM_CUDSS_COMM_LIB "${CUDSS_COMM_LIB}")
endif()
message(STATUS "CUDSS communication layer library: ${MFEM_CUDSS_COMM_LIB}")
endif()
+10 -8
View File
@@ -18,17 +18,19 @@
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
enable_language(C)
set(GSLIB_FETCH_VERSION 1.0.9)
add_library(GSLIB STATIC IMPORTED)
# set options (technically flags because GSLIB does not use cmake)
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(GSLIB_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
if (BUILD_SHARED_LIBS)
set(GSLIB_FLAGS "${GSLIB_FLAGS} -fPIC")
set(GSLIB_FETCH_VERSION 1.0.9)
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
if (CMAKE_C_FLAGS)
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
endif()
if (BUILD_SHARED_LIBS)
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
endif()
add_library(GSLIB STATIC IMPORTED)
# define external project and create future include directory so it is present
# to pass CMake checks at end of MFEM configuration step
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_FLAGS}")
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
include(ExternalProject)
ExternalProject_Add(gslib
@@ -38,7 +40,7 @@ if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND ""
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS=${GSLIB_FLAGS}"
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
INSTALL_COMMAND "")
file(MAKE_DIRECTORY ${PREFIX}/include)
# set imported library target properties
+1 -3
View File
@@ -44,9 +44,6 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
# set options and associated dependencies
set(HYPRE_CMAKE_OPTIONS "")
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
if (BUILD_SHARED_LIBS)
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_POSITION_INDEPENDENT_CODE:BOOL=ON)
endif()
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
get_cmake_property(all_vars VARIABLES)
foreach(var ${all_vars})
@@ -98,6 +95,7 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
SOURCE_SUBDIR src
PREFIX ${HYPRE_INSTALL}
BUILD_COMMAND ${CMAKE_COMMAND} --build . -- -j${CMAKE_BUILD_PARALLEL_LEVEL}
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${HYPRE_INSTALL} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${HYPRE_CMAKE_OPTIONS})
file(MAKE_DIRECTORY ${HYPRE_INSTALL}/include)
# set imported library target properties
+2 -10
View File
@@ -19,18 +19,10 @@
# - METIS_VERSION_5 (cache variable)
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
enable_language(C)
set(METIS_FETCH_VERSION 4.0.3)
add_library(METIS STATIC IMPORTED)
# set options (technically flags because METIS does not use cmake)
set(METIS_FLAGS "-Wno-implicit-int -Wno-incompatible-pointer-types")
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(METIS_FLAGS "${METIS_FLAGS} ${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
if (BUILD_SHARED_LIBS)
set(METIS_FLAGS "${METIS_FLAGS} -fPIC")
endif()
# define external project
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with ${METIS_FLAGS}")
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with default options")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/metis)
include(ExternalProject)
ExternalProject_Add(metis
@@ -40,7 +32,7 @@ if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND tar -xzf ../metis/metis-${METIS_FETCH_VERSION}-mac.tgz --strip=1
BUILD_COMMAND $(MAKE) clean && $(MAKE) "OPTFLAGS=${METIS_FLAGS}"
BUILD_COMMAND $(MAKE) COPTIONS=-Wno-incompatible-pointer-types
INSTALL_COMMAND mkdir -p ${PREFIX}/lib && cp libmetis.a ${PREFIX}/lib/)
# set imported library target properties
add_dependencies(METIS metis)
+9 -9
View File
@@ -22,15 +22,15 @@ include(MfemCmakeUtilities)
mfem_find_package(SuiteSparse SuiteSparse SuiteSparse_DIR "" "" "" ""
"Paths to headers required by SuiteSparse."
"Libraries required by SuiteSparse."
ADD_COMPONENT "UMFPACK" "include;include/suitesparse;suitesparse" umfpack.h "lib" umfpack
ADD_COMPONENT "KLU" "include;include/suitesparse;suitesparse" klu.h "lib" klu
ADD_COMPONENT "AMD" "include;include/suitesparse;suitesparse" amd.h "lib" amd
ADD_COMPONENT "BTF" "include;include/suitesparse;suitesparse" btf.h "lib" btf
ADD_COMPONENT "CHOLMOD" "include;include/suitesparse;suitesparse" cholmod.h "lib" cholmod
ADD_COMPONENT "COLAMD" "include;include/suitesparse;suitesparse" colamd.h "lib" colamd
ADD_COMPONENT "CAMD" "include;include/suitesparse;suitesparse" camd.h "lib" camd
ADD_COMPONENT "CCOLAMD" "include;include/suitesparse;suitesparse" ccolamd.h "lib" ccolamd
ADD_COMPONENT "config" "include;include/suitesparse;suitesparse" SuiteSparse_config.h "lib"
ADD_COMPONENT "UMFPACK" "include;suitesparse" umfpack.h "lib" umfpack
ADD_COMPONENT "KLU" "include;suitesparse" klu.h "lib" klu
ADD_COMPONENT "AMD" "include;suitesparse" amd.h "lib" amd
ADD_COMPONENT "BTF" "include;suitesparse" btf.h "lib" btf
ADD_COMPONENT "CHOLMOD" "include;suitesparse" cholmod.h "lib" cholmod
ADD_COMPONENT "COLAMD" "include;suitesparse" colamd.h "lib" colamd
ADD_COMPONENT "CAMD" "include;suitesparse" camd.h "lib" camd
ADD_COMPONENT "CCOLAMD" "include;suitesparse" ccolamd.h "lib" ccolamd
ADD_COMPONENT "config" "include;suitesparse" SuiteSparse_config.h "lib"
suitesparseconfig)
if (SuiteSparse_FOUND AND METIS_VERSION_5)
+16 -4
View File
@@ -157,10 +157,22 @@ constexpr real_t operator""_r(unsigned long long v)
#endif
#endif // MFEM_USE_MPI not defined
#ifndef MFEM_USE_CUDA
#ifdef MFEM_USE_CUDSS
#error Building with cuDSS (MFEM_USE_CUDSS=YES) requires CUDA (MFEM_USE_CUDA=YES)
#ifdef NVTX_DBG_HPP
#include NVTX_DBG_HPP
#else
#define db1(...)
#define dbg(...)
#define dbl(...)
#define dba(...)
#define dbc(...)
#define NVTX_MARK_FUNCTION
#define NVTX_MARK_BEGIN(...)
#define NVTX_INI(...)
#define NVTX_END(...)
#define NVTX_MARK_INI(...)
#define NVTX_MARK_END(...)
#define NVTX_MARK(...)
#define NVTX(...)
#endif
#endif // MFEM_USE_CUDSS not defined
#endif // MFEM_CONFIG_HPP
-9
View File
@@ -108,15 +108,6 @@
// Enable MFEM functionality based on the STRUMPACK library.
// #define MFEM_USE_STRUMPACK
// Enable MFEM functionality based on the cuDSS library.
// #define MFEM_USE_CUDSS
// CUDSS communication layer library path
// #define MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
// CUDSS threading layer library path
// #define MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
// Enable MFEM features based on the Ginkgo library.
// #define MFEM_USE_GINKGO
-3
View File
@@ -36,9 +36,6 @@ MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_SUPERLU5 = @MFEM_USE_SUPERLU5@
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_CUDSS = @MFEM_USE_CUDSS@
MFEM_CUDSS_COMM_LIB = @MFEM_CUDSS_COMM_LIB@
MFEM_CUDSS_THREADING_LIB = @MFEM_CUDSS_THREADING_LIB@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_AMGX = @MFEM_USE_AMGX@
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
-1
View File
@@ -38,7 +38,6 @@ option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
option(MFEM_USE_SUPERLU5 "Use the old SuperLU_DIST 5.1 version" OFF)
option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
option(MFEM_USE_CUDSS "Enable cuDSS usage" OFF)
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
+1 -15
View File
@@ -153,7 +153,6 @@ MFEM_USE_SUPERLU = NO
MFEM_USE_SUPERLU5 = NO
MFEM_USE_MUMPS = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_CUDSS = NO
MFEM_USE_GINKGO = NO
MFEM_USE_AMGX = NO
MFEM_USE_MAGMA = NO
@@ -369,19 +368,6 @@ STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
$(SCOTCH_LIB) $(SCALAPACK_LIB)
# CUDSS library configuration
CUDSS_DIR = @MFEM_DIR@/../cudss
CUDSS_INCLUDE_DIR = $(CUDSS_DIR)/include
CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
CUDSS_LIB = \
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
# The cuDSS communication and threading libraries.
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
$(subst @MFEM_DIR@,$(MFEM_DIR),$(CUDSS_LIBRARY_DIR)/libcudss_mtlayer_gomp.so))))
# Ginkgo library configuration
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
GINKGO_SEARCH_DIR = $(subst @MFEM_DIR@,$(MFEM_DIR),$(GINKGO_DIR))
@@ -635,7 +621,7 @@ PARELAG_LIB = -L$(PARELAG_DIR)/build/src -lParELAG
AXOM_DIR = @MFEM_DIR@/../axom
TRIBOL_DIR = @MFEM_DIR@/../tribol
TRIBOL_OPT = -I$(TRIBOL_DIR)/include -I$(AXOM_DIR)/include
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -ltribol_shared -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
-laxom_slam -laxom_slic -laxom_core
# Enzyme configuration
+2 -1
View File
@@ -47,6 +47,7 @@ list(APPEND ALL_EXE_SRCS
ex39.cpp
ex40.cpp
ex41.cpp
jitplayground.cpp
)
if (MFEM_USE_MPI)
@@ -215,7 +216,7 @@ if (MFEM_ENABLE_TESTING)
add_test(NAME ex1p_ceed_np=${MFEM_MPI_NP}
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
${MPIEXEC_PREFLAGS}
$<TARGET_FILE:ex1p> "-no-vis" "-d" "ceed-cpu" "-pa" "-a"
$<TARGET_FILE:ex1p> "-no-vis" "-d ceed-cpu" "-pa" "-a"
${MPIEXEC_POSTFLAGS})
endif()
endif()
+1 -1
View File
@@ -64,7 +64,7 @@ PARALLEL_NAME := Parallel AMGX example
$(MFEM_LIB_FILE):
$(error The MFEM library is not build)
clean: clean-build clean-exec
clean: clean-build
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
+3 -3
View File
@@ -64,12 +64,12 @@ ex1p-test-par: ex1p
$(MFEM_LIB_FILE):
$(error The MFEM library is not built)
clean: clean-build clean-exec
clean: clean-build clean-exec $(SUBDIRS_CLEAN)
clean-build:
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh mesh.*
@rm -f sol.*
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
+21 -34
View File
@@ -50,10 +50,6 @@
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cpu
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
//
// Device simplices sample runs:
// ex1 -pa -d gpu -m ../data/inline-tet.mesh
// ex1 -pa -d gpu -m ../data/inline-tri.mesh
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
@@ -142,25 +138,25 @@ int main(int argc, char *argv[])
}
// 5. Define a finite element space on the mesh. Here we use continuous
// Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
// Lagrange finite elements of the specified order. If order < 1, we
// instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
auto basis_type = (pa && mesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim, basis_type);
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (mesh.GetNodes())
{
fec = mesh.GetNodes()->OwnFEC();
delete_fec = false;
cout << "Using isoparametric FEs: " << fec->Name() << endl;
}
else
{
fec = new H1_FECollection(order = 1, dim, basis_type);
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
FiniteElementSpace fespace(&mesh, fec);
cout << "Number of finite element unknowns: "
@@ -228,29 +224,17 @@ int main(int argc, char *argv[])
// 11. Solve the linear system A X = B.
if (!pa)
{
#ifdef MFEM_USE_CUDSS
if (Device::Allows(Backend::CUDA_MASK))
{
// Use cuDSS to solve the system.
CuDSSSolver cudss_solver;
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
#ifndef MFEM_USE_SUITESPARSE
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
GSSmoother M((SparseMatrix&)(*A));
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
#else
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
UMFPackSolver umf_solver;
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
umf_solver.SetOperator(*A);
umf_solver.Mult(B, X);
#endif
}
}
else
{
@@ -289,14 +273,17 @@ int main(int argc, char *argv[])
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "solution\n" << mesh << x << flush;
}
// 15. Free the used memory.
if (order > 0) { delete fec; }
if (delete_fec)
{
delete fec;
}
return 0;
}
+34 -60
View File
@@ -42,11 +42,7 @@
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/square-mixed.mesh
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/fichera-mixed.mesh
// mpirun -np 4 ex1p -pa -d ceed-cpu -m ../data/beam-tet.mesh
//
// Device simplices sample runs:
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tet.mesh
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tri.mesh
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
//
// Description: This example code demonstrates the use of MFEM to define a
// simple finite element discretization of the Poisson problem
@@ -87,9 +83,6 @@ int main(int argc, char *argv[])
const char *device_config = "cpu";
bool visualization = true;
bool algebraic_ceed = false;
#ifdef MFEM_USE_CUDSS
bool cudss_solver = false;
#endif
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -109,10 +102,6 @@ int main(int argc, char *argv[])
args.AddOption(&algebraic_ceed, "-a", "--algebraic",
"-no-a", "--no-algebraic",
"Use algebraic Ceed solver");
#endif
#ifdef MFEM_USE_CUDSS
args.AddOption(&cudss_solver, "-cudss", "--cudss-solver", "-no-cudss",
"--no-cudss-solver", "Use the cuDSS Solver.");
#endif
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
@@ -169,20 +158,19 @@ int main(int argc, char *argv[])
}
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange finite elements of the specified order.
// - If order < 1, we instead use an isoparametric/isogeometric space.
// - If the mesh is simplicial and partial assembly is requested,
// we use the positive basis, which supports device execution.
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec;
auto basis_type = (pa && pmesh.IsSimplexMesh()) ?
BasisType::Positive : BasisType::GaussLobatto;
bool delete_fec;
if (order > 0)
{
fec = new H1_FECollection(order, dim, basis_type);
fec = new H1_FECollection(order, dim);
delete_fec = true;
}
else if (pmesh.GetNodes())
{
fec = pmesh.GetNodes()->OwnFEC();
delete_fec = false;
if (myid == 0)
{
cout << "Using isoparametric FEs: " << fec->Name() << endl;
@@ -190,7 +178,8 @@ int main(int argc, char *argv[])
}
else
{
fec = new H1_FECollection(order = 1, dim, basis_type);
fec = new H1_FECollection(order = 1, dim);
delete_fec = true;
}
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_BigInt size = fespace.GlobalTrueVSize();
@@ -259,51 +248,33 @@ int main(int argc, char *argv[])
// 13. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use Jacobi smoothing, for now.
#ifdef MFEM_USE_CUDSS
if (!pa && (Device::Allows(Backend::CUDA_MASK) && cudss_solver))
Solver *prec = NULL;
if (pa)
{
// Solve using a direct solver with cuDSS
CuDSSSolver cudss_solver(MPI_COMM_WORLD);
cudss_solver.SetMatrixSymType(
CuDSSSolver::SYMMETRIC_POSITIVE_DEFINITE);
cudss_solver.SetMatrixViewType(CuDSSSolver::UPPER);
cudss_solver.SetOperator(*A);
cudss_solver.Mult(B, X);
}
else
#endif
{
Solver *prec = NULL;
if (pa)
if (UsesTensorBasis(fespace))
{
if (UsesTensorBasis(fespace))
if (algebraic_ceed)
{
if (algebraic_ceed)
{
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
}
else
{
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
}
}
else
{
prec = new HypreBoomerAMG;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec)
{
cg.SetPreconditioner(*prec);
}
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
}
else
{
prec = new HypreBoomerAMG;
}
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-12);
cg.SetMaxIter(2000);
cg.SetPrintLevel(1);
if (prec) { cg.SetPreconditioner(*prec); }
cg.SetOperator(*A);
cg.Mult(B, X);
delete prec;
// 14. Recover the parallel grid function corresponding to X. This is the
// local finite element solution on each processor.
@@ -337,7 +308,10 @@ int main(int argc, char *argv[])
}
// 17. Free the used memory.
if (order > 0) { delete fec; }
if (delete_fec)
{
delete fec;
}
return 0;
}
-9
View File
@@ -95,15 +95,6 @@ int main(int argc, char *argv[])
args.PrintOptions(cout);
}
if (amg_elast && !static_cond && reorder_space)
{
if (myid == 0)
cerr << "\nThe AMG elasticity solver requires ordering byVDIM! "
<< "Ignoring the specified option -nodes/--by-nodes.\n"
<< endl;
reorder_space = false;
}
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
+1 -1
View File
@@ -76,4 +76,4 @@ clean-build:
rm -rf *.dSYM *.TVD.*breakpoints
clean-exec:
@rm -f refined.mesh sol.gf mesh.* sol.*
@rm -f refined.mesh sol.gf
+536
View File
@@ -0,0 +1,536 @@
#include <mfem.hpp>
#include "../fem/dfem/util.hpp"
#include <proteus/CppJitModule.h>
#include "jitplayground.hpp"
#include <algorithm>
#include <array>
#include <cctype>
#include <cmath>
#include <fstream>
#include <initializer_list>
#include <iostream>
#include <memory>
#include <sstream>
#include <string>
#include <string_view>
#include <type_traits>
#include <unordered_map>
#include <unordered_set>
#include <utility>
#include <vector>
namespace util
{
constexpr std::string_view Dirname(std::string_view path)
{
const size_t last_sep = path.find_last_of("/\\");
if (last_sep == std::string_view::npos) { return {}; }
return path.substr(0, last_sep);
}
constexpr std::string_view thisFileDir = Dirname(__FILE__);
}
template <typename T>
static std::string TypeNameString()
{
return std::string(mfem::future::get_type_name<T>());
}
template <typename Tuple, size_t... Is>
static auto ParamTypeStringsImpl(std::index_sequence<Is...>)
{
return std::array<std::string, sizeof...(Is)>
{
TypeNameString<std::remove_reference_t<decltype(mfem::future::get<Is>(std::declval<Tuple&>()))>>()...
};
}
template <typename Tuple>
static auto ParamTypeStrings()
{
return ParamTypeStringsImpl<Tuple>(
std::make_index_sequence<mfem::future::tuple_size<Tuple>::value> {});
}
static std::string_view Trim(std::string_view s)
{
size_t begin = 0;
while (begin < s.size() && std::isspace(static_cast<unsigned char>(s[begin])))
{
++begin;
}
size_t end = s.size();
while (end > begin &&
std::isspace(static_cast<unsigned char>(s[end - 1])))
{
--end;
}
return s.substr(begin, end - begin);
}
static bool IsValidIdentifier(std::string_view s)
{
if (s.empty()) { return false; }
const unsigned char c0 = static_cast<unsigned char>(s[0]);
if (!(std::isalpha(c0) || c0 == '_')) { return false; }
for (size_t i = 1; i < s.size(); ++i)
{
const unsigned char c = static_cast<unsigned char>(s[i]);
if (!(std::isalnum(c) || c == '_')) { return false; }
}
return true;
}
static bool ParseJitDirective(std::string_view line,
std::string &type,
std::string &var,
std::string &kind)
{
const size_t jit_pos = line.find("$JIT");
if (jit_pos == std::string_view::npos) { return false; }
const size_t open = line.find('[', jit_pos);
const size_t close = line.find(']', jit_pos);
MFEM_VERIFY(open != std::string_view::npos &&
close != std::string_view::npos &&
close > open,
"malformed $JIT directive (expected brackets): " << line);
const std::string_view payload = line.substr(open + 1, close - open - 1);
const size_t comma1 = payload.find(',');
const size_t comma2 = (comma1 == std::string_view::npos)
? std::string_view::npos
: payload.find(',', comma1 + 1);
MFEM_VERIFY(comma1 != std::string_view::npos &&
comma2 != std::string_view::npos,
"malformed $JIT directive (expected 3 comma-separated fields): "
<< line);
const std::string_view f0 = Trim(payload.substr(0, comma1));
const std::string_view f1 = Trim(payload.substr(comma1 + 1,
comma2 - comma1 - 1));
const std::string_view f2 = Trim(payload.substr(comma2 + 1));
MFEM_VERIFY(!f0.empty() && !f1.empty() && !f2.empty(),
"malformed $JIT directive (empty field): " << line);
type.assign(f0);
var.assign(f1);
kind.assign(f2);
return true;
}
static std::string ReadFileOrEmpty(const std::string &fn)
{
std::ifstream file(fn);
if (!file.is_open())
{
std::cerr << "could not open file " << fn << "\n";
return {};
}
std::stringstream buffer;
buffer << file.rdbuf();
return buffer.str();
}
static std::vector<std::string> ExtractJitVarNames(const std::string
&kernel_code)
{
std::stringstream ss(kernel_code);
std::string line;
std::vector<std::string> var_names;
std::unordered_set<std::string> seen_vars;
while (std::getline(ss, line))
{
std::string type, var, kind;
if (ParseJitDirective(line, type, var, kind))
{
MFEM_VERIFY(IsValidIdentifier(var),
"$JIT variable must be a valid identifier: " << var);
MFEM_VERIFY(seen_vars.insert(var).second,
"duplicate $JIT variable name: " << var);
var_names.push_back(var);
}
}
return var_names;
}
static std::string RewriteKernelForJit(std::string kernel_code,
const std::vector<std::string> &jit_values)
{
std::stringstream ss(kernel_code);
std::string line;
std::string out;
out.reserve(kernel_code.size() + 128);
bool have_pending = false;
size_t pending_index = 0;
std::string pending_type;
std::string pending_var;
std::unordered_set<std::string> seen_vars;
while (std::getline(ss, line))
{
line.push_back('\n');
if (have_pending)
{
MFEM_VERIFY(pending_index < jit_values.size(),
"not enough JIT values provided");
const size_t indent_end = line.find_first_not_of(" \t");
const std::string indent =
(indent_end == std::string::npos) ? std::string() :
line.substr(0, indent_end);
out += indent + "const " + pending_type + " " + pending_var + " = " +
jit_values[pending_index] + ";\n";
have_pending = false;
++pending_index;
continue;
}
std::string type, var, kind;
if (ParseJitDirective(line, type, var, kind))
{
MFEM_VERIFY(IsValidIdentifier(var),
"$JIT variable must be a valid identifier: " << var);
MFEM_VERIFY(kind == "generic",
"unsupported $JIT kind: " << kind);
MFEM_VERIFY(seen_vars.insert(var).second,
"duplicate $JIT variable name: " << var);
pending_type = std::move(type);
pending_var = std::move(var);
have_pending = true;
continue; // drop directive line
}
out += line;
}
MFEM_VERIFY(!have_pending,
"$JIT directive must annotate a following line");
MFEM_VERIFY(jit_values.size() == pending_index,
"JIT value count must match number of $JIT directives");
return out;
}
static std::string GeneratedOutputPath(std::string_view original_path)
{
const size_t last_sep = original_path.find_last_of("/\\");
const size_t dot = original_path.find_last_of('.');
const bool dot_in_filename =
(dot != std::string_view::npos) &&
(last_sep == std::string_view::npos || dot > last_sep);
const std::string_view base =
dot_in_filename ? original_path.substr(0, dot) : original_path;
return std::string(base) + "_generated.hpp";
}
static void WriteFileOrWarn(const std::string &path,
const std::string &contents)
{
std::ofstream out(path);
if (!out.is_open())
{
std::cerr << "could not write generated file " << path << "\n";
return;
}
out << contents;
}
class JitQFunction
{
public:
template <typename ImplT, size_t N>
JitQFunction(ImplT, const std::string &fn,
const std::array<bool, N> &activity_map)
{
using qf_signature = typename
mfem::future::get_function_signature<
decltype(&ImplT::operator())>::type;
using qf_param_ts = typename qf_signature::parameter_ts;
constexpr size_t nparams = mfem::future::tuple_size<qf_param_ts>::value;
static_assert(N == nparams, "activity_map size must match qfunc arity");
this->fn = fn;
this->nparams = nparams;
this->activity_map.reserve(N);
for (size_t i = 0; i < N; ++i)
{
this->activity_map.push_back(activity_map[i]);
}
{
const auto param_types_arr = ParamTypeStrings<qf_param_ts>();
this->param_types.assign(param_types_arr.begin(), param_types_arr.end());
}
this->return_type = TypeNameString<typename qf_signature::return_t>();
this->return_is_void = std::is_same_v<typename qf_signature::return_t, void>;
this->impl_type_name = TypeNameString<ImplT>();
this->jit_var_names = ExtractJitVarNames(ReadFileOrEmpty(fn));
}
template <typename ReturnT, typename... Args>
ReturnT run(std::string_view name,
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
Args&&... args)
{
auto ordered_values = MatchJitValues(jit_values);
auto &mod = GetOrCreateModule(ordered_values);
auto &instance = mod.instantiate(std::string(name), std::string());
return instance.template run<ReturnT>(std::forward<Args>(args)...);
}
template <typename ReturnT, typename... Args>
ReturnT run_primal(
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
Args&&... args)
{
return run<ReturnT>(qfunc_name, jit_values,
std::forward<Args>(args)...);
}
template <typename ReturnT, typename... Args>
ReturnT run_derivative(
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
Args&&... args)
{
return run<ReturnT>(qfunc_name + "_fwddiff", jit_values,
std::forward<Args>(args)...);
}
private:
std::vector<std::string_view> MatchJitValues(
std::initializer_list<std::pair<std::string_view, std::string_view>>
named_values) const
{
std::unordered_map<std::string_view, std::string_view> value_map;
for (const auto &[name, value] : named_values)
{
value_map[name] = value;
}
std::vector<std::string_view> ordered_values;
ordered_values.reserve(jit_var_names.size());
for (const auto &var_name : jit_var_names)
{
auto it = value_map.find(var_name);
MFEM_VERIFY(it != value_map.end(),
"missing JIT value for variable: " << var_name);
ordered_values.push_back(it->second);
}
MFEM_VERIFY(ordered_values.size() == named_values.size(),
"provided " << named_values.size() << " JIT values but expected "
<< jit_var_names.size());
return ordered_values;
}
std::string BuildModuleCode(const std::vector<std::string> &jit_values) const
{
std::string module_code =
RewriteKernelForJit(ReadFileOrEmpty(fn), jit_values);
module_code += "\n\n";
module_code += "// --- generated ---\n";
module_code +=
"template <typename return_type, typename... Args>\n"
"return_type __enzyme_fwddiff(Args...);\n"
"\n"
"extern int enzyme_const;\n"
"extern int enzyme_dup;\n"
"\n";
// Generate a primal wrapper with the requested symbol name, so the kernel
// header can just define the qfunc as a functor.
//
// Note: Proteus instantiates entrypoints via `qfunc_wrapper<>(...)` even
// when there are no user template args, so keep the wrapper itself a
// template (with a default parameter) while still doing literal `$JIT`
// replacements in the kernel code.
module_code += "template <typename = void>\n";
module_code += return_type + " " +
std::string(qfunc_name) + "(";
bool first = true;
for (size_t i = 0; i < nparams; ++i)
{
if (!first) { module_code += ", "; }
first = false;
module_code += param_types[i] + " Arg" + std::to_string(i);
}
module_code += ")\n";
module_code += "{\n";
module_code += " " + impl_type_name + " qf;\n";
if (return_is_void)
{
module_code += " ";
}
else
{
module_code += " return ";
}
module_code += "qf(";
for (size_t i = 0; i < nparams; ++i)
{
if (i) { module_code += ", "; }
module_code += "Arg" + std::to_string(i);
}
module_code += ");\n";
module_code += "}\n\n";
module_code += "template <typename = void>\n";
module_code += return_type + " " +
std::string(qfunc_name) + "_fwddiff(";
first = true;
for (size_t i = 0; i < nparams; ++i)
{
if (!first) { module_code += ", "; }
first = false;
module_code += param_types[i] + " Arg" + std::to_string(i);
if (activity_map[i])
{
module_code += ", " + param_types[i] + " dArg" + std::to_string(i);
}
}
module_code += ")\n";
module_code += "{\n";
if (return_is_void)
{
module_code += " __enzyme_fwddiff<void>(\n";
}
else
{
module_code += " return __enzyme_fwddiff<" +
return_type + ">(\n";
}
module_code += " (void*)" + std::string(qfunc_name) + "<>";
module_code += ",\n";
for (size_t i = 0; i < nparams; ++i)
{
if (activity_map[i])
{
module_code += " enzyme_dup, Arg" + std::to_string(i) +
", dArg" + std::to_string(i);
}
else
{
module_code += " enzyme_const, Arg" + std::to_string(i);
}
module_code += (i + 1 == nparams) ? ");\n" : ",\n";
}
module_code += "}\n";
WriteFileOrWarn(GeneratedOutputPath(fn), module_code);
return module_code;
}
proteus::CppJitModule &GetOrCreateModule(
const std::vector<std::string_view> &jit_values)
{
std::string key;
for (const auto &val : jit_values)
{
if (!key.empty()) { key += ","; }
key += val;
}
auto it = modules.find(key);
if (it != modules.end())
{
return *it->second;
}
std::vector<std::string> values(jit_values.begin(), jit_values.end());
std::string code = BuildModuleCode(values);
auto mod = std::make_unique<proteus::CppJitModule>("host", code,
DefaultExtraArgs());
auto [inserted, ok] = modules.emplace(key, std::move(mod));
MFEM_VERIFY(ok, "failed to cache JIT module");
return *inserted->second;
}
static std::vector<std::string> DefaultExtraArgs()
{
return {"-fplugin=/Users/andrej1/local/enzyme/lib/ClangEnzyme-20.dylib"};
}
std::string qfunc_name = "qfunc_wrapper";
std::string fn;
size_t nparams = 0;
std::vector<bool> activity_map;
std::vector<std::string> param_types;
std::string return_type;
bool return_is_void = false;
std::string impl_type_name;
std::vector<std::string> jit_var_names;
std::unordered_map<std::string, std::unique_ptr<proteus::CppJitModule>> modules;
};
int main()
{
const size_t N = 4;
const size_t M = 5;
const double A = 123.4;
std::vector<double> X(N);
std::vector<double> Y(N);
for (size_t i = 0; i < N; ++i)
{
X[i] = static_cast<double>(i + 1);
Y[i] = static_cast<double>(N - i);
}
// // >>> user interface calls
// const std::string kernel_path = std::string(util::thisFileDir) +
// "/jitplayground.hpp";
// JitQFunction qf(daxpy_op{}, kernel_path, std::array{false, true, false});
// // <<< user interface calls
// // this will happen internally in dFEM
daxpy_op op;
printf("\n\nfunction call\n");
op(&A, X.data(), Y.data(), &N);
// reset X for the derivative test
for (size_t i = 0; i < N; ++i)
{
X[i] = static_cast<double>(i + 1);
Y[i] = static_cast<double>(N - i);
}
std::vector<double> dX(N, 1.0);
printf("\n\nforward diff call\n");
daxpy_op_fwddiff(&A, X.data(), dX.data(), Y.data(), &N);
std::vector<double> dX_manual(N, A);
printf("\n\nderivative checks\n");
std::cout << "dX: ";
for (size_t i = 0; i < N; ++i)
{
std::cout << dX[i] << (i + 1 == N ? '\n' : ' ');
}
std::cout << "dX_manual: ";
for (size_t i = 0; i < N; ++i)
{
std::cout << dX_manual[i] << (i + 1 == N ? '\n' : ' ');
}
double max_abs_err = 0.0;
for (size_t i = 0; i < N; ++i)
{
max_abs_err = std::max(max_abs_err, std::abs(dX[i] - dX_manual[i]));
}
std::cout << "max |dX - dX_manual| = " << max_abs_err << "\n";
return 0;
}
+58
View File
@@ -0,0 +1,58 @@
#pragma once
#include <cstddef>
#include <vector>
#include <type_traits>
#include "proteus/JitInterface.h"
struct daxpy_op
{
void operator()(
const double *a,
double *x,
const double *y,
const size_t *N) const
{
const size_t n = *N;
auto lam = [=, n = proteus::jit_variable(n)]
() __attribute__((annotate("jit")))
{
printf("N = %zu\n", n);
for (size_t i = 0; i < n; ++i)
{
printf("x[%zu] = %f, y[%zu] = %f\n", i, x[i], i, y[i]);
x[i] = *a * x[i] + y[i];
printf("updated x[%zu] = %f\n", i, x[i]);
}
};
proteus::register_lambda(lam);
lam();
}
};
template <typename return_type, typename... Args>
return_type __enzyme_fwddiff(Args...);
extern int enzyme_const;
extern int enzyme_dup;
void daxpy_op_wrapper(const double * Arg0, double * Arg1,
const double * Arg2, const size_t *Arg3)
{
daxpy_op qf;
qf(Arg0, Arg1, Arg2, Arg3);
}
void daxpy_op_fwddiff(const double * Arg0, double * Arg1,
double * dArg1, const double * Arg2, const size_t *Arg3)
{
__enzyme_fwddiff<void>(
(void*)daxpy_op_wrapper,
enzyme_const, Arg0,
enzyme_dup, Arg1, dArg1,
enzyme_const, Arg2,
enzyme_const, Arg3);
}
+2 -5
View File
@@ -71,7 +71,6 @@ endif
SUBDIRS_ALL = $(addsuffix /all,$(SUBDIRS))
SUBDIRS_TEST = $(addsuffix /test,$(SUBDIRS))
SUBDIRS_TEST_NOCLEAN = $(addsuffix /test-noclean,$(SUBDIRS))
SUBDIRS_CLEAN = $(addsuffix /clean,$(SUBDIRS))
SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
@@ -88,9 +87,8 @@ SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
all: $(EXAMPLES) $(SUBDIRS_ALL)
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) \
$(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) $(SUBDIRS_CLEAN):
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN):
$(MAKE) -C $(@D) $(@F)
$(SUBDIRS_TPRINT):
@$(MAKE) -C $(@D) $(@F)
@@ -109,7 +107,6 @@ endif
MFEM_TESTS = EXAMPLES
include $(MFEM_TEST_MK)
test: $(SUBDIRS_TEST)
test-noclean: $(SUBDIRS_TEST_NOCLEAN)
test-print: $(SUBDIRS_TPRINT)
# Testing: Parallel vs. serial runs
+19 -19
View File
@@ -121,6 +121,11 @@ set(SRCS
qinterp/eval_hdiv.cpp
qinterp/grad_by_nodes.cpp
qinterp/grad_by_vdim.cpp
qinterp/grad_transpose.cpp
qinterp/grad_transpose_by_nodes.cpp
qinterp/grad_transpose_by_vdim.cpp
qinterp/eval_transpose.cpp
qinterp/eval_transpose_by_vdim.cpp
qspace.cpp
quadinterpolator.cpp
quadinterpolator_face.cpp
@@ -171,12 +176,8 @@ set(SRCS
tmop_tools.cpp
tmop_amr.cpp
gslib.cpp
gslib/findptsedge_local_2.cpp
gslib/findptsedge_local_3.cpp
gslib/findptssurf_local_3.cpp
gslib/findpts_local_2.cpp
gslib/findpts_local_3.cpp
gslib/interpolate_local_1.cpp
gslib/interpolate_local_2.cpp
gslib/interpolate_local_3.cpp
transfer.cpp
@@ -195,14 +196,12 @@ set(HDRS
integ/bilininteg_dgtrace_kernels.hpp
integ/bilininteg_vecdiffusion_kernels.hpp
integ/bilininteg_convection_kernels.hpp
integ/bilininteg_diffusion_pa_simplices.hpp
integ/bilininteg_diffusion_kernels.hpp
integ/bilininteg_elasticity_kernels.hpp
integ/bilininteg_hcurl_kernels.hpp
integ/bilininteg_hdiv_kernels.hpp
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
integ/bilininteg_mass_pa_simplices.hpp
integ/bilininteg_vecdiffusion_pa.hpp
integ/bilininteg_vecmass_pa.hpp
coefficient.hpp
@@ -284,8 +283,10 @@ set(HDRS
qfunction.hpp
qinterp/det.hpp
qinterp/eval.hpp
qinterp/eval_transpose.hpp
qinterp/eval_hdiv.hpp
qinterp/grad.hpp
qinterp/grad_transpose.hpp
qspace.hpp
quadinterpolator.hpp
quadinterpolator_face.hpp
@@ -311,7 +312,6 @@ set(HDRS
tmop_tools.hpp
tmop_amr.hpp
gslib.hpp
gslib/gslib_kernel_helpers.hpp
transfer.hpp
hyperbolic.hpp
integrator.hpp
@@ -320,36 +320,36 @@ set(HDRS
)
if (MFEM_USE_SIDRE)
list(APPEND SRCS sidredatacollection.cpp)
list(APPEND HDRS sidredatacollection.hpp)
list(APPEND SRCS sidredatacollection.cpp)
list(APPEND HDRS sidredatacollection.hpp)
endif()
if (MFEM_USE_CONDUIT)
list(APPEND SRCS conduitdatacollection.cpp)
list(APPEND HDRS conduitdatacollection.hpp)
list(APPEND SRCS conduitdatacollection.cpp)
list(APPEND HDRS conduitdatacollection.hpp)
endif()
if (MFEM_USE_ADIOS2)
list(APPEND SRCS adios2datacollection.cpp)
list(APPEND HDRS adios2datacollection.hpp)
list(APPEND SRCS adios2datacollection.cpp)
list(APPEND HDRS adios2datacollection.hpp)
endif()
if (MFEM_USE_FMS)
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
endif()
if (MFEM_USE_MPI)
list(APPEND SRCS
list(APPEND SRCS
pbilinearform.cpp
pfespace.cpp
pgridfunc.cpp
plinearform.cpp
pnonlinearform.cpp
prestriction.cpp)
# If this list (HDRS -> HEADERS) is used for install, we probably want the
# headers added all the time.
list(APPEND HDRS
# If this list (HDRS -> HEADERS) is used for install, we probably want the
# headers added all the time.
list(APPEND HDRS
pbilinearform.hpp
pfespace.hpp
pgridfunc.hpp
+4 -22
View File
@@ -1345,8 +1345,7 @@ real_t DiffusionIntegrator::ComputeFluxEnergy
}
const IntegrationRule &DiffusionIntegrator::GetRule(
const FiniteElement &trial_fe, const FiniteElement &test_fe,
const bool stroud)
const FiniteElement &trial_fe, const FiniteElement &test_fe)
{
int order;
if (trial_fe.Space() == FunctionSpace::Pk)
@@ -1363,15 +1362,7 @@ const IntegrationRule &DiffusionIntegrator::GetRule(
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
}
MassIntegrator::MassIntegrator(const IntegrationRule *ir)
@@ -1458,8 +1449,7 @@ void MassIntegrator::AssembleElementMatrix2(
const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans,
const bool stroud)
const ElementTransformation &Trans)
{
// int order = trial_fe.GetOrder() + test_fe.GetOrder();
const int order = trial_fe.GetOrder() + test_fe.GetOrder() + Trans.OrderW();
@@ -1468,15 +1458,7 @@ const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
{
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
}
if (stroud)
{
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
}
else
{
return IntRules.Get(trial_fe.GetGeomType(), order);
}
return IntRules.Get(trial_fe.GetGeomType(), order);
}
+19 -52
View File
@@ -2178,29 +2178,22 @@ class DiffusionIntegrator: public BilinearFormIntegrator
{
public:
using ApplyKernelType = void(*)(const int, const bool, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&,
const Vector&, const Vector&,
Vector&, const int, const int);
using DiffusionApplyKernelType = void(*)(const int, const bool,
const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&,
const Vector&, const Vector&,
Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const bool, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&,
Vector&, const int, const int);
using DiffusionDiagonalKernelType = void(*)(const int, const bool,
const Array<real_t>&,
const Array<real_t>&, const Vector&, Vector&,
const int, const int);
using DiagonalKernelType = void(*)(const int, const bool, const Array<real_t>&,
const Array<real_t>&, const Vector&, Vector&,
const int, const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(DiffusionApplyPAKernel, DiffusionApplyKernelType,
(int, int, int));
MFEM_REGISTER_KERNELS(DiffusionDiagonalPAKernel, DiffusionDiagonalKernelType,
(int, int, int));
struct Kernels { Kernels(); };
protected:
@@ -2220,6 +2213,7 @@ private:
const FiniteElementSpace *fespace;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
public:
int dim, ne, dofs1D, quad1D;
Vector pa_data;
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
@@ -2352,8 +2346,7 @@ public:
void AddMultPatchPA(const int patch, const Vector &x, Vector &y) const;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const bool stroud = false);
const FiniteElement &test_fe);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2362,15 +2355,8 @@ public:
template <int DIM, int D1D, int Q1D>
static void AddSpecialization()
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiffusionApplyPAKernel::Specialization<DIM,D1D,Q1D>::Add();
DiffusionDiagonalPAKernel::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
const IntegrationRule* GetDefaultIntegrationRule(
@@ -2407,22 +2393,11 @@ public:
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
using ApplySimplexKernelType = void(*)(const int, const Array<int>&,
const Array<int>&,
const Array<int>&, const Array<int>&, const Array<int>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Array<real_t>&, const Array<real_t>&,
const Vector&, const Vector&, Vector&,
const int, const int);
using DiagonalKernelType = void(*)(const int, const Array<real_t>&,
const Vector&, Vector&, const int,
const int);
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
int));
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
struct Kernels { Kernels(); };
@@ -2471,8 +2446,7 @@ public:
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans,
const bool stroud = false);
const ElementTransformation &Trans);
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
@@ -2483,13 +2457,6 @@ public:
{
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
AddSimplexSpecialization<DIM,D1D,Q1D>();
}
template <int DIM, int D1D, int Q1D>
static void AddSimplexSpecialization()
{
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
}
protected:
-6
View File
@@ -54,8 +54,6 @@ void Coefficient::Project(QuadratureFunction &qf)
QuadratureSpaceBase &qspace = *qf.GetSpace();
const int ne = qspace.GetNE();
Vector values;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
@@ -329,8 +327,6 @@ void VectorCoefficient::Project(QuadratureFunction &qf)
const int ne = qspace.GetNE();
DenseMatrix values;
Vector col;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
@@ -699,8 +695,6 @@ void MatrixCoefficient::Project(QuadratureFunction &qf, bool transpose)
QuadratureSpaceBase &qspace = *qf.GetSpace();
const int ne = qspace.GetNE();
DenseMatrix values, matrix;
// GetValues makes a reference, but we need it to be valid on Host
qf.HostWrite();
for (int iel = 0; iel < ne; ++iel)
{
qf.GetValues(iel, values);
+26 -88
View File
@@ -237,81 +237,6 @@ ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
gfi->SyncAliasMemory(*this);
}
real_t
ComplexGridFunction::ComputeLpError(const real_t p,
Coefficient &exsolr,
Coefficient &exsoli,
Coefficient *weight,
const IntegrationRule *irs[],
const Array<int> *elems) const
{
real_t error = 0.0;
const FiniteElement *fe;
ElementTransformation *T;
Vector valsr;
Vector valsi;
const GridFunction& gf_r = real();
const GridFunction& gf_i = imag();
for (int i = 0; i < fes->GetNE(); i++)
{
if (elems != NULL && (*elems)[i] == 0) { continue; }
fe = fes->GetFE(i);
const IntegrationRule *ir;
if (irs)
{
ir = irs[fe->GetGeomType()];
}
else
{
int intorder = 2*fe->GetOrder() + 3;
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
real_t elem_error = 0.0;
gf_r.GetValues(i, *ir, valsr);
gf_i.GetValues(i, *ir, valsi);
T = fes->GetElementTransformation(i);
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
T->SetIntPoint(&ip);
real_t diffr = valsr(j) - exsolr.Eval(*T, ip);
real_t diffi = valsi(j) - exsoli.Eval(*T, ip);
real_t diff = hypot(diffr, diffi);
if (p < infinity())
{
diff = pow(diff, p);
if (weight)
{
diff *= weight->Eval(*T, ip);
}
elem_error += ip.weight * T->Weight() * diff;
}
else
{
if (weight)
{
diff *= weight->Eval(*T, ip);
}
error = std::max(error, diff);
}
}
if (p < infinity())
{
// negative quadrature weights may cause the error to be negative
error += fabs(elem_error);
}
}
if (p < infinity())
{
error = pow(error, 1./p);
}
return error;
}
void ComplexGridFunction::Save(std::ostream &os) const
{
os << "ComplexGridFunction\n";
@@ -905,9 +830,15 @@ ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
int vsize = pfes->GetVSize();
Vector::Load(input, 2*vsize);
real_t *h_data = HostReadWrite();
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
@@ -1120,14 +1051,15 @@ void ParComplexGridFunction::Save(std::ostream &os) const
os << '\n';
int vsize = pfes->GetVSize();
// We use const_cast + HostRead (instead of HostReadWrite) because we only
// need to change the host data temporarily and this way we do not invalidate
// the data if it is on device. If we use HostReadWrite here, later calls to
// Read or ReadWrite will need to copy the data from host to device. With the
// approach used here, the host-to-device copy is avoided.
real_t *h_data = const_cast<real_t*>(HostRead());
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
if (pfes->GetOrdering() == Ordering::byNODES)
{
@@ -1138,8 +1070,14 @@ void ParComplexGridFunction::Save(std::ostream &os) const
Vector::Print(os, pfes->GetVDim());
}
pfes->ApplyDofSigns(h_data);
pfes->ApplyDofSigns(h_data + vsize);
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
os.flush();
}
-69
View File
@@ -166,75 +166,6 @@ public:
return sqrt(err_r * err_r + err_i * err_i);
}
/// @brief Returns Max|u_ex - u_h| error for complex-valued H1 or L2 elements
///
/// Compute the $L_\infty$ error across the entire domain.
///
/// @param[in] exsolr Coefficient object reproducing the real part of the
/// anticipated values of the scalar field, Re(u_ex).
/// @param[in] exsoli Coefficient object reproducing the imaginary part of
/// the anticipated values of the scalar field, Im(u_ex).
/// @param[in] irs Optional pointer to an array of custom integration
/// rules e.g. higher order than the default rules. If
/// present the array will be indexed by
/// Geometry::Type.
///
/// @note Uses ComputeLpError internally. See the ComputeLpError
/// documentation for generalizations of this error computation.
///
/// @note If an array of integration rules is provided through @a irs, be
/// sure to include valid rules for each element type that may occur
/// in the list of elements.
///
virtual real_t ComputeMaxError(Coefficient &exsolr,
Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
return ComputeLpError(infinity(), exsolr, exsoli, NULL, irs);
}
/// @brief Returns ||u_ex - u_h||_Lp for complex-valued H1 or L2 elements
///
/// Computes:
/// $$(\sum_{elems} \int_{elem} w \, |u_{ex} - u_h|^p)^{1/p}$$
/// Where:
/// $$|u_{ex} - u_h| = \sqrt{Re(u_{ex} - u_h)^2 + Im(u_{ex} - u_h)^2}$$
///
/// @param[in] p Real value indicating the exponent of the $L^p$ norm.
/// To avoid domain errors p should have a positive value,
/// either finite or infinite.
/// @param[in] exsolr Coefficient object reproducing the real part of the
/// anticipated values of the scalar field, Re(u_ex).
/// @param[in] exsoli Coefficient object reproducing the imaginary part of
/// the anticipated values of the scalar field, Im(u_ex).
/// @param[in] weight Optional pointer to a Coefficient object reproducing
/// a weighting function, w.
/// @param[in] irs Optional pointer to an array of custom integration
/// rules e.g. higher order than the default rules. If
/// present the array will be indexed by Geometry::Type.
/// @param[in] elems Optional pointer to a marker array, with a length
/// equal to the number of local elements, indicating
/// which elements to integrate over. Only those elements
/// corresponding to non-zero entries in @a elems will
/// contribute to the computed L2 error.
///
/// @note If an array of integration rules is provided through @a irs, be
/// sure to include valid rules for each element type that may occur
/// in the list of elements.
///
/// @note Quadratures with negative weights (as in some simplex integration
/// rules in MFEM) can produce negative integrals even with
/// non-negative integrands. To avoid returning negative errors this
/// function uses the absolute values of the element-wise integrals.
/// This may lead to results which are not entirely consistent with
/// such integration rules.
virtual real_t ComputeLpError(const real_t p,
Coefficient &exsolr,
Coefficient &exsoli,
Coefficient *weight = NULL,
const IntegrationRule *irs[] = NULL,
const Array<int> *elems = NULL) const;
/// Save the ComplexGridFunction to an output stream.
virtual void Save(std::ostream &out) const;
-4
View File
@@ -114,10 +114,6 @@ void ConduitDataCollection::Save()
n_mesh["fields"][name]);
}
// TODO: in parallel, we need to call ParFiniteElementSpace::ApplyDofSigns
// for all ParGridFunction objects before and after saving, see
// ParGridFunction::Save.
// save mesh data
SaveMeshAndFields(myid,
n_mesh,
+1 -3
View File
@@ -1181,14 +1181,12 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
DenseMatrix vval, pmat;
std::vector<char> buf;
int vec_dim = it->second->VectorDim();
int map_type = it->second->FESpace()->GetTypicalFE()->GetMapType();
os << "<DataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << it->first
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
<< VTKComponentLabels(vec_dim) << " "
<< "format=\"" << GetDataFormatString() << "\" >" << '\n';
if (vec_dim == 1 && (map_type == FiniteElement::VALUE ||
map_type == FiniteElement::INTEGRAL))
if (vec_dim == 1)
{
for (int i = 0; i < mesh->GetNE(); i++)
{
+30 -79
View File
@@ -25,35 +25,21 @@
namespace mfem
{
/// Lightweight adaptor over an std::map from type K to type to V
template<typename K, typename V,
typename = typename std::enable_if<std::is_default_constructible<V>::value>::type>
class GenericFieldMap
/// Lightweight adaptor over an std::map from strings to pointer to T
template<typename T>
class NamedFieldsMap
{
private:
static constexpr bool ValueIsPointer = std::is_pointer<V>::value;
public:
typedef std::map<K, V> MapType;
typedef std::map<std::string, T*> MapType;
typedef typename MapType::iterator iterator;
typedef typename MapType::const_iterator const_iterator;
/// Register field @a field with name @a key
/// Only enabled if the template parameter V is not a pointer
template<typename = std::enable_if<!ValueIsPointer, bool>>
void Register(const K& key, V field)
/// Register field @a field with name @a fname
/** Replace existing field associated with @a fname (and optionally
delete associated pointer if @a own_data is true) */
void Register(const std::string& fname, T* field, bool own_data)
{
field_map[key] = field;
}
/// Register field @a field with name @a key
/** Replace existing field associated with @a key (and optionally
delete associated pointer if @a own_data is true).
Only enabled if the template parameter V is a pointer*/
template<typename = std::enable_if<ValueIsPointer, bool>>
void Register(const K& key, V field, bool own_data)
{
V& ref = field_map[key];
T*& ref = field_map[fname];
if (own_data)
{
delete ref; // if newly allocated -> ref is null -> OK
@@ -61,40 +47,23 @@ public:
ref = field;
}
/// Unregister association between field @a field and name @a key
/// Only enabled if the template parameter V is not a pointer
template<typename = std::enable_if<!ValueIsPointer, bool>>
void Deregister(const K& key)
/// Unregister association between field @a field and name @a fname
/** Optionally delete associated pointer if @a own_data is true */
void Deregister(const std::string& fname, bool own_data)
{
iterator it = field_map.find(key);
if ( it != field_map.end() )
{
field_map.erase(it);
}
}
/// Unregister association between field @a field and name @a key
/** Optionally delete associated pointer if @a own_data is true.
Only enabled if the template parameter V is a pointer */
template<typename = std::enable_if<ValueIsPointer, bool>>
void Deregister(const K& key, bool own_data)
{
iterator it = field_map.find(key);
iterator it = field_map.find(fname);
if ( it != field_map.end() )
{
if (own_data)
{
delete it->second;
it->second = nullptr;
}
field_map.erase(it);
}
}
/// Clear all associations between names and fields
/** Delete associated pointers when @a own_data is true.
Only enabled if the template parameter V is a pointer */
template<typename = std::enable_if<ValueIsPointer, bool>>
/** Delete associated pointers when @a own_data is true */
void DeleteData(bool own_data)
{
for (iterator it = field_map.begin(); it != field_map.end(); ++it)
@@ -107,37 +76,22 @@ public:
}
}
/// Predicate to check if a field is associated with name @a key
bool Has(const K& key) const
/// Predicate to check if a field is associated with name @a fname
bool Has(const std::string& fname) const
{
return field_map.find(key) != field_map.end();
return field_map.find(fname) != field_map.end();
}
/// Get a pointer to the field associated with name @a key
/** @return Field associated with @a key or NULL,
if value is pointer and key not found */
V Get(const K& key) const
/// Get a pointer to the field associated with name @a fname
/** @return Pointer to field associated with @a fname or NULL */
T* Get(const std::string& fname) const
{
const_iterator it = field_map.find(key);
if (it != field_map.end())
{
return it->second;
}
else
{
if constexpr (ValueIsPointer)
{
return nullptr;
}
else
{
return V(); // Return default-constructed value for non-pointer types
}
}
const_iterator it = field_map.find(fname);
return it != field_map.end() ? it->second : NULL;
}
/// Returns a const reference to the underlying map
const MapType &GetMap() const { return field_map; }
const MapType& GetMap() const { return field_map; }
/// Returns the number of registered fields
int NumFields() const { return field_map.size(); }
@@ -152,24 +106,21 @@ public:
/// Returns an end const iterator to the registered fields
const_iterator end() const { return field_map.end(); }
/// Returns an iterator to the field @a key
iterator find(const K& key)
{ return field_map.find(key); }
/// Returns an iterator to the field @a fname
iterator find(const std::string& fname)
{ return field_map.find(fname); }
/// Returns a const iterator to the field @a key
const_iterator find(const K& key) const
{ return field_map.find(key); }
/// Returns a const iterator to the field @a fname
const_iterator find(const std::string& fname) const
{ return field_map.find(fname); }
/// Clears the map of registered fields
/// Clears the map of registered fields without reclaiming memory
void clear() { field_map.clear(); }
protected:
MapType field_map;
};
/// Lightweight adaptor over an std::map from strings to pointer to T
template<typename T>
using NamedFieldsMap = GenericFieldMap<std::string, T*>;
/** A class for collecting finite element data that is part of the same
simulation. Currently, this class groups together grid functions (fields),
+587
View File
@@ -0,0 +1,587 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include <cassert>
#include <cstddef>
// #include "fem/kernels.hpp"
#include "fem/kernels3d.hpp"
namespace ker = mfem::kernels::internal;
namespace low = mfem::kernels::internal::low;
#include "fem/kernel_dispatch.hpp"
// #include "linalg/kernels.hpp"
#include "util.hpp"
#undef NVTX_COLOR
#define NVTX_COLOR ::nvtx::kOrchid
namespace mfem::future
{
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1>` */
template<typename T, int n1>
MFEM_HOST_DEVICE
const tensor<T, n1>& as_tensor(const T* ptr)
{
// std::launder makes this defined behavior under strict aliasing rules
return *std::launder(reinterpret_cast<const tensor<T, n1>*>(ptr));
}
// convenience overload if you prefer a mutable view
template<typename T, int n1>
MFEM_HOST_DEVICE
tensor<T, n1>& as_tensor(T* ptr)
{
return *std::launder(reinterpret_cast<tensor<T, n1>*>(ptr));
}
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2>` */
template<typename T, int n1, int n2>
MFEM_HOST_DEVICE
const tensor<T, n1, n2>& as_tensor(const T* ptr)
{
// std::launder makes this defined behavior under strict aliasing rules
return *std::launder(reinterpret_cast<const tensor<T, n1, n2>*>(ptr));
}
// convenience overload if you prefer a mutable view
template<typename T, int n1, int n2>
MFEM_HOST_DEVICE
tensor<T, n1, n2>& as_tensor(T* ptr)
{
return *std::launder(reinterpret_cast<tensor<T, n1, n2>*>(ptr));
}
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3>` */
template<typename T, int n1, int n2, int n3>
MFEM_HOST_DEVICE
const tensor<T, n1, n2, n3>& as_tensor(const T* ptr)
{
// std::launder makes this defined behavior under strict aliasing rules
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3>*>(ptr));
}
// convenience overload if you prefer a mutable view
template<typename T, int n1, int n2, int n3>
MFEM_HOST_DEVICE
tensor<T, n1, n2, n3>& as_tensor(T* ptr)
{
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3>*>(ptr));
}
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3, n4>` */
template<typename T, int n1, int n2, int n3, int n4>
MFEM_HOST_DEVICE
const tensor<T, n1, n2, n3, n4>& as_tensor(const T* ptr)
{
// std::launder makes this defined behavior under strict aliasing rules
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3, n4>*>(ptr));
}
// convenience overload if you prefer a mutable view
template<typename T, int n1, int n2, int n3, int n4>
MFEM_HOST_DEVICE
tensor<T, n1, n2, n3, n4>& as_tensor(T* ptr)
{
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3, n4>*>(ptr));
}
template <std::size_t N>
MFEM_HOST_DEVICE inline
std::array<real_t*, N>
load_field_e_ptr(const std::array<DeviceTensor<2>, N> &fields_e,
const int e)
{
std::array<real_t*, N> f;
for_constexpr<N>([&](auto i) { f[i] = &fields_e[i](0, e); });
return f;
}
namespace qf
{
template <int T_Q1D,
size_t num_args,
typename reg_t,
typename qfunc_t,
typename args_ts>
MFEM_HOST_DEVICE inline
void apply_kernel(reg_t &res /*output*/,
reg_t &reg,
const real_t *rd,
const int qx, const int qy, const int qz,
const qfunc_t &qfunc, args_ts &args)
{
if constexpr (num_args == 2) // PAApply
{
// ∇u
tensor<real_t, 3> &arg_0 = get<0>(args);
arg_0[0] = reg[qz][qy][qx][0];
arg_0[1] = reg[qz][qy][qx][1];
arg_0[2] = reg[qz][qy][qx][2];
// D (PA data)
tensor<real_t, 3, 3> &arg_1 = get<1>(args);
if constexpr (T_Q1D > 0)
{
const auto *D = (const real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
for (int k = 0; k < 3; k++)
{
for (int j = 0; j < 3; j++)
{
arg_1[k][j] = D[qx][qy][qz][k][j];
}
}
}
else
{
static_assert(false);
// const auto D = Reshape(r2, 3, 3, Q1D, Q1D, Q1D);
// for (int j = 0; j < 3; j++)
// {
// for (int k = 0; k < 3; k++)
// {
// arg_1[k][j] = D(j, k, qz, qy, qx);
// }
// }
}
}
else
{
// MFApply comes here
assert(false);
// MFEM_ABORT("Only two arguments (∇u and D) are supported in apply_kernel for now");
}
const auto r = get<0>(apply(qfunc, args));
if constexpr (decltype(r)::ndim == 1)
{
// process_qf_result_from_reg(r0, qx, qy, qz, r);
as_tensor<real_t, 3>(&res[qz][qy][qx][0]) = r;
}
else
{
static_assert(false);
}
}
} // namespace qf
#define MFEM_D2Q_MAX_SIZE 4
static MFEM_CONSTANT real_t Bi[MFEM_D2Q_MAX_SIZE][8*8], Bo[8*8];
static MFEM_CONSTANT real_t Gi[MFEM_D2Q_MAX_SIZE][8*8], Go[8*8];
template<size_t num_fields,
size_t num_inputs,
size_t num_outputs,
typename restriction_cb_t,
typename qfunc_t,
typename input_t,
typename output_fop_t>
class NewActionCallback
{
restriction_cb_t &restriction_cb;
qfunc_t &qfunc;
input_t &inputs;
const std::array<size_t, num_inputs> &input_to_field;
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps;
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps;
const int num_entities;
const int test_vdim;
const int num_test_dof;
const int dimension;
const ThreadBlocks &thread_blocks;
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info;
const Array<int> &attributes;
const output_fop_t &output_fop;
const Array<int> *elem_attributes;
// refs
std::vector<Vector> &fields_e;
Vector &residual_e;
std::function<void(Vector &, Vector &)> &output_restriction_transpose;
// args
std::vector<Vector> &solutions_l;
const std::vector<Vector> &parameters_l;
Vector &residual_l;
public:
NewActionCallback() = delete;
NewActionCallback(const bool use_kernels_specialization,
restriction_cb_t &restriction_cb,
qfunc_t &qfunc,
input_t &inputs,
const std::array<size_t, num_inputs> &input_to_field,
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
const int num_entities,
const int test_vdim,
const int num_test_dof,
const int dimension,
const ThreadBlocks &thread_blocks,
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info,
const Array<int> &attributes,
const output_fop_t &output_fop,
const Array<int> *elem_attributes,
// refs
std::vector<Vector> &fields_e,
Vector &residual_e,
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
// args
std::vector<Vector> &solutions_l,
const std::vector<Vector> &parameters_l,
Vector &residual_l):
restriction_cb(restriction_cb),
qfunc(qfunc),
inputs(inputs),
input_to_field(input_to_field),
input_dtq_maps(input_dtq_maps),
output_dtq_maps(output_dtq_maps),
num_entities(num_entities),
test_vdim(test_vdim),
num_test_dof(num_test_dof),
dimension(dimension),
thread_blocks(thread_blocks),
shmem_info(shmem_info),
attributes(attributes),
output_fop(output_fop),
elem_attributes(elem_attributes),
fields_e(fields_e),
residual_e(residual_e),
output_restriction_transpose(output_restriction_transpose),
solutions_l(solutions_l),
parameters_l(parameters_l),
residual_l(residual_l)
{
if (!use_kernels_specialization) { return; }
NewActionCallbackKernels::template Specialization<3>::Add(); // 1
NewActionCallbackKernels::template Specialization<4>::Add(); // 2
NewActionCallbackKernels::template Specialization<5>::Add(); // 3
NewActionCallbackKernels::template Specialization<6>::Add(); // 4
NewActionCallbackKernels::template Specialization<7>::Add(); // 5
NewActionCallbackKernels::template Specialization<8>::Add(); // 6
}
template<int T_Q1D = 0>
static void action_callback_new(const int d1d,
restriction_cb_t &restriction_cb,
qfunc_t &qfunc,
[[maybe_unused]] input_t &inputs,
[[maybe_unused]] const std::array<size_t, num_inputs> &input_to_field,
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
[[maybe_unused]] const int dimension,
const int num_entities,
[[maybe_unused]] const int test_vdim,
[[maybe_unused]] const int num_test_dof,
const ThreadBlocks &thread_blocks,
[[maybe_unused]] SharedMemoryInfo<num_fields, num_inputs, num_outputs>
&shmem_info,
[[maybe_unused]] const Array<int> &attributes,
[[maybe_unused]] const output_fop_t &output_fop,
[[maybe_unused]] const Array<int> *elem_attributes,
// refs
std::vector<Vector> &fields_e,
Vector &residual_e,
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
// args
std::vector<Vector> &solutions_l,
const std::vector<Vector> &parameters_l,
Vector &residual_l,
// fallback arguments
const int q1d)
{
NVTX_MARK_FUNCTION;
assert(dimension == 3);
static_assert(MFEM_D2Q_MAX_SIZE >= num_inputs, "MFEM_D2Q_MAX_SIZE error");
constexpr int DIM = 3;
[[maybe_unused]] static bool ini = (for_constexpr<num_inputs>([&](auto i)
{
const auto dtq = input_dtq_maps[i];
{
const auto [q, _, p] = dtq.B.GetShape();
const auto B = (const real_t*)input_dtq_maps[i].B;
dbg("Loading Bi[{}]: q={} p={}", i.value, q, p);
if (B) { Gpu(MemcpyToSymbol)(Bi[i], B, (p*q)*sizeof(real_t)); }
}
{
const auto [q, _, p] = dtq.G.GetShape();
const auto G = (const real_t*)input_dtq_maps[i].G;
if (G) { Gpu(MemcpyToSymbol)(Gi[i], G, (p*q)*sizeof(real_t)); }
}
if constexpr (i == 0) // output B
{
const auto dtq_o = output_dtq_maps[0];
const auto [q, _, p] = dtq_o.B.GetShape();
const auto B = (const real_t*)dtq_o.B;
if (B) { Gpu(MemcpyToSymbol)(Bo, B, (p*q)*sizeof(real_t)); }
}
if constexpr (i == 0) // output G
{
const auto dtq_o = output_dtq_maps[0];
const auto [q, _, p] = dtq_o.G.GetShape();
const auto G = (const real_t*)dtq_o.G;
if (G) { Gpu(MemcpyToSymbol)(Go, G, (p*q)*sizeof(real_t)); }
dbg("Loaded B and G to constant memory");
}
}), true);
// types
using qf_signature =
typename create_function_signature<decltype(&qfunc_t::operator())>::type;
using qf_param_ts = typename qf_signature::parameter_ts;
restriction_cb(solutions_l, parameters_l, fields_e);
NVTX_INI("res=0");
residual_e = 0.0;
NVTX_END("res=0");
// auto wrapped_fields_e =
// wrap_fields(fields_e, shmem_info.field_sizes, num_entities);
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
// const int vdim = input.vdim;
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
const int NE = num_entities;
constexpr int VDIM = 1;
const auto XE = Reshape(fields_e[0].Read(), d1d, d1d, d1d, VDIM, NE);
const real_t *dx_ptr = fields_e[1].Read();
auto YE = Reshape(residual_e.ReadWrite(), d1d, d1d, d1d, VDIM, NE);
const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
NVTX_INI("forall");
dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
{
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
// real_t (&sm0_ptr)[MQ1][MQ1][MQ1][3] = sm0;
// real_t (&sm1_ptr)[MQ1][MQ1][MQ1][3] = sm1;
low::regs3d_t<DIM, MQ1> reg;
const real_t *rd = dx_ptr;
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
// real_t (&sB_ptr)[MD1][MQ1] = sB;
// real_t (&sG_ptr)[MD1][MQ1] = sG;
// Interpolate
// for_constexpr<num_inputs>(
// [ D1D, Q1D, MQ1, e,
// &input_dtq_maps,
// &sm0_ptr, &sm1_ptr,
// &sB = sB_ptr, &sG = sG_ptr,
// &inputs,
// // &fields_e_ptr,
// &reg, &rd,
// &input_to_field ] (auto i)
{
// const auto input = get<0/*i*/>(inputs);
// using field_operator_t = std::decay_t<decltype(input)>;
// if constexpr (is_gradient_fop<field_operator_t>::value) // Grad
{
// const int vdim = input.vdim;
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
// const auto XE = Reshape(field_e_r, D1D, D1D, D1D, vdim);
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bi[i]);
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Gi[i]);
low::LoadMatrix(d1d, q1d, B, sB);
low::LoadMatrix(d1d, q1d, G, sG);
// for (int c = 0; c < vdim; c++)
// constexpr int c = 0;
{
low::LoadDofs3d(e, d1d, XE, sm0);
low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
}
}
// else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
{
// db1("Identity");
// rd = fields_e_ptr[input_to_field[i]];
// rd = dx_ptr;
}
// else if constexpr (is_weight_fop<field_operator_t>::value) // Weight
// {
// dbg("Weight");
// rw = fields_e_ptr[input_to_field[i]]; // 🔥
// }
// else
{
// MFApply comes here
// assert(false);
// MFEM_ABORT("Only Grad and Identity field operators are supported");
}
}//); // for_constexpr<num_inputs>
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
{
#if 0
auto qf_args = decay_tuple<qf_param_ts> {};
qf::apply_kernel<T_Q1D, num_inputs>
(reg, reg, rd, qx, qy, qz, qfunc, qf_args);
#elif 0
real_t v[3], u[3] = { reg[qz][qy][qx][0],
reg[qz][qy][qx][1],
reg[qz][qy][qx][2]
};
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
kernels::Mult(3, 3, &D[qx][qy][qz][0][0], u, v);
reg[qz][qy][qx][0] = v[0];
reg[qz][qy][qx][1] = v[1];
reg[qz][qy][qx][2] = v[2];
#elif 0
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
const auto args = decay_tuple<qf_param_ts>
{
{{ reg[qz][qy][qx][0], reg[qz][qy][qx][1], reg[qz][qy][qx][2] }},
{{
{{ D[qx][qy][qz][0][0], D[qx][qy][qz][0][1], D[qx][qy][qz][0][2] }},
{{ D[qx][qy][qz][1][0], D[qx][qy][qz][1][1], D[qx][qy][qz][1][2] }},
{{ D[qx][qy][qz][2][0], D[qx][qy][qz][2][1], D[qx][qy][qz][2][2] }}
}
}
};
const auto r = get<0>(apply(qfunc, args));
reg[qz][qy][qx][0] = r[0];
reg[qz][qy][qx][1] = r[1];
reg[qz][qy][qx][2] = r[2];
#elif 0
auto u = as_tensor<real_t, 3>(&reg[qz][qy][qx][0]);
const auto *d = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
auto D = as_tensor<real_t, 3, 3>(&d[qx][qy][qz][0][0]);
auto r = D * u;
reg[qz][qy][qx][0] = r[0];
reg[qz][qy][qx][1] = r[1];
reg[qz][qy][qx][2] = r[2];
#else
auto args = decay_tuple<qf_param_ts> {};
get<0>(args) = as_tensor<real_t, 3>(&reg[qz][qy][qx][0]);
if constexpr (T_Q1D > 0)
{
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
}
else
{
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
}
auto r = get<0>(apply(qfunc, args));
if constexpr (decltype(r)::ndim == 1)
{
as_tensor<real_t, 3>(&reg[qz][qy][qx][0]) = r;
}
else { static_assert(false); }
#endif
}
}
}
MFEM_SYNC_THREAD;
// Integrate
// if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
{
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
low::WriteDofs3d(d1d, 0, e, reg, YE);
}
},
num_entities, thread_blocks, 0, nullptr);
NVTX_END("forall");
NVTX_INI("out^T");
output_restriction_transpose(residual_e, residual_l);
NVTX_END("out^T");
}
using NewActionKernelType = decltype(&NewActionCallback::action_callback_new<>);
MFEM_REGISTER_KERNELS(NewActionCallbackKernels, NewActionKernelType, (int));
void Apply(const int d1d, const int q1d)
{
db1();
NewActionCallbackKernels::Run(q1d,
// args
d1d,
restriction_cb,
qfunc,
inputs,
input_to_field,
input_dtq_maps,
output_dtq_maps,
dimension,
num_entities,
test_vdim,
num_test_dof,
thread_blocks,
shmem_info,
attributes,
output_fop,
elem_attributes,
fields_e,
residual_e,
output_restriction_transpose,
solutions_l,
parameters_l,
residual_l,
// fallback arguments
q1d);
}
};
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
template<int T_Q1D>
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Kernel()
{
return action_callback_new<T_Q1D>;
}
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Fallback
(int q1d)
{
dbg("\x1b[33mFallback q1d:{}", q1d);
// MFEM_ABORT("No kernel for q1d=" << q1d);
// return nullptr;
return action_callback_new<>;
}
} // namespace mfem::future
+111
View File
@@ -0,0 +1,111 @@
#pragma once
#include "../util.hpp"
#include "../../integrator_ctx.hpp"
#include <utility>
namespace mfem::future
{
namespace GlobalQFImpl
{
template<
typename qfunc_t,
typename inputs_t,
typename outputs_t,
size_t ninputs = tuple_size<inputs_t>::value,
size_t noutputs = tuple_size<outputs_t>::value>
struct Action
{
Action(
IntegratorContext ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs) :
ctx(ctx),
qfunc(std::move(qfunc)),
inputs(inputs),
outputs(outputs)
{
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
check_consistency(inputs, input_to_infd, ctx.infds);
check_consistency(outputs, output_to_outfd, ctx.outfds);
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
const int nqp = ctx.ir.GetNPoints();
gnqp = nqp * ctx.nentities;
xq_offsets.SetSize(ninputs + 1);
xq_offsets[0] = 0;
constexpr_for<0, ninputs>([&](auto i)
{
const auto input = get<i>(inputs);
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
});
xq_offsets.PartialSum();
xq.Update(xq_offsets);
yq_offsets.SetSize(noutputs + 1);
yq_offsets[0] = 0;
constexpr_for<0, noutputs>([&](auto i)
{
const auto output = get<i>(outputs);
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
});
yq_offsets.PartialSum();
yq.Update(yq_offsets);
}
void operator()(
const std::vector<Vector *> &xe,
std::vector<Vector *> &ye) const
{
if (ctx.attr.Size() == 0) { return; }
// E -> Q
interpolate(input_to_infd, input_bases, xe, xq);
// Q -> Q
static_assert(
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
"qfunc signature not supported by default backend Action");
detail::call_qfunc(
qfunc, xq, yq, gnqp, input_qlayouts, output_qlayouts,
std::make_index_sequence<ninputs> {},
std::make_index_sequence<noutputs> {});
// Q -> E
integrate(output_to_outfd, output_bases, yq, ye);
}
IntegratorContext ctx;
qfunc_t qfunc;
inputs_t inputs;
outputs_t outputs;
std::array<size_t, ninputs> input_to_infd;
std::array<size_t, noutputs> output_to_outfd;
std::array<FieldBasis, ninputs> input_bases;
std::array<FieldBasis, noutputs> output_bases;
std::array<std::vector<int>, ninputs> input_qlayouts;
std::array<std::vector<int>, noutputs> output_qlayouts;
int gnqp = 0;
Array<int> xq_offsets, yq_offsets;
mutable BlockVector xq, yq;
};
}
}
@@ -0,0 +1,131 @@
#pragma once
#include "../fem/quadinterpolator.hpp"
#include "../../integrator_ctx.hpp"
#include "../util.hpp"
#include <utility>
namespace mfem::future
{
namespace GlobalQFImpl
{
template<
int derivative_id,
typename qfunc_t,
typename inputs_t,
typename outputs_t,
size_t ninputs = tuple_size<inputs_t>::value,
size_t noutputs = tuple_size<outputs_t>::value>
struct DerivativeActionEnzyme
{
DerivativeActionEnzyme(
IntegratorContext ctx,
qfunc_t &qfunc,
inputs_t inputs,
outputs_t outputs) :
ctx(ctx),
qfunc(qfunc),
inputs(inputs),
outputs(outputs)
{
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
check_consistency(inputs, input_to_infd, ctx.infds);
check_consistency(outputs, output_to_outfd, ctx.outfds);
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
const int nqp = ctx.ir.GetNPoints();
gnqp = nqp * ctx.nentities;
xq_offsets.SetSize(ninputs + 1);
xq_offsets[0] = 0;
constexpr_for<0, ninputs>([&](auto i)
{
const auto input = get<i>(inputs);
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
});
xq_offsets.PartialSum();
xq.Update(xq_offsets);
yq_offsets.SetSize(noutputs + 1);
yq_offsets[0] = 0;
constexpr_for<0, noutputs>([&](auto i)
{
const auto output = get<i>(outputs);
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
});
yq_offsets.PartialSum();
yq.Update(yq_offsets);
// For each dependent input in the dependency map we create a shadow
// memory variable at the quadrature point level.
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
shadow_xq_offsets.SetSize(ninputs + 1);
shadow_xq_offsets = 0;
constexpr_for<0, ninputs>([&](auto i)
{
if (activity_map[i])
{
shadow_xq_offsets[i + 1] =
xq_offsets[i + 1] - xq_offsets[i];;
}
});
shadow_xq_offsets.PartialSum();
shadow_xq.Update(shadow_xq_offsets);
}
void operator()(
const std::vector<Vector *> &xe,
const Vector *de,
std::vector<Vector *> &ye) const
{
if (ctx.attr.Size() == 0) { return; }
// E -> Q
interpolate(input_to_infd, input_bases, xe, xq);
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
interpolate(input_to_infd, input_bases, xe, shadow_xq, activity_map);
// Q -> Q
static_assert(
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
"qfunc signature not supported by default backend Action");
detail::enzyme_fwddiff<derivative_id, qfunc_t, inputs_t, outputs_t>(
qfunc, xq, shadow_xq, yq, gnqp, input_qlayouts, output_qlayouts,
std::make_index_sequence<ninputs> {},
std::make_index_sequence<noutputs> {});
// Q -> E
integrate(output_to_outfd, output_bases, yq, ye);
}
IntegratorContext ctx;
qfunc_t &qfunc;
inputs_t inputs;
outputs_t outputs;
std::array<size_t, ninputs> input_to_infd;
std::array<size_t, noutputs> output_to_outfd;
std::array<FieldBasis, ninputs> input_bases;
std::array<FieldBasis, noutputs> output_bases;
std::array<std::vector<int>, ninputs> input_qlayouts;
std::array<std::vector<int>, noutputs> output_qlayouts;
int gnqp = 0;
Array<int> xq_offsets, shadow_xq_offsets, yq_offsets;
mutable BlockVector xq, shadow_xq, yq;
};
}
}
+42
View File
@@ -0,0 +1,42 @@
#pragma once
#include "action.hpp"
#include "derivative_action_enzyme.hpp"
namespace mfem::future
{
struct GlobalQFBackend
{
template<
typename qfunc_t,
typename inputs_t,
typename outputs_t>
auto static MakeAction(
const IntegratorContext &ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs)
{
return GlobalQFImpl::Action(ctx, qfunc, inputs, outputs);
}
template<
int derivative_id,
typename qfunc_t,
typename inputs_t,
typename outputs_t>
auto static MakeDerivativeAction(
const IntegratorContext &ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs)
{
return GlobalQFImpl::DerivativeActionEnzyme<
derivative_id, qfunc_t, inputs_t, outputs_t>(
ctx, qfunc, inputs, outputs);
}
};
}
+166
View File
@@ -0,0 +1,166 @@
#pragma once
#include "../util.hpp"
#include "../../integrator_ctx.hpp"
#include <utility>
namespace mfem::future
{
namespace LocalQFImpl
{
template<
typename qfunc_t,
typename inputs_t,
typename outputs_t,
size_t ninputs = tuple_size<inputs_t>::value,
size_t noutputs = tuple_size<outputs_t>::value>
struct Action
{
Action(
IntegratorContext ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs) :
ctx(ctx),
qfunc(std::move(qfunc)),
inputs(inputs),
outputs(outputs)
{
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
check_consistency(inputs, input_to_infd, ctx.infds);
check_consistency(outputs, output_to_outfd, ctx.outfds);
const int nqp = ctx.ir.GetNPoints();
// Initialize DofToQuad maps for inputs
for_constexpr<ninputs>([&](auto i)
{
const auto &fd = ctx.infds[input_to_infd[i]];
std::visit([&](auto* space_ptr)
{
using T = std::decay_t<decltype(*space_ptr)>;
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
std::is_same_v<T, ParFiniteElementSpace>)
{
const auto *fe = space_ptr->GetTypicalFE();
input_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
}
}, fd.data);
});
// Initialize DofToQuad maps for outputs
for_constexpr<noutputs>([&](auto i)
{
const auto &fd = ctx.outfds[output_to_outfd[i]];
std::visit([&](auto* space_ptr)
{
using T = std::decay_t<decltype(*space_ptr)>;
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
std::is_same_v<T, ParFiniteElementSpace>)
{
const auto *fe = space_ptr->GetTypicalFE();
output_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
}
}, fd.data);
});
}
void operator()(
const std::vector<Vector *> &xe,
std::vector<Vector *> &ye) const
{
if (ctx.attr.Size() == 0) { return; }
// input_dtq_maps
// const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
// const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
// dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
// {
// if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
// MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
// MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
// low::regs3d_t<DIM, MQ1> reg;
// const real_t *rd = dx_ptr;
// MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
// {
// low::LoadMatrix(d1d, q1d, B, sB);
// low::LoadMatrix(d1d, q1d, G, sG);
// {
// low::LoadDofs3d(e, d1d, XE, sm0);
// low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
// }
// }
// // else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
// {
// // db1("Identity");
// // rd = fields_e_ptr[input_to_field[i]];
// // rd = dx_ptr;
// }
// }
// MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
// {
// MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
// {
// MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
// {
// auto args = decay_tuple<qf_param_ts> {};
// get<0>(args) = as_tensor<real_t, 3>(&reg[qz][qy][qx][0]);
// if constexpr (T_Q1D > 0)
// {
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
// }
// else
// {
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
// }
// auto r = get<0>(apply(qfunc, args));
// if constexpr (decltype(r)::ndim == 1)
// {
// as_tensor<real_t, 3>(&reg[qz][qy][qx][0]) = r;
// }
// else { static_assert(false); }
// }
// }
// }
// MFEM_SYNC_THREAD;
// // Integrate
// // if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
// {
// // const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
// // const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
// low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
// low::WriteDofs3d(d1d, 0, e, reg, YE);
// }
// },
// num_entities, thread_blocks, 0, nullptr);
}
IntegratorContext ctx;
qfunc_t qfunc;
inputs_t inputs;
outputs_t outputs;
std::array<size_t, ninputs> input_to_infd;
std::array<size_t, noutputs> output_to_outfd;
std::array<const DofToQuad*, ninputs> input_dtq_maps;
std::array<const DofToQuad*, noutputs> output_dtq_maps;
};
}
}
+39
View File
@@ -0,0 +1,39 @@
#pragma once
#include "../../integrator_ctx.hpp"
#include "action.hpp"
namespace mfem::future
{
struct LocalQFBackend
{
template<
typename qfunc_t,
typename inputs_t,
typename outputs_t>
auto static MakeAction(
const IntegratorContext &ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs)
{
return LocalQFImpl::Action(ctx, qfunc, inputs, outputs);
}
template<
int derivative_id,
typename qfunc_t,
typename inputs_t,
typename outputs_t>
auto static MakeDerivativeAction(
const IntegratorContext &ctx,
qfunc_t qfunc,
inputs_t inputs,
outputs_t outputs)
{
MFEM_ABORT("LocalQFBackend does not support derivative actions.");
}
};
}
+659
View File
@@ -0,0 +1,659 @@
#pragma once
#include "../fem/quadinterpolator.hpp"
#include "../util.hpp"
#include "general/enzyme.hpp"
namespace mfem::future
{
template <size_t N, size_t... Is>
constexpr std::array<bool, N> all_true_impl(std::index_sequence<Is...>)
{
return {{((void)Is, true)...}};
}
template <size_t N>
constexpr std::array<bool, N> all_true()
{
return all_true_impl<N>(std::make_index_sequence<N> {});
}
struct FieldBasis
{
// E-vector -> Q-vector
std::function<void(const Vector &, Vector &)> forward;
// Q-vector -> E-vector
std::function<void(const Vector &, Vector &)> transpose;
};
inline FieldBasis FromQI(const QuadratureInterpolator *qi,
QuadratureInterpolator::EvalFlags mode)
{
return
{
[qi, mode](const Vector &xe, Vector &xq)
{
qi->SetOutputLayout(QVectorLayout::byVDIM);
if (mode == QuadratureInterpolator::VALUES)
{
qi->Values(xe, xq);
}
else
{
qi->Derivatives(xe, xq);
}
},
[qi, mode](const Vector &yq, Vector &ye)
{
Vector empty;
qi->SetOutputLayout(QVectorLayout::byVDIM);
if (mode == QuadratureInterpolator::VALUES)
{
qi->AddMultTranspose(QuadratureInterpolator::VALUES, yq, empty, ye);
}
else
{
qi->AddMultTranspose(QuadratureInterpolator::DERIVATIVES, empty, yq, ye);
}
}
};
}
// QuadratureFunction identity copy
inline FieldBasis FromQF()
{
return
{
[](const Vector &xe, Vector &xq) { xq = xe; },
[](const Vector &yq, Vector &ye) { ye = yq; }
};
}
// User-defined parameter space B
inline FieldBasis FromPS(const Operator *B, const Operator *Bt)
{
return
{
[B](const Vector &xe, Vector &xq) { B->Mult(xe, xq); },
[Bt](const Vector &yq, Vector &ye) { Bt->Mult(yq, ye); }
};
}
inline FieldBasis FieldBasisFromWeight(const IntegrationRule &ir)
{
return
{
[&ir](const Vector &, Vector &xq)
{
const int nqp = ir.GetNPoints();
MFEM_ASSERT(xq.Size() % nqp == 0, "weight block has unexpected size");
const int ne = xq.Size() / nqp;
const real_t *wref = ir.GetWeights().Read();
for (int e = 0; e < ne; e++)
{
std::memcpy(xq.GetData() + e*nqp, wref, nqp*sizeof(real_t));
}
},
[](const Vector &, Vector &) {}
};
}
inline const FieldBasis GetFieldBasis(const FieldDescriptor &f,
const IntegrationRule &ir,
QuadratureInterpolator::EvalFlags mode)
{
return std::visit([&ir, &mode](auto && arg) -> FieldBasis
{
using T = std::decay_t<decltype(arg)>;
if constexpr (std::is_same_v<T, const FiniteElementSpace *>)
{
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
}
else if constexpr (std::is_same_v<T, const ParFiniteElementSpace *>)
{
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return FromQF();
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return FromPS(arg->GetB(), arg->GetBt());
}
else if constexpr (std::is_same_v<T, const IntegrationRule *>)
{
return FieldBasis{};
}
else
{
static_assert(dfem::always_false<T>, "internal error");
}
}, f.data);
}
template <typename fops_t, size_t nfops>
void create_fieldbases(
fops_t &fops,
const std::array<size_t, nfops> &fop_to_fd,
const std::vector<FieldDescriptor> &fds,
const IntegrationRule &ir,
std::array<FieldBasis, nfops> &bases)
{
constexpr_for<0, nfops>([&](auto i)
{
const auto fop = get<i>(fops);
using fop_t = std::decay_t<decltype(fop)>;
const auto fd = fds[fop_to_fd[i]];
constexpr QuadratureInterpolator::EvalFlags dummy_mode =
QuadratureInterpolator::VALUES;
if constexpr (is_identity_fop<fop_t>::value)
{
bases[i] = GetFieldBasis(fd, ir, dummy_mode);
}
else if constexpr (is_weight_fop<fop_t>::value)
{
bases[i] = FieldBasisFromWeight(ir);
}
else if constexpr (is_value_fop<fop_t>::value)
{
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::VALUES);
}
else if constexpr (is_gradient_fop<fop_t>::value)
{
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::DERIVATIVES);
}
});
}
template <typename fops_t, size_t nfops>
void check_consistency(
fops_t &fops,
const std::array<size_t, nfops> &fop_to_fd,
const std::vector<FieldDescriptor> &fields)
{
constexpr_for<0, nfops>([&](auto i)
{
const auto input = get<i>(fops);
using input_t = std::decay_t<decltype(input)>;
const auto fd = fields[fop_to_fd[i]];
if constexpr (is_identity_fop<input_t>::value)
{
MFEM_ASSERT(std::holds_alternative<const QuadratureFunction *>(fd.data),
"Identity FieldOperator requested on non "
"QuadratureFunction");
}
else if constexpr (is_weight_fop<input_t>::value)
{
}
else if constexpr (is_value_fop<input_t>::value)
{
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data) ||
std::holds_alternative<const ParameterSpace *>(fd.data),
"Value FieldOperator requested on non "
"QuadratureFunction");
}
else if constexpr (is_gradient_fop<input_t>::value)
{
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data),
"Value FieldOperator requested on non "
"QuadratureFunction");
}
});
}
template <size_t ninputs>
void interpolate(
const std::array<size_t, ninputs> &input_to_infd,
const std::array<FieldBasis, ninputs> &input_bases,
const std::vector<Vector *> &xe,
BlockVector &xq,
const std::array<bool, ninputs> &conditional = all_true<ninputs>())
{
constexpr_for<0, ninputs>([&](auto i)
{
if (!conditional.empty() && !conditional[i]) { return; }
input_bases[i].forward(*xe[input_to_infd[i]], xq.GetBlock(i));
});
}
template <size_t noutputs>
void integrate(
const std::array<size_t, noutputs> &output_to_outfd,
const std::array<FieldBasis, noutputs> &output_bases,
const BlockVector &yq,
std::vector<Vector *> &ye)
{
for (auto v : ye) { *v = 0.0; }
constexpr_for<0, noutputs>([&](auto i)
{
output_bases[i].transpose(yq.GetBlock(i), *ye[output_to_outfd[i]]);
});
}
namespace detail
{
template <typename T>
struct is_tensor_array : std::false_type {};
template <typename scalar_t, int... Dims>
struct is_tensor_array<tensor_array<scalar_t, Dims...>> : std::true_type {};
template <typename T>
struct is_tensor_array_mut : std::false_type {};
template <typename scalar_t, int... Dims>
struct is_tensor_array_mut<tensor_array<scalar_t, Dims...>> :
std::bool_constant<!std::is_const_v<scalar_t>> {};
template <typename ndarray_t>
inline void set_layout_default(ndarray_t &a)
{
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
constexpr std::size_t nd = ndarray_t::rank();
constexpr std::size_t td = ndarray_t::tensor_rank();
std::array<std::size_t, nd + td> perm{};
for (std::size_t i = 0; i < td; i++) { perm[i] = nd + i; }
for (std::size_t i = 0; i < nd; i++) { perm[td + i] = i; }
a.set_layout(perm);
}
template <typename ndarray_t>
inline void set_layout(ndarray_t& a, const std::vector<int>& layout)
{
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
constexpr std::size_t nd = ndarray_t::rank();
constexpr std::size_t td = ndarray_t::tensor_rank();
constexpr std::size_t N = nd + td;
// missing means default
if (layout.empty()) { set_layout_default(a); return; }
MFEM_VERIFY(layout.size() == N,
"layout size mismatch: expected " << N << " got " << layout.size());
// TODO: make a version of set_layout that takes `std::vector<int>`
std::array<std::size_t, N> perm{};
for (std::size_t i = 0; i < N; i++)
{
MFEM_VERIFY(layout[i] >= 0, "layout index must be >=0");
perm[i] = static_cast<std::size_t>(layout[i]);
}
a.set_layout(perm);
}
/// Primary template: intentionally undefined — gives a clear error for unsupported types.
template <typename T>
struct tensor_array_traits;
/// Matches tensor<scalar_t, sizes...>
template <typename scalar_t, int... sizes>
struct tensor_array_traits<tensor<scalar_t, sizes...>>
{
using scalar_type = scalar_t;
template <std::size_t ndims>
using array_type = tensor_ndarray<scalar_t, ndims, sizes...>;
};
/// Matches tensor_ndarray<scalar_t, ndims, tensor_sizes...>
template <typename scalar_t, int ndims, int... tensor_sizes>
struct tensor_array_traits<tensor_ndarray<scalar_t, ndims, tensor_sizes...>>
{
using scalar_type = scalar_t;
template <std::size_t N>
using array_type = tensor_ndarray<scalar_t, N, tensor_sizes...>;
};
/// Entry point: explicit tensor type T as template argument.
template <typename T, typename ptr_scalar_t, typename... dyn_sizes_t>
decltype(auto) make_tensor_array(ptr_scalar_t *ptr,
const std::vector<int>* layout,
dyn_sizes_t... dynamic_sizes)
{
using traits = tensor_array_traits<T>;
using array_t = typename traits::template array_type<sizeof...(dynamic_sizes)>;
auto a = array_t(ptr, {std::size_t(dynamic_sizes)...});
if (layout) { set_layout(a, *layout); }
else { set_layout_default(a); }
return a;
}
template <typename qfunc_t, typename inputs_t, typename outputs_t>
struct supports_tensor_array_qfunc
{
using qf_signature = typename get_function_signature<qfunc_t>::type;
using qf_param_ts = typename qf_signature::parameter_ts;
static constexpr int ninputs = tuple_size<inputs_t>::value;
static constexpr int noutputs = tuple_size<outputs_t>::value;
static constexpr int nparams = tuple_size<qf_param_ts>::value;
template <std::size_t... Is>
static constexpr bool InputsOk(std::index_sequence<Is...>)
{
return (is_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<Is, qf_param_ts>::type>>>::value && ...);
}
template <std::size_t... Is>
static constexpr bool OutputsOk(std::index_sequence<Is...>)
{
return (is_tensor_array_mut<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<ninputs + Is, qf_param_ts>::type>>>::value && ...);
}
static constexpr bool value =
(nparams == ninputs + noutputs) &&
InputsOk(std::make_index_sequence<ninputs> {}) &&
OutputsOk(std::make_index_sequence<noutputs> {});
};
template <typename qfunc_t, std::size_t... Is, std::size_t... Os>
inline void call_qfunc(
const qfunc_t &qfunc,
const BlockVector &xq,
BlockVector &yq,
int gnqp,
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
std::index_sequence<Is...>,
std::index_sequence<Os...>)
{
constexpr std::size_t ninputs = sizeof...(Is);
using qf_signature = typename get_function_signature<qfunc_t>::type;
using qf_param_ts = typename qf_signature::parameter_ts;
auto inputs = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<Is, qf_param_ts>::type>>>(
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
auto outputs = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
std::apply([&](auto&&... args)
{
qfunc(args...);
}, std::tuple_cat(inputs, outputs));
}
template <typename func_t, typename... arg_ts>
MFEM_HOST_DEVICE inline
auto qfunction_wrapper(const func_t &f, arg_ts...args)
{
return f(args...);
}
template <std::size_t derivative_id, std::size_t I, typename Tuple, std::size_t... Is>
constexpr std::array<bool, sizeof...(Is)>
make_activity_array(std::index_sequence<Is...>)
{
return { (std::decay_t<typename tuple_element<Is, Tuple>::type>::GetFieldId() == derivative_id)... };
}
template <std::size_t derivative_id, typename inputs_t, std::size_t... Is>
constexpr auto make_activity_map_impl(std::index_sequence<Is...>)
{
constexpr std::size_t N = sizeof...(Is);
if constexpr (N == 0)
return std::array<bool, 0> {};
return make_activity_array<derivative_id, 0, inputs_t>
(std::make_index_sequence<N> {});
}
template <std::size_t derivative_id, typename inputs_t>
constexpr auto make_activity_map(inputs_t)
{
return make_activity_map_impl<derivative_id, inputs_t>(
std::make_index_sequence<tuple_size<inputs_t>::value> {});
}
namespace enzyme_detail
{
template <auto wrapper_fn, typename qf_return_t, typename... AccArgs>
__attribute__((always_inline)) inline void
do_enzyme_call(AccArgs... acc)
{
__enzyme_fwddiff<qf_return_t>(wrapper_fn, acc...);
}
template <auto wrapper_fn, typename qf_return_t,
size_t CurO, size_t NO,
typename primals_t, typename derivs_t,
typename... AccArgs>
__attribute__((always_inline)) inline void
process_outputs(primals_t &primals, derivs_t &derivs, AccArgs... acc)
{
if constexpr (CurO == NO)
{
do_enzyme_call<wrapper_fn, qf_return_t>(acc...);
}
else
{
process_outputs<wrapper_fn, qf_return_t, CurO + 1, NO>(
primals, derivs,
acc...,
enzyme_dupnoneed,
&std::get<CurO>(primals),
&std::get<CurO>(derivs));
}
}
template <auto wrapper_fn, typename qf_return_t,
size_t CurI, size_t NI, bool... ActivityMap,
typename inputs_t, typename shadows_t,
typename primals_t, typename derivs_t,
typename... AccArgs>
__attribute__((always_inline)) inline void
process_inputs(inputs_t &inputs, shadows_t &shadows,
primals_t &primals, derivs_t &derivs,
AccArgs... acc)
{
if constexpr (CurI == NI)
{
constexpr size_t NO = std::tuple_size_v<primals_t>;
process_outputs<wrapper_fn, qf_return_t, 0, NO>(
primals, derivs, acc...);
}
else
{
constexpr bool active =
std::array<bool, sizeof...(ActivityMap)> {ActivityMap...} [CurI];
if constexpr (active)
{
std::cout << "Input[" << CurI << "]: ACTIVE (enzyme_dup)\n"
<< " primal ptr type: "
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n"
<< " shadow ptr type: "
<< get_type_name<decltype(&std::get<CurI>(shadows))>() << "\n";
}
else
{
std::cout << "Input[" << CurI << "]: INACTIVE (enzyme_const)\n"
<< " primal ptr type: "
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n";
}
if constexpr (active)
{
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
inputs, shadows, primals, derivs,
acc...,
enzyme_dup,
&std::get<CurI>(inputs),
&std::get<CurI>(shadows));
}
else
{
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
inputs, shadows, primals, derivs,
acc...,
enzyme_const,
&std::get<CurI>(inputs));
}
}
}
} // namespace enzyme_detail
template <size_t derivative_id, typename qfunc_t, typename inputs_t, typename outputs_t,
std::size_t... Is, std::size_t... Os>
inline void enzyme_fwddiff(
qfunc_t &qfunc,
const BlockVector &xq,
const BlockVector &shadow_xq,
BlockVector &yq,
const int &gnqp,
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
std::index_sequence<Is...>,
std::index_sequence<Os...>)
{
#ifdef MFEM_USE_ENZYME
constexpr std::size_t ninputs = sizeof...(Is);
constexpr std::size_t noutputs = sizeof...(Os);
using qf_signature = typename get_function_signature<qfunc_t>::type;
using qf_param_ts = typename qf_signature::parameter_ts;
using qf_return_t = typename qf_signature::return_t;
constexpr auto activity_map = make_activity_map<derivative_id>(inputs_t{});
static_assert(activity_map.size() == ninputs, "activity map size mismatch");
std::cout << "activity_map: ";
for (const auto &v : activity_map)
{
std::cout << v << " ";
}
std::cout << "\n";
auto inputs = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<Is, qf_param_ts>::type>>>(
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
auto shadows = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<Is, qf_param_ts>::type>>>(
shadow_xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
std::array<Vector, noutputs> primal_storage;
((primal_storage[Os].SetSize(yq.GetBlock(Os).Size())), ...);
auto primals_out = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
primal_storage[Os].ReadWrite(), &out_layouts[Os], gnqp)...);
auto derivs_out = std::make_tuple(
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
using wrapper_fn_t = qf_return_t (*)(
const qfunc_t &,
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...);
constexpr wrapper_fn_t wrapper_fn =
qfunction_wrapper<qfunc_t,
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...>;
// wrapper_fn travels as a non-type template parameter throughout without
// being stored.
enzyme_detail::process_inputs<
wrapper_fn,
qf_return_t,
0,
ninputs,
activity_map[Is]...
>(inputs, shadows,
primals_out, derivs_out,
enzyme_const, &qfunc // seed: qfunc is always inactive
);
#else
MFEM_ABORT("enzyme_fwddiff requires MFEM_USE_ENZYME");
#endif
}
} // namespace detail
// Create quadrature function fop to fields map
template <typename fops_t, size_t N = tuple_size<fops_t>::value, size_t M>
void create_fop_to_fd(const fops_t &fops,
const std::vector<FieldDescriptor> &fields,
std::array<size_t, M> &fop_to_fd)
{
static_assert(N == M, "sizes must match");
constexpr_for<0, N>([&](auto i)
{
const auto fop = get<i>(fops);
fop_to_fd[i] = std::numeric_limits<size_t>::max();
for (size_t j = 0; j < fields.size(); j++)
{
// TODO: output.GetFieldId() should probably store/return size_t
if (static_cast<int>(fields[j].id) == fop.GetFieldId())
{
fop_to_fd[i] = j;
}
}
// Handle Weight type. There is no FieldDescriptor for the weight.
// TODO: Create weight descriptor for the weight for internal use?
// TODO: this is a hack...
if (is_weight_fop<std::remove_cv_t<decltype(fop)>>::value)
{
fop_to_fd[i] = 0;
}
else if (fop_to_fd[i] == std::numeric_limits<size_t>::max())
{
MFEM_ABORT("not found");
}
});
}
template <typename fops_t, size_t nfops>
void create_qlayouts(const fops_t &fops,
const std::unordered_map<std::type_index, std::vector<int>> &a,
std::array<std::vector<int>, nfops> &b)
{
constexpr_for<0, nfops>([&](auto i)
{
using fop_t =
std::remove_cv_t<std::remove_reference_t<decltype(get<i>(fops))>>;
auto it = a.find(std::type_index(typeid(fop_t)));
if (it != a.end()) { b[i] = it->second; }
else { b[i].clear(); }
});
}
}
+96 -21
View File
@@ -11,44 +11,119 @@
#include "doperator.hpp"
#include <algorithm>
#ifdef MFEM_USE_MPI
using namespace mfem;
using namespace mfem::future;
void DifferentiableOperator::SetParameters(std::vector<Vector *> p) const
DifferentiableOperator::DifferentiableOperator(
const std::vector<FieldDescriptor> &infds,
const std::vector<FieldDescriptor> &outfds,
const ParMesh &mesh) :
Operator(),
mesh(mesh),
infds(infds),
outfds(outfds)
{
MFEM_ASSERT(parameters.size() == p.size(),
"number of parameters doesn't match descriptors");
for (size_t i = 0; i < parameters.size(); i++)
unionfds.clear();
unionfds.insert(unionfds.end(), infds.begin(), infds.end());
unionfds.insert(unionfds.end(), outfds.begin(), outfds.end());
std::sort(unionfds.begin(), unionfds.end());
auto last = std::unique(unionfds.begin(), unionfds.end());
unionfds.erase(last, unionfds.end());
infields_l.resize(infds.size());
for (size_t i = 0; i < infds.size(); i++)
{
p[i]->Read();
parameters_l[i] = *p[i];
infields_l[i] = new Vector(GetVSize(infds[i]));
}
infields_e.resize(infds.size());
}
DifferentiableOperator::DifferentiableOperator(
const std::vector<FieldDescriptor> &solutions,
const std::vector<FieldDescriptor> &parameters,
const ParMesh &mesh) :
mesh(mesh),
solutions(solutions),
parameters(parameters)
void DifferentiableOperator::SetMultLevel(MultLevel level)
{
fields.resize(solutions.size() + parameters.size());
fields_e.resize(fields.size());
solutions_l.resize(solutions.size());
parameters_l.resize(parameters.size());
mult_level = level;
}
for (size_t i = 0; i < solutions.size(); i++)
void DifferentiableOperator::Mult(const Vector &x, Vector &y) const
{
MFEM_ASSERT(!action_callbacks.empty(),
"no integrators have been set");
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&x),
"x needs to be a BlockVector");
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&y),
"y needs to be a BlockVector");
const auto &bx = static_cast<const BlockVector &>(x);
auto &by = static_cast<BlockVector &>(y);
Mult(bx, by);
}
void DifferentiableOperator::DisableTensorProductStructure(bool disable)
{
use_tensor_product_structure = !disable;
}
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
size_t derivative_id, const Vector &x)
{
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
derivative_action_callbacks.end(),
"no derivative action has been found for ID " << derivative_id);
const size_t dfidx = FindIdx(derivative_id, infds);
// Get transpose callbacks if available, otherwise pass empty vector
std::vector<derivative_action_t> transpose_callbacks;
auto it = daction_transpose_callbacks.find(derivative_id);
if (it != daction_transpose_callbacks.end())
{
fields[i] = solutions[i];
transpose_callbacks = it->second;
}
for (size_t i = 0; i < parameters.size(); i++)
return std::make_shared<DerivativeOperator>(
height,
GetTrueVSize(infds[dfidx]),
derivative_action_callbacks[derivative_id],
transpose_callbacks,
infds[dfidx],
x,
infds,
outfds);
}
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
size_t derivative_id, const MultiVector &x)
{
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
derivative_action_callbacks.end(),
"no derivative action has been found for ID " << derivative_id);
const size_t dfidx = FindIdx(derivative_id, infds);
// Get transpose callbacks if available, otherwise pass empty vector
std::vector<derivative_action_t> transpose_callbacks;
auto it = daction_transpose_callbacks.find(derivative_id);
if (it != daction_transpose_callbacks.end())
{
fields[i + solutions.size()] = parameters[i];
transpose_callbacks = it->second;
}
return std::make_shared<DerivativeOperator>(
height,
GetTrueVSize(infds[dfidx]),
derivative_action_callbacks[derivative_id],
transpose_callbacks,
infds[dfidx],
x,
infds,
outfds);
}
#endif // MFEM_USE_MPI
+236 -896
View File
File diff suppressed because it is too large Load Diff
+63
View File
@@ -0,0 +1,63 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../fespace.hpp"
#include "parameterspace.hpp"
namespace mfem::future
{
/// @brief FieldDescriptor struct
///
/// This struct is used to store information about a field.
struct FieldDescriptor
{
using data_variant_t =
std::variant<const FiniteElementSpace *,
const ParFiniteElementSpace *,
const QuadratureFunction *,
const ParameterSpace *>;
/// Field ID
std::size_t id;
/// Field variant
data_variant_t data;
/// Default constructor
FieldDescriptor() :
id(SIZE_MAX), data(data_variant_t{}) {}
/// Constructor
template <typename T>
FieldDescriptor(std::size_t field_id, const T* v) :
id(field_id), data(v) {}
bool operator==(const FieldDescriptor& other) const
{
return id == other.id;
}
bool operator<(const FieldDescriptor& other) const
{
return id < other.id;
}
friend void swap(FieldDescriptor& a, FieldDescriptor& b)
{
using std::swap;
swap(a.id, b.id);
swap(a.data, b.data);
}
};
}
+22
View File
@@ -0,0 +1,22 @@
#pragma once
#include "util.hpp"
namespace mfem::future
{
struct IntegratorContext
{
const ParMesh &mesh;
const Array<int> *elem_attr;
Array<int> attr;
int nentities;
const std::vector<FieldDescriptor> &infds;
const std::vector<FieldDescriptor> &outfds;
const std::vector<FieldDescriptor> &unionfds;
const IntegrationRule &ir;
std::unordered_map<std::type_index, std::vector<int>> &in_qlayouts;
std::unordered_map<std::type_index, std::vector<int>> &out_qlayouts;
};
}
+94 -9
View File
@@ -9,8 +9,23 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
// #define NVTX_COLOR nvtx::kPeru
#include "util.hpp"
#include "fem/kernels.hpp"
///////////////////////////////////////////////////////////////////////////////
template <class T>
inline std::enable_if_t<!std::numeric_limits<T>::is_integer, bool>
AlmostEq(T x, T y, T tolerance = 15.0 * std::numeric_limits<T>::epsilon())
{
const T neg = std::abs(x - y);
constexpr T min = std::numeric_limits<T>::min();
constexpr T eps = std::numeric_limits<T>::epsilon();
const T min_abs = std::min(std::abs(x), std::abs(y));
if (std::abs(min_abs) == 0.0) { return neg < eps; }
return (neg / (1.0 + std::max(min, min_abs))) < tolerance;
}
namespace mfem::future
{
@@ -30,6 +45,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
{
dbg("Value");
auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
@@ -94,10 +110,11 @@ void map_field_to_quadrature_data_tensor_product_3d(
else if constexpr (
is_gradient_fop<std::decay_t<field_operator_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
// dbg("Gradient");
const auto [q1d, B_dim, d1d] = B.GetShape();
const int vdim = input.vdim;
const int dim = input.dim;
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
const auto field = Reshape(&std::as_const(field_e[0]), d1d, d1d, d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
@@ -106,7 +123,30 @@ void map_field_to_quadrature_data_tensor_product_3d(
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
for (int vd = 0; vd < vdim; vd++)
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
// static constexpr int DIM = 3;
// MFEM_VERIFY(q1d <= MQ1, "q1d > MQ1");
// MFEM_SHARED real_t smem[MQ1][MQ1];
// kernels::internal::d_regs3d_t<DIM, MQ1> r0, r1;
// real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
/*
{
assert(B_dim == 1 && "1D B required!");
kernels::internal::LoadMatrix(d1d, q1d, B, sB);
kernels::internal::LoadMatrix(d1d, q1d, G, sG);
for (int qx = 0; qx < q1d; qx++)
{
for (int dx = 0; dx < d1d; dx++)
{
assert(AlmostEq(B(qx, 0, dx), sB[dx][qx]));
assert(AlmostEq(G(qx, 0, dx), sG[dx][qx]));
}
}
}*/
for (int c = 0; c < vdim; c++)
{
MFEM_FOREACH_THREAD(dz, z, d1d)
{
@@ -117,7 +157,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
real_t uv[2] = {0.0, 0.0};
for (int dx = 0; dx < d1d; dx++)
{
const real_t f = field(dx, dy, dz, vd);
const real_t f = field(dx, dy, dz, c);
uv[0] += f * B(qx, 0, dx);
uv[1] += f * G(qx, 0, dx);
}
@@ -163,19 +203,59 @@ void map_field_to_quadrature_data_tensor_product_3d(
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
}
fqp(vd, 0, qx, qy, qz) = uvw[0];
fqp(vd, 1, qx, qy, qz) = uvw[1];
fqp(vd, 2, qx, qy, qz) = uvw[2];
fqp(c, 0, qx, qy, qz) = uvw[0];
fqp(c, 1, qx, qy, qz) = uvw[1];
fqp(c, 2, qx, qy, qz) = uvw[2];
}
}
}
MFEM_SYNC_THREAD;
}
/*
{
for (int c = 0; c < vdim; c++)
{
kernels::internal::LoadDofs3d(d1d, c, field, r0);
for (int d = 0; d < DIM; d++)
{
for (int dz = 0; dz < d1d; dz++)
{
for (int dy = 0; dy < d1d; dy++)
{
for (int dx = 0; dx < d1d; dx++)
{
const real_t f = field(dx, dy, dz, c);
assert(AlmostEq(f, r0[d][dz][dy][dx]));
}
}
}
}
kernels::internal::Grad3d(d1d, q1d, smem, sB, sG, r0, r1, c);
for (int qz = 0; qz < q1d; qz++)
{
for (int qy = 0; qy < q1d; qy++)
{
for (int qx = 0; qx < q1d; qx++)
{
if (!AlmostEq(fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]))
{
dbg("\x1b[31m[{}:d] {} {}", c, fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]);
dbg("❌❌❌"), std::exit(EXIT_FAILURE);
}
}
}
}
}
// dbg("✅✅✅✅✅✅✅✅✅✅✅✅✅✅✅");//, std::exit(EXIT_SUCCESS);
}*/
}
// TODO: Create separate function for clarity
else if constexpr (
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
{
// dbg("None");
const int num_qp = integration_weights.GetShape()[0];
// TODO: eeek
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
@@ -518,6 +598,9 @@ void map_fields_to_quadrature_data(
const int &dimension,
const bool &use_sum_factorization = false)
{
// dbg();
assert(use_sum_factorization && "❌ use_sum_factorization required");
// When the input_to_field map returns -1, this means the requested input
// is the integration weight. Weights don't have a user defined field
// attached to them and we create a dummy field which is not accessed
@@ -578,6 +661,7 @@ void map_field_to_quadrature_data_conditional(
const int &dimension,
const bool &use_sum_factorization = false)
{
assert(false && "❌ condition not implemented");
if (condition)
{
if (use_sum_factorization)
@@ -619,6 +703,7 @@ void map_fields_to_quadrature_data_conditional(
const std::array<bool, num_inputs> &conditions,
const bool &use_sum_factorization = false)
{
assert(false && "❌ condition not implemented");
for_constexpr<num_inputs>([&](auto i)
{
map_field_to_quadrature_data_conditional(
@@ -627,7 +712,7 @@ void map_fields_to_quadrature_data_conditional(
});
}
template <size_t num_inputs, typename field_operator_ts>
template <int T_Q1D, size_t num_inputs, typename field_operator_ts>
MFEM_HOST_DEVICE
void map_direction_to_quadrature_data_conditional(
std::array<DeviceTensor<2>, num_inputs> &directions_qp,
@@ -660,7 +745,7 @@ void map_direction_to_quadrature_data_conditional(
}
else if (dimension == 3)
{
map_field_to_quadrature_data_tensor_product_3d(
map_field_to_quadrature_data_tensor_product_3d<T_Q1D>(
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
integration_weights, scratch_mem);
}
+18 -3
View File
@@ -43,7 +43,7 @@ public:
/// Get spatial dimension
///
/// returns always 1.
int Dimension() const
constexpr int Dimension() const
{
return 1;
}
@@ -74,11 +74,14 @@ public:
return elem_restr.get();
}
virtual const Operator* GetB() const = 0;
virtual const Operator* GetBt() const = 0;
protected:
int vdim;
DofToQuad dtq;
mutable std::unique_ptr<Operator> prolongation;
mutable std::unique_ptr<Operator> elem_restr;
mutable std::unique_ptr<Operator> prolongation, elem_restr, B, Bt;
};
/// @brief Uniform parameter space
@@ -122,6 +125,18 @@ public:
return lsize;
}
const Operator* GetB() const override
{
MFEM_ABORT("UniformParameterSpace does not support GetB");
return nullptr;
}
const Operator* GetBt() const override
{
MFEM_ABORT("UniformParameterSpace does not support GetBt");
return nullptr;
}
private:
/// T-vector size
int tsize;
+2
View File
@@ -243,6 +243,8 @@ void process_qf_arg(
}
}
// const tensor<real_t, DIM> ∇u
// const tensor<real_t, DIM, DIM> D (PA_DATA)
template <typename arg_type>
MFEM_HOST_DEVICE inline
void process_qf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
+76
View File
@@ -0,0 +1,76 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "tuple.hpp"
#include "../linalg/tensor.hpp"
using namespace mfem::future;
using mfem::future::tensor;
// Helper to add dimension to tensor type
template<typename T, int qp>
struct AddQPDimension;
// Specialization for tensor<real_t, dim>
template<typename real_t, int dim, int qp>
struct AddQPDimension<tensor<real_t, dim>, qp>
{
using type = tensor<real_t, dim, qp>;
};
// Specialization for tensor<real_t, dim, dim>
template<typename real_t, int dim, int qp>
struct AddQPDimension<tensor<real_t, dim, dim>, qp>
{
using type = tensor<real_t, dim, dim, qp>;
};
// Specialization for real_t (transforms to tensor<real_t, qp>)
template<typename real_t, int qp>
struct AddQPDimension
{
using type = tensor<real_t, qp>;
};
// Helper to transform tuple
template<typename Tuple, int qp>
struct TransformTupleQP {};
// Specialization for mfem::future::tuple
template<int qp, typename... Types>
struct TransformTupleQP<mfem::future::tuple<Types...>, qp>
{
using type = mfem::future::tuple<typename AddQPDimension<Types, qp>::type...>;
};
template<int qp, typename... Types>
struct TransformTupleQP<std::tuple<Types...>, qp>
{
using type = std::tuple<typename AddQPDimension<Types, qp>::type...>;
};
// Function to transform tuple type with qp dimension
template<int qp, typename qf_param_ts>
struct add_qp_dimension
{
using type = typename TransformTupleQP<qf_param_ts, qp>::type;
};
// Helper alias template for cleaner usage
template<int qp, typename qf_param_ts>
using add_qp_dimension_t = typename add_qp_dimension<qp, qf_param_ts>::type;
// ...AddDomainIntegrator...
// {
// constexpr int Q1D = 4;
// using qf_param_augmentd_ts = add_qp_dimension_t<Q1D, decay_tuple<qf_param_ts>>;
// }
+535 -66
View File
@@ -21,6 +21,7 @@
#include <type_traits>
#include <numeric>
#include <iomanip>
#include <typeindex>
#include "../../general/communication.hpp"
#include "../../general/forall.hpp"
@@ -28,13 +29,19 @@
#include "../fe/fe_base.hpp"
#include "../fespace.hpp"
#include "../pfespace.hpp"
#include "../qfunction.hpp"
#include "../../mesh/mesh.hpp"
#include "../../linalg/dtensor.hpp"
#include "../quadinterpolator.hpp"
#include "fielddescriptor.hpp"
#include "fieldoperator.hpp"
#include "parameterspace.hpp"
#include "tuple.hpp"
#undef NVTX_COLOR
#define NVTX_COLOR ::nvtx::kLightBlue
namespace mfem::future
{
@@ -75,7 +82,7 @@ constexpr void for_constexpr(lambda&& f,
}
template <typename lambda>
constexpr void for_constexpr(lambda&& f, std::integer_sequence<std::size_t>) {}
constexpr void for_constexpr(lambda&&, std::integer_sequence<std::size_t>) {}
template <int... n, typename lambda>
constexpr void for_constexpr(lambda&& f)
@@ -84,7 +91,7 @@ constexpr void for_constexpr(lambda&& f)
}
template <typename lambda, typename arg_t>
constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg,
constexpr void for_constexpr_with_arg(lambda&&, arg_t&&,
std::integer_sequence<std::size_t>)
{
// Base case - do nothing for empty sequence
@@ -108,6 +115,16 @@ constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg)
indices{});
}
template <auto start, auto end, auto inc = 1, typename F>
constexpr void constexpr_for(F&& f)
{
if constexpr (start < end)
{
f(std::integral_constant<decltype(start), start>());
constexpr_for<start + inc, end, inc>(f);
}
}
template <std::size_t I, typename Tuple, std::size_t... Is>
std::array<bool, sizeof...(Is)>
make_dependency_array(const Tuple& inputs, std::index_sequence<Is...>)
@@ -444,6 +461,21 @@ struct create_function_signature<output_t (*)(input_ts...)>
using type = FunctionSignature<output_t(input_ts...)>;
};
template <typename...>
using void_t = void;
template <typename T, typename = void>
struct get_function_signature
{
using type = typename create_function_signature<T>::type;
};
template <typename T>
struct get_function_signature<T, void_t<decltype(&T::operator())>>
{
using type = typename create_function_signature<decltype(&T::operator())>::type;
};
template <typename T>
constexpr int GetFieldId()
{
@@ -538,38 +570,12 @@ auto get_marked_entries(
/// @param t the tuple to filter fields from.
/// @returns a tuple containing only the fields with field IDs not equal to -1.
template <typename... Ts>
constexpr auto filter_fields(const std::tuple<Ts...>& t)
constexpr auto filter_fields(const std::tuple<Ts...>&)
{
return std::tuple_cat(
std::conditional_t<Ts::GetFieldId() != -1, std::tuple<Ts>, std::tuple<>> {}...);
}
/// @brief FieldDescriptor struct
///
/// This struct is used to store information about a field.
struct FieldDescriptor
{
using data_variant_t =
std::variant<const FiniteElementSpace *,
const ParFiniteElementSpace *,
const ParameterSpace *>;
/// Field ID
std::size_t id;
/// Field variant
data_variant_t data;
/// Default constructor
FieldDescriptor() :
id(SIZE_MAX), data(data_variant_t{}) {}
/// Constructor
template <typename T>
FieldDescriptor(std::size_t field_id, const T* v) :
id(field_id), data(v) {}
};
namespace dfem
{
template <class... T> constexpr bool always_false = false;
@@ -599,7 +605,7 @@ struct ThreadBlocks
#if defined(MFEM_USE_CUDA_OR_HIP)
template <typename func_t>
__global__ void forall_kernel_shmem(func_t f, int n)
__global__ void forall_kernel_extern_shmem(func_t f, int n)
{
int i = blockIdx.x;
extern __shared__ real_t shmem[];
@@ -608,23 +614,48 @@ __global__ void forall_kernel_shmem(func_t f, int n)
f(i, shmem);
}
}
template <typename func_t>
__global__ void forall_kernel_static_smem(func_t f, int n)
{
int i = blockIdx.x;
if (i >= n) { return; }
f(i, nullptr);
}
template <int MAX_THREADS_PER_BLOCK, typename func_t>
__global__
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
static void forall_kernel_static_smem_launch_bounds(func_t f, int n)
{
for (int k = blockIdx.x; k < n; k += gridDim.x) { f(k, nullptr); }
}
#endif
template <typename func_t>
template </*typename kernel_tag,*/ typename func_t>
void forall(func_t f,
const int &N,
const ThreadBlocks &blocks,
int num_shmem = 0,
[[maybe_unused]] const ThreadBlocks &blocks,
[[maybe_unused]] int num_shmem = 0,
real_t *shmem = nullptr)
{
db1();
if (Device::Allows(Backend::CUDA_MASK) ||
Device::Allows(Backend::HIP_MASK))
{
#if defined(MFEM_USE_CUDA_OR_HIP)
// int gridsize = (N + Z - 1) / Z;
int num_bytes = num_shmem * sizeof(decltype(shmem));
db1("num_bytes:{}", num_bytes);
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
dim3 block_size(blocks.x, blocks.y, blocks.z);
forall_kernel_shmem<<<N, block_size, num_bytes>>>(f, N);
// ForallKernel<kernel_tag>::run<<<N, block_size, num_bytes>>>(f, N);
if (num_bytes > 0)
{
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
}
else
{
forall_kernel_static_smem<<<N, block_size>>>(f, N);
}
#if defined(MFEM_USE_CUDA)
MFEM_GPU_CHECK(cudaGetLastError());
#elif defined(MFEM_USE_HIP)
@@ -635,6 +666,7 @@ void forall(func_t f,
}
else if (Device::Allows(Backend::CPU_MASK))
{
db1("CPU_MASK");
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
"Backend::CPU needs a pre-allocated shared memory block");
for (int i = 0; i < N; i++)
@@ -648,6 +680,69 @@ void forall(func_t f,
}
}
namespace dfem
{
template <int MAX_THREADS_PER_BLOCK = 0, typename func_t>
void forall(func_t f,
const int &N,
[[maybe_unused]] const ThreadBlocks &blocks,
[[maybe_unused]] int num_shmem = 0,
real_t *shmem = nullptr)
{
db1();
if (Device::Allows(Backend::CUDA_MASK) ||
Device::Allows(Backend::HIP_MASK))
{
#if defined(MFEM_USE_CUDA_OR_HIP)
int num_bytes = num_shmem * sizeof(decltype(shmem));
db1("num_bytes:{}", num_bytes);
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
db1("MAX_THREADS_PER_BLOCK:{}", MAX_THREADS_PER_BLOCK);
dim3 block_size(blocks.x, blocks.y, blocks.z);
if constexpr (MAX_THREADS_PER_BLOCK > 0)
{
assert(num_bytes == 0);
forall_kernel_static_smem_launch_bounds
<MAX_THREADS_PER_BLOCK><<<N, block_size>>> (f, N);
}
else
{
static_assert(MAX_THREADS_PER_BLOCK == 0);
if (num_bytes == 0)
{
forall_kernel_static_smem<<<N, block_size>>>(f, N);
}
else
{
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
}
}
#if defined(MFEM_USE_CUDA)
MFEM_GPU_CHECK(cudaGetLastError());
#elif defined(MFEM_USE_HIP)
MFEM_GPU_CHECK(hipGetLastError());
#endif
// MFEM_DEVICE_SYNC; // ⚠️
#endif
}
else if (Device::Allows(Backend::CPU_MASK))
{
db1("CPU_MASK");
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
"Backend::CPU needs a pre-allocated shared memory block");
for (int i = 0; i < N; i++)
{
f(i, shmem);
}
}
else
{
MFEM_ABORT("no compute backend available");
}
}
}
/// @todo To be removed.
class FDJacobian : public Operator
{
@@ -671,20 +766,6 @@ public:
MPI_COMM_WORLD);
}
Operator& GetGradient(const Vector &x0) const override
{
x = x0;
f.UseDevice(x.UseDevice());
xpev.UseDevice(x.UseDevice());
op.Mult(x, f);
const real_t xnorm_local = x.Norml2();
MPI_Allreduce(&xnorm_local, &xnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
MPI_COMM_WORLD);
return const_cast<FDJacobian&>(*this);
}
void Mult(const Vector &v, Vector &y) const override
{
// See [1] for choice of eps.
@@ -739,11 +820,11 @@ public:
private:
const Operator &op;
mutable Vector x, f;
Vector x, f;
mutable Vector xpev;
real_t lambda = 1.0e-6;
real_t fixed_eps;
mutable real_t xnorm;
real_t xnorm;
};
/// @brief Find the index of a field descriptor in a vector of field descriptors.
@@ -786,6 +867,10 @@ int GetVSize(const FieldDescriptor &f)
{
return arg->GetVSize();
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return arg->Size();
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->GetVSize();
@@ -824,6 +909,10 @@ void GetElementVDofs(const FieldDescriptor &f, int el, Array<int> &vdofs)
{
arg->GetElementVDofs(el, vdofs);
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
MFEM_ABORT("internal error");
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
MFEM_ABORT("internal error");
@@ -858,6 +947,10 @@ int GetTrueVSize(const FieldDescriptor &f)
{
return arg->GetTrueVSize();
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return arg->Size();
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->GetTrueVSize();
@@ -888,6 +981,10 @@ int GetVDim(const FieldDescriptor &f)
{
return arg->GetVDim();
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return arg->GetVDim();
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->GetVDim();
@@ -923,6 +1020,10 @@ int GetDimension(const FieldDescriptor &f)
return arg->GetMesh()->Dimension() - 1;
}
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return arg->GetSpace()->GetMesh()->Dimension();
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->Dimension();
@@ -935,6 +1036,36 @@ int GetDimension(const FieldDescriptor &f)
}, f.data);
}
inline
std::variant<const QuadratureInterpolator *, const Operator *>get_qinterp(
const FieldDescriptor &f,
const IntegrationRule &ir)
{
return std::visit([&ir](auto && arg) -> const QuadratureInterpolator*
{
using T = std::decay_t<decltype(arg)>;
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
std::is_same_v<T, const ParFiniteElementSpace *>)
{
return arg->GetQuadratureInterpolator(ir);
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
// QuadratureFunction doesn't need a QuadratureInterpolator
return nullptr;
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return nullptr;
}
else
{
static_assert(dfem::always_false<T>, "internal error");
}
return nullptr; // Unreachable, but avoids compiler warning
}, f.data);
}
/// @brief Get the prolongation operator for a field descriptor.
///
@@ -943,6 +1074,7 @@ int GetDimension(const FieldDescriptor &f)
inline
const Operator *get_prolongation(const FieldDescriptor &f)
{
NVTX("get P");
return std::visit([](auto&& arg) -> const Operator*
{
using T = std::decay_t<decltype(arg)>;
@@ -951,6 +1083,10 @@ const Operator *get_prolongation(const FieldDescriptor &f)
{
return arg->GetProlongationMatrix();
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return nullptr;
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->GetProlongationMatrix();
@@ -973,6 +1109,7 @@ inline
const Operator *get_element_restriction(const FieldDescriptor &f,
ElementDofOrdering o)
{
NVTX("get ER");
return std::visit([&o](auto&& arg) -> const Operator*
{
using T = std::decay_t<decltype(arg)>;
@@ -981,6 +1118,10 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
{
return arg->GetElementRestriction(o);
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return nullptr;
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return arg->GetElementRestriction(o);
@@ -1008,6 +1149,7 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
FaceType ft,
L2FaceValues m)
{
NVTX("get FR");
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
{
using T = std::decay_t<decltype(arg)>;
@@ -1016,6 +1158,11 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
{
return arg->GetFaceRestriction(o, ft, m);
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
// QuadratureFunction does not support face restrictions
MFEM_ABORT("internal error");
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
// ParameterSpace does not support face restrictions
@@ -1041,6 +1188,7 @@ inline
const Operator *get_restriction(const FieldDescriptor &f,
const ElementDofOrdering &o)
{
NVTX("get R");
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
return get_element_restriction(f, o);
@@ -1066,12 +1214,14 @@ inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
get_restriction_transpose(
const FieldDescriptor &f,
const ElementDofOrdering &o,
const fop_t &fop)
[[maybe_unused]] const fop_t &fop)
{
NVTX("get R^T");
if constexpr (is_sum_fop<fop_t>::value)
{
auto RT = [=](const Vector &v_e, Vector &v_l)
{
NVTX("R^T sum");
v_l += v_e;
};
return std::make_tuple(RT, 1);
@@ -1081,6 +1231,7 @@ get_restriction_transpose(
const Operator *R = get_restriction<entity_t>(f, o);
std::function<void(const Vector&, Vector&)> RT = [=](const Vector &x, Vector &y)
{
NVTX("R^T+");
R->AddMultTranspose(x, y);
};
return std::make_tuple(RT, R->Height());
@@ -1100,11 +1251,26 @@ get_restriction_transpose(
inline
void prolongation(const FieldDescriptor field, const Vector &x, Vector &field_l)
{
NVTX("P");
const auto P = get_prolongation(field);
NVTX_INI("SetSize");
field_l.SetSize(P->Height());
NVTX_END("SetSize");
NVTX_INI("P->Mult");
P->Mult(x, field_l);
}
inline
void prolongation_transpose(
const FieldDescriptor &field, const Vector &field_l, Vector &x)
{
const auto P = get_prolongation(field);
x.SetSize(P->Width());
P->MultTranspose(field_l, x);
}
/// @brief Apply the prolongation operator to a vector of fields.
///
/// x is a long vector containing the data for all fields on tdofs and
@@ -1121,6 +1287,7 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
const Vector &x,
std::array<Vector, M> &fields_l)
{
NVTX("P");
int data_offset = 0;
for (int i = 0; i < N; i++)
{
@@ -1128,9 +1295,14 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
const int width = P->Width();
// const Vector x_i(x.GetData() + data_offset, width);
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
fields_l[i].SetSize(P->Height());
NVTX_INI("SetSize");
fields_l[i].SetSize(P->Height());
NVTX_END("SetSize");
NVTX_INI("P->Mult");
P->Mult(x_i, fields_l[i]);
NVTX_END("P->Mult");
data_offset += width;
}
}
@@ -1144,20 +1316,259 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
/// @param fields the array of field descriptors.
/// @param x the input vector in tdofs.
/// @param fields_l the array of output vectors in vdofs.
// inline
// void prolongation(const std::vector<FieldDescriptor> fields,
// const Vector &x,
// std::vector<Vector> &fields_l)
// {
// int data_offset = 0;
// for (std::size_t i = 0; i < fields.size(); i++)
// {
// const auto P = get_prolongation(fields[i]);
// const int width = P->Width();
// const Vector x_i(const_cast<Vector&>(x), data_offset, width);
// fields_l[i].SetSize(P->Height());
// P->Mult(x_i, fields_l[i]);
// data_offset += width;
// }
// }
inline
void prolongation(const std::vector<FieldDescriptor> fields,
const Vector &x,
std::vector<Vector> &fields_l)
void prolongation(
const std::vector<FieldDescriptor> fields,
const BlockVector &x,
std::vector<Vector *> &x_l)
{
int data_offset = 0;
for (std::size_t i = 0; i < fields.size(); i++)
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
"error " << x.NumBlocks() << " vs " << x_l.size());
for (int i = 0; i < x.NumBlocks(); i++)
{
const auto P = get_prolongation(fields[i]);
const int width = P->Width();
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
fields_l[i].SetSize(P->Height());
P->Mult(x_i, fields_l[i]);
data_offset += width;
// If nullptr, assume Identity.
if (P == nullptr)
{
*x_l[i] = x.GetBlock(i);
}
else
{
const auto P = get_prolongation(fields[i]);
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
"prolongation not applicable to given input data size " <<
P->Width() << " vs " << x.GetBlock(i).Size());
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
"prolongation not applicable to given output data size " <<
P->Height() << " vs " << x_l[i]->Size());
P->Mult(x.GetBlock(i), *x_l[i]);
}
}
}
inline
void prolongation(
const std::vector<FieldDescriptor> fields,
const MultiVector &x,
std::vector<Vector *> &x_l)
{
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
"error " << x.NumBlocks() << " vs " << x_l.size());
for (int i = 0; i < x.NumBlocks(); i++)
{
const auto P = get_prolongation(fields[i]);
// If nullptr, assume Identity.
if (P == nullptr)
{
*x_l[i] = x[i];
}
else
{
const auto P = get_prolongation(fields[i]);
MFEM_ASSERT(P->Width() == x[i].Size(),
"prolongation not applicable to given input data size " <<
P->Width() << " vs " << x[i].Size());
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
"prolongation not applicable to given output data size " <<
P->Height() << " vs " << x_l[i]->Size());
P->Mult(x[i], *x_l[i]);
}
}
}
inline
void prolongation_transpose(
const std::vector<FieldDescriptor> fields,
const std::vector<Vector *> &x_l,
BlockVector &x)
{
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
"error " << x_l.size() << " vs " << x.NumBlocks());
for (size_t i = 0; i < x_l.size(); i++)
{
const auto P = get_prolongation(fields[i]);
// If nullptr, assume Identity.
if (P == nullptr)
{
x.GetBlock(i) = *x_l[i];
}
else
{
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
"prolongation not applicable to given input data size " <<
P->Height() << " vs " << x_l[i]->Size());
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
"prolongation not applicable to given output data size " <<
P->Width() << " vs " << x.GetBlock(i).Size());
P->MultTranspose(*x_l[i], x.GetBlock(i));
}
}
}
inline
void prolongation_transpose(
const std::vector<FieldDescriptor> fields,
const std::vector<Vector *> &x_l,
MultiVector &x)
{
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
"error " << x_l.size() << " vs " << x.NumBlocks());
for (size_t i = 0; i < x_l.size(); i++)
{
const auto P = get_prolongation(fields[i]);
// If nullptr, assume Identity.
if (P == nullptr)
{
x[i] = *x_l[i];
}
else
{
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
"prolongation not applicable to given input data size " <<
P->Height() << " vs " << x_l[i]->Size());
MFEM_ASSERT(P->Width() == x[i].Size(),
"prolongation not applicable to given output data size " <<
P->Width() << " vs " << x[i].Size());
P->MultTranspose(*x_l[i], x[i]);
}
}
}
template <typename entity_t>
void restriction(
const std::vector<FieldDescriptor> fields,
const std::vector<Vector *> &x_l,
std::vector<Vector *> &x_e)
{
MFEM_ASSERT(x_l.size() == x_e.size(),
"internal error " << x_l.size() << " vs " << x_e.size());
for (size_t i = 0; i < fields.size(); i++)
{
int s = 0;
const auto R = get_restriction<entity_t>(
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
// If nullptr, assume Identity.
if (R == nullptr)
{
s = x_l[i]->Size();
}
else
{
s = R->Height();
}
// TODO
if (x_e[i] == nullptr)
{
x_e[i] = new Vector(s);
}
x_e[i]->SetSize(s);
if (R == nullptr)
{
x_e[i] = x_l[i];
}
else
{
MFEM_ASSERT(R->Width() == x_l[i]->Size(),
"restriction not applicable to given input data size " <<
R->Width() << " vs " << x_l[i]->Size());
R->Mult(*x_l[i], *x_e[i]);
}
}
}
template <typename entity_t>
void prepare_residual(
const std::vector<FieldDescriptor> &fields,
std::vector<Vector *> &r_e)
{
for (size_t i = 0; i < fields.size(); i++)
{
int s = 0;
if (std::holds_alternative<const QuadratureFunction *>(fields[i].data))
{
const auto fd = std::get<const QuadratureFunction *>(fields[i].data);
s = fd->Size();
}
else
{
const auto R = get_restriction<entity_t>(
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
s = R->Height();
}
// TODO
if (r_e[i] == nullptr)
{
r_e[i] = new Vector(s);
}
else
{
r_e[i]->SetSize(s);
}
}
}
template <typename entity_t>
void restriction_transpose(
const std::vector<FieldDescriptor> &fields,
const std::vector<Vector *> &x_e,
std::vector<Vector *> &x_l)
{
for (size_t i = 0; i < fields.size(); i++)
{
int s = 0;
const auto R = get_restriction<entity_t>(
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
// TODO: if nullptr, assume Identity
if (R == nullptr)
{
s = x_e[i]->Size();
}
else
{
s = R->Width();
}
// TODO
if (x_l[i] == nullptr)
{
x_l[i] = new Vector(s);
}
x_l[i]->SetSize(s);
// TODO: if nullptr, assume Identity
if (R == nullptr)
{
x_l[i] = x_e[i];
}
else
{
R->MultTranspose(*x_e[i], *x_l[i]);
}
}
}
@@ -1166,6 +1577,7 @@ void get_lvectors(const std::vector<FieldDescriptor> fields,
const Vector &x,
std::vector<Vector> &fields_l)
{
NVTX("get_lvectors");
int data_offset = 0;
for (std::size_t i = 0; i < fields.size(); i++)
{
@@ -1192,13 +1604,15 @@ template <typename fop_t>
inline
std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
const FieldDescriptor &f,
const fop_t &fop,
[[maybe_unused]] const fop_t &fop,
MPI_Comm mpi_comm)
{
NVTX("get P^T");
if constexpr (is_sum_fop<fop_t>::value)
{
auto PT = [=](const Vector &r_local, Vector &y)
{
NVTX("P^T sum");
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
real_t local_sum = r_local.Sum();
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM, mpi_comm);
@@ -1209,6 +1623,7 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
{
auto PT = [=](const Vector &r_local, Vector &y)
{
NVTX("P^T Identity");
y = r_local;
};
return PT;
@@ -1216,6 +1631,7 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
const Operator *P = get_prolongation(f);
auto PT = [=](const Vector &r_local, Vector &y)
{
NVTX("P^T");
P->MultTranspose(r_local, y);
};
return PT;
@@ -1234,12 +1650,19 @@ void restriction(const FieldDescriptor u,
Vector &field_e,
ElementDofOrdering ordering)
{
NVTX("R");
const auto R = get_restriction<entity_t>(u, ordering);
MFEM_ASSERT(R->Width() == u_l.Size(),
"restriction not applicable to given data size");
const int height = R->Height();
NVTX_INI("SetSize");
field_e.SetSize(height);
NVTX_END("SetSize");
NVTX_INI("R->Mult");
R->Mult(u_l, field_e);
NVTX_END("R->Mult");
}
/// @brief Apply the restriction operator to a vector of fields.
@@ -1257,14 +1680,29 @@ void restriction(const std::vector<FieldDescriptor> u,
ElementDofOrdering ordering,
const int offset = 0)
{
NVTX("R");
for (std::size_t i = 0; i < u.size(); i++)
{
const auto R = get_restriction<entity_t>(u[i], ordering);
MFEM_ASSERT(R->Width() == u_l[i].Size(),
"restriction not applicable to given data size");
const int height = R->Height();
// NVTX_INI("SetSize");
fields_e[i + offset].SetSize(height);
R->Mult(u_l[i], fields_e[i + offset]);
// NVTX_END("SetSize");
// NVTX_INI("R->Mult");
if (dynamic_cast<const IdentityOperator*>(R))
{
NVTX("Identity");
fields_e[i + offset].NewMemoryAndSize(u_l[i].GetMemory(), u_l[i].Size(), false);
}
else
{
R->Mult(u_l[i], fields_e[i + offset]);
}
// NVTX_END("R->Mult");
}
}
@@ -1276,14 +1714,21 @@ void element_restriction(const std::array<FieldDescriptor, N> u,
ElementDofOrdering ordering,
const int offset = 0)
{
NVTX("ER");
for (int i = 0; i < N; i++)
{
const auto R = get_element_restriction(u[i], ordering);
MFEM_ASSERT(R->Width() == u_l[i].Size(),
"element restriction not applicable to given data size");
const int height = R->Height();
NVTX_INI("SetSize");
fields_e[i + offset].SetSize(height);
NVTX_END("SetSize");
NVTX_INI("R->Mult");
R->Mult(u_l[i], fields_e[i + offset]);
NVTX_END("R->Mult");
}
}
@@ -1340,6 +1785,10 @@ const DofToQuad *GetDofToQuad(const FieldDescriptor &f,
return &arg->GetTypicalTraceElement()->GetDofToQuad(ir, mode);
}
}
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
{
return nullptr;
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
return &arg->GetDofToQuad();
@@ -1471,7 +1920,7 @@ create_descriptors_to_fields_map(
auto f = [&](auto &fop, auto &map)
{
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
if constexpr (is_weight_fop<std::decay_t<decltype(fop)>>::value)
{
// TODO-bug: stealing dimension from the first field
fop.dim = GetDimension<entity_t>(fields[0]);
@@ -1601,7 +2050,7 @@ get_shmem_info(
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
const std::vector<FieldDescriptor> &fields,
const int &num_entities,
const input_t &inputs,
[[maybe_unused]] const input_t &inputs,
const int &num_qp,
const std::vector<int> &input_size_on_qp,
const int &residual_size_on_qp,
@@ -2356,5 +2805,25 @@ std::array<DofToQuadMap, num_fields> create_dtq_maps(
std::make_index_sequence<num_fields> {});
}
struct QLayoutEntry
{
std::type_index type;
std::vector<int> layout;
template <class Fop>
QLayoutEntry(Fop, std::initializer_list<int> idx) :
type(typeid(Fop)), layout(idx) {}
};
static void ExtractQLayouts(
const std::initializer_list<QLayoutEntry> entries,
std::unordered_map<std::type_index, std::vector<int>>& out)
{
for (const auto& e : entries)
{
out[e.type] = e.layout;
}
}
} // namespace mfem::future
#endif
+1 -1
View File
@@ -57,7 +57,7 @@ void DGMassApply(const int e,
}
else if (DIM == 3)
{
SmemPAMassApply3D_Element<TD1D,TQ1D,NBZ,ACCUM>(e, NE, B, pa_data, x, y);
SmemPAMassApply3D_Element<TD1D,TQ1D,ACCUM>(e, NE, B, pa_data, x, y);
}
else
{
+1 -42
View File
@@ -167,15 +167,7 @@ public:
/** @brief Full multidimensional representation which does not use tensor
product structure. The ordering of the degrees of freedom is the
same as TENSOR, but the sizes of B and G are the same as FULL.*/
LEXICOGRAPHIC_FULL,
/** @brief Ragged tensor product representation using 1D matrices/tensors
with dimensions using 1D number of quadrature points and ragged tensor degrees of
freedom. */
/** Used only for partial assembly of the H1 positive basis. The
size of B is d1d x qnpt x dim. Since different Gauss-Jacobi quadrature rules
are employed in each dimension, we need to store dim arrays. */
RAGGED_TENSOR
LEXICOGRAPHIC_FULL
};
/// Describes the contents of the #B, #Bt, #G, and #Gt arrays, see #Mode.
@@ -236,39 +228,6 @@ public:
const Array<DofToQuad*> &dof2quad_array,
const IntegrationRule &ir,
DofToQuad::Mode mode);
virtual ~DofToQuad() = default;
};
/** @brief Structure representing the matrices/tensors needed to evaluate (in
reference space) the values, gradients, divergences, or curls of a positive
FiniteElement on simplices at the quadrature points of Stroud conical quadrature. */
class RaggedDofToQuad : public DofToQuad
{
public:
/** @brief Special basis function structures for positive (Bernstein) basis with
partial assembly. The storage layout of Ba1 is ndof x nqpt for scalar elements.
The storage layout of Ba2 is ndof x ndof x nqpt. In particular, we have
Ba2(iqpt, a1, a2) = B^{p-a1}_{a2}(x_{iqpt}). */
Array<real_t> Ba1, Ba2, Ba3;
Array<real_t> Ba1t, Ba2t, Ba3t;
/** @brief Special structures for gradients of positive basis with partial assembly.
The gradient arrays exploit properties of the Bernstein basis which allow grad(B^p_alpha)
to be expressed as the sum of products of B^{p-1}_alpha and the barycentric coordinates.
Thus, Ga1 and Ga2 simply contain the ragged tensor product components of B^{p-1}_alpha */
Array<real_t> Ga1, Ga2, Ga3;
Array<real_t> Ga1t, Ga2t, Ga3t;
/** @brief Mapping from the Bernstein multi-index (a_1, ..., a_d) to the lexicographic
dof index. */
Array<int> lex_map;
Array<int> forward_map2d_diff, forward_map3d_diff;
Array<int> inverse_map2d_diff, inverse_map3d_diff;
Array<int> forward_map2d_mass, forward_map3d_mass;
Array<int> inverse_map2d_mass, inverse_map3d_mass;
};
/// Describes the function space on each element
-302
View File
@@ -557,101 +557,6 @@ H1Pos_TriangleElement::H1Pos_TriangleElement(const int p)
}
}
const DofToQuad &H1Pos_TriangleElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt*ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize((int)nqpt*ndof*ndof);
rd2q->Ba1t.SetSize(nqpt*ndof);
rd2q->Ba2t.SetSize((int)nqpt*ndof*ndof);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt*(ndof -1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->Ga1t.SetSize(nqpt*(ndof -1));
rd2q->Ga2t.SetSize(nqpt*(ndof-1)*(ndof -1));
rd2q->lex_map.SetSize(ndof * ndof);
Vector shape_a1(ndof), shape_a2(ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2((ndof-1) * (ndof-1));
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). Additionally, the Bernstein PA algorithms expect evaluation of the
// component 1D bases at the Stroud nodes pulled back to the unit square, so perform the pullback
// on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
rd2q->Ba2t[i + nqpt*(j + ndof*k)] = rd2q->Ba2[k + ndof*(j + ndof*i)] = shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
rd2q->Ga2t[i + nqpt*(j + (ndof-1)*k)] = rd2q->Ga2[k + (ndof-1)*(j +
(ndof-1)*i)] = shape_Ga2(k);
}
}
}
}
// stores the mapping from 2D Bernstein multi-index (i,j,p-i-j) to the
// lexicographic DOF ordering
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
int idx = ((2 * (ndof-1) + 3) - j) * j / 2 + i;
rd2q->lex_map[j + ndof*i] = idx;
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TriangleElement::CalcShape(
const int p, const real_t l1, const real_t l2, real_t *shape)
@@ -844,213 +749,6 @@ H1Pos_TetrahedronElement::H1Pos_TetrahedronElement(const int p)
}
}
const DofToQuad &H1Pos_TetrahedronElement::GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array)
{
DofToQuad *d2q = nullptr;
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
d2q = new RaggedDofToQuad;
const int ndof = fe.GetOrder() + 1; // verify
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
const int basis_dim2d = ndof*(ndof+1) / 2;
const int basis_dim3d = ndof*(ndof+1)*(ndof+2) / 6;
const int basis_dim2d_diff = (ndof-1)*(ndof) / 2;
const int basis_dim3d_diff = (ndof-1)*(ndof)*(ndof+1) / 6;
d2q->FE = &fe;
d2q->IntRule = &ir;
d2q->mode = mode;
d2q->ndof = ndof;
d2q->nqpt = nqpt;
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
rd2q->Ba1.SetSize(nqpt * ndof);
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba2.SetSize(nqpt * basis_dim2d);
// third component of ragged tensor basis, technically dof*(dof-1)/2 entries
rd2q->Ba3.SetSize(nqpt * basis_dim3d);
rd2q->Ba1t.SetSize(nqpt * ndof);
rd2q->Ba2t.SetSize(nqpt * basis_dim2d);
rd2q->Ba3t.SetSize(nqpt * basis_dim3d);
// stores first component of ragged tensor basis with order p-1, for gradients only
rd2q->Ga1.SetSize(nqpt * (ndof-1));
// stores second component of ragged tensor basis with order p-1
rd2q->Ga2.SetSize(nqpt * basis_dim2d_diff);
// stores third component of ragged tensor basis with order p-1
rd2q->Ga3.SetSize(nqpt * basis_dim3d_diff);
rd2q->Ga1t.SetSize(nqpt * (ndof-1));
rd2q->Ga2t.SetSize(nqpt * basis_dim2d_diff);
rd2q->Ga3t.SetSize(nqpt * basis_dim3d_diff);
rd2q->lex_map.SetSize(ndof * ndof * ndof);
rd2q->forward_map2d_diff.SetSize((ndof-1) * (ndof-1));
rd2q->forward_map3d_diff.SetSize((ndof-1) * (ndof-1) * (ndof-1));
rd2q->inverse_map2d_diff.SetSize(2 * basis_dim2d_diff);
rd2q->inverse_map3d_diff.SetSize(3 * basis_dim3d_diff);
rd2q->forward_map2d_mass.SetSize(ndof * ndof);
rd2q->forward_map3d_mass.SetSize(ndof * ndof * ndof);
rd2q->inverse_map2d_mass.SetSize(2 * basis_dim2d);
rd2q->inverse_map3d_mass.SetSize(2 * basis_dim3d);
// forward and inverse maps for multi-index to collpased 1d index for diffusion, can combine
// these four loops, but need four idx's and clause for shorter diff loops
int idx = 0;
for (int i = 0; i < ndof-1; i++)
{
for (int j = 0; j < ndof-i-1; j++)
{
rd2q->forward_map2d_diff[j + (ndof-1)*i] = idx;
rd2q->inverse_map2d_diff[2*idx] = i;
rd2q->inverse_map2d_diff[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof-1; k++)
{
for (int j = 0; j < ndof-k-1; j++)
{
for (int i = 0; i < ndof-k-j-1; i++)
{
rd2q->forward_map3d_diff[k + (ndof-1)*(j + (ndof-1)*i)] = idx;
rd2q->inverse_map3d_diff[3*idx] = i;
rd2q->inverse_map3d_diff[1 + 3*idx] = j;
rd2q->inverse_map3d_diff[2 + 3*idx] = k;
idx++;
}
}
}
// forward and inverse maps for multi-index to collpased 1d index for mass
idx = 0;
for (int j = 0; j < ndof; j++)
{
for (int i = 0; i < ndof-j; i++)
{
rd2q->forward_map2d_mass[j + ndof*i] = idx;
rd2q->inverse_map2d_mass[2*idx] = i;
rd2q->inverse_map2d_mass[1 + 2*idx] = j;
idx++;
}
}
idx = 0;
for (int k = 0; k < ndof; k++)
{
for (int j = 0; j < ndof-k; j++)
{
for (int i = 0; i < ndof-k-j; i++)
{
rd2q->forward_map3d_mass[k + ndof*(j + ndof*i)] = idx;
rd2q->inverse_map3d_mass[2*idx] = i;
rd2q->inverse_map3d_mass[1 + 2*idx] = j;
// d2q->inverse_map3d_mass[2 + 3*idx] = k;
idx++;
}
}
}
Vector shape_a1(ndof), shape_a2(ndof * ndof), shape_a3(ndof * ndof * ndof);
Vector shape_Ga1(ndof-1), shape_Ga2(ndof-1), shape_Ga3(ndof-1);
for (int i = 0; i < nqpt; i++)
{
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
// Gauss-Jacobi rule). The first 'nqpt' points in the third dimension have the same z-coordinates
// as those of the 1D rule for the third dimension (i.e. Gauss-Legendre rule). Additionally,
// the Bernstein PA algorithms expect evaluation of the component 1D bases at the Stroud nodes
// pulled back to the unit cube, so perform the pullback on the fly.
const real_t x = ir.IntPoint(i).x;
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
const real_t z = ir.IntPoint(nqpt*nqpt*i).z / (1.0 - ir.IntPoint(
nqpt*nqpt*i).x - ir.IntPoint(nqpt*nqpt*i).y);
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
for (int j = 0; j < ndof; j++)
{
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
if (j < ndof-1)
{
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
}
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
for (int k = 0; k < ndof-j; k++)
{
const int a_2d_mass = rd2q->forward_map2d_mass[k + ndof*j];
rd2q->Ba2t[i + nqpt*a_2d_mass] = rd2q->Ba2[a_2d_mass + basis_dim2d*i] =
shape_a2(
k);
if (j < ndof-1 && k < ndof-j-1)
{
const int a_2d_diff = rd2q->forward_map2d_diff[k + (ndof-1)*j];
rd2q->Ga2t[i + nqpt*a_2d_diff] = rd2q->Ga2[a_2d_diff + basis_dim2d_diff*i] =
shape_Ga2(k);
Poly_1D::CalcBernstein(ndof-2-j-k, z, shape_Ga3);
}
Poly_1D::CalcBernstein(ndof-1-j-k, z, shape_a3);
for (int m = 0; m < ndof-j-k; m++)
{
const int a_3d_mass = rd2q->forward_map3d_mass[m + ndof*(k + ndof*j)];
rd2q->Ba3t[i + nqpt*a_3d_mass] = rd2q->Ba3[a_3d_mass + basis_dim3d*i] =
shape_a3(
m);
if (j < ndof-1 && k < ndof-j-1 && m < ndof-j-k-1)
{
// // collapsed 1D access
// d2q->Ga3[i + nqpt*(m + d2q->offset3d[k + (ndof-1)*j])] = shape_Ga3(m);
// collapsed 1D access with forward mapping
const int a_3d_diff = rd2q->forward_map3d_diff[m + (ndof-1)*(k + (ndof-1)*j)];
rd2q->Ga3t[i + nqpt*a_3d_diff] = rd2q->Ga3[a_3d_diff + basis_dim3d_diff*i] =
shape_Ga3(m);
}
}
}
}
}
// stores the mapping from 3D Bernstein multi-index (i,j,k,p-i-j-k) to the
// lexicographic DOF ordering
int p = ndof - 1;
for (int i = 0; i < ndof; i++)
{
for (int j = 0; j < ndof-i; j++)
{
for (int k = 0; k < ndof-i-j; k++)
{
int dof = (p+1)*(p+2)*(p+3) / 6;
int tet = (p-k)*(p-k+1)*(p-k+2) / 6;
int tri = (p+1-k-j)*(p+2-k-j)/2;
int multi_idx = dof - tet - tri + i;
rd2q->lex_map[k + ndof*(j + ndof*i)] = multi_idx;
}
}
}
dof2quad_array.Append(d2q);
}
}
return *d2q;
}
// static method
void H1Pos_TetrahedronElement::CalcShape(
const int p, const real_t l1, const real_t l2, const real_t l3,
-30
View File
@@ -191,21 +191,6 @@ public:
/// Construct the H1Pos_TriangleElement of order @a p
H1Pos_TriangleElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)/2 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
real_t *shape);
@@ -235,21 +220,6 @@ public:
/// Construct the H1Pos_TetrahedronElement of order @a p
H1Pos_TetrahedronElement(const int p);
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const override
{
return (mode == DofToQuad::RAGGED_TENSOR) ?
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
FiniteElement::GetDofToQuad(ir, mode);
}
static const DofToQuad &GetRaggedTensorDofToQuad(
const FiniteElement &fe, const IntegrationRule &ir,
DofToQuad::Mode mode,
Array<DofToQuad*> &dof2quad_array);
const Array<int> &GetDofMap() const { return dof_map; }
// The size of shape is (p+1)(p+2)(p+3)/6 (dof).
static void CalcShape(const int p, const real_t x, const real_t y,
const real_t z, real_t *shape);
-27
View File
@@ -250,14 +250,6 @@ public:
its GetOrder() method. */
virtual FiniteElementCollection *Clone(int p) const;
/** @brief Return the order parameter used to construct this collection.
* This differs from GetOrder() depending on the collection type. */
virtual int GetConstructorOrder() const
{
MFEM_ABORT("Collection " << Name() << " does not support GetConstructorOrder");
return -1;
}
protected:
const int base_p; ///< Order as returned by GetOrder().
@@ -322,9 +314,6 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new H1_FECollection(p, dim, b_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~H1_FECollection();
};
@@ -354,10 +343,6 @@ class H1_Trace_FECollection : public H1_FECollection
public:
H1_Trace_FECollection(const int p, const int dim,
const int btype = BasisType::GaussLobatto);
FiniteElementCollection *Clone(int p) const override
{ return new H1_Trace_FECollection(p, dim+1, b_type); }
};
/// Arbitrary order "L2-conforming" discontinuous finite elements.
@@ -411,9 +396,6 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new L2_FECollection(p, dim, b_type, m_type); }
int GetConstructorOrder() const override
{ return base_p; }
virtual ~L2_FECollection();
};
@@ -474,9 +456,6 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new RT_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return base_p-1; }
virtual ~RT_FECollection();
};
@@ -557,9 +536,6 @@ public:
FiniteElementCollection *Clone(int p) const override
{ return new ND_FECollection(p, dim, cb_type, ob_type); }
int GetConstructorOrder() const override
{ return dim>1 ? base_p : base_p+1; }
virtual ~ND_FECollection();
};
@@ -572,9 +548,6 @@ public:
ND_Trace_FECollection(const int p, const int dim,
const int cb_type = BasisType::GaussLobatto,
const int ob_type = BasisType::GaussLegendre);
FiniteElementCollection *Clone(int p) const override
{ return new ND_Trace_FECollection(p, dim+1, cb_type, ob_type); }
};
/// Arbitrary order 3D H(curl)-conforming Nedelec finite elements in 1D.
+1 -1
View File
@@ -52,7 +52,7 @@
#include "bounds.hpp"
#include "particleset.hpp"
#include "dfem/doperator.hpp"
// #include "dfem/doperator.hpp"
#ifdef MFEM_USE_MPI
#include "pfespace.hpp"
+2 -1
View File
@@ -4631,8 +4631,9 @@ FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
ElementDofOrdering GetEVectorOrdering(const FiniteElementSpace& fes)
{
return (UsesTensorBasis(fes) || fes.UsesRaggedTensorBasis()) ?
return UsesTensorBasis(fes)?
ElementDofOrdering::LEXICOGRAPHIC:
ElementDofOrdering::NATIVE;
}
} // namespace mfem
-12
View File
@@ -1514,18 +1514,6 @@ public:
return dynamic_cast<const L2_FECollection*>(fec) != NULL;
}
/// @brief Return true if the mesh contains only one topology, the elements are
/// all triangles or tetrahedrons, and the elements are ragged tensor elements
/// i.e. Bernstein/positive basis.
bool UsesRaggedTensorBasis() const
{
bool simplex = this->GetMesh()->IsSimplexMesh();
bool positive =
dynamic_cast<const mfem::H1Pos_TriangleElement *>(this->GetTypicalFE()) ||
dynamic_cast<const mfem::H1Pos_TetrahedronElement *>(this->GetTypicalFE());
return simplex && positive;
}
/** In variable-order spaces on nonconforming (NC) meshes, this function
controls whether strict conformity is enforced in cases where coarse
edges/faces have higher polynomial order than their fine NC neighbors.
-167
View File
@@ -2256,104 +2256,6 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
}
}
void GridFunction::AccumulateAndCountTraceValues(
Coefficient *coeff[], VectorCoefficient *vcoeff,
Array<int> &values_counter)
{
if (vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
"vcoeff vdim != fes VDim");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::VALUE &&
fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR,
"Can only call ProjectTraceCoefficient on scalar value-type "
"trace elements. "
"Use ProjectTraceCoefficientNormal for RT and "
"ProjectTraceCoefficientTangent for ND finite elements.");
}
Array<int> vdofs;
Vector vc;
values_counter.SetSize(Size());
values_counter = 0;
const int vdim = fes->GetVDim();
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
const FiniteElement *fe = fes->GetFaceElement(i);
const int fdof = fe->GetDof();
ElementTransformation *transf = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
fes->GetFaceVDofs(i, vdofs);
for (int j = 0; j < fdof; j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
transf->SetIntPoint(&ip);
if (vcoeff) { vcoeff->Eval(vc, *transf, ip); }
for (int d = 0; d < vdim; d++)
{
if (!vcoeff && !coeff[d]) { continue; }
real_t val = vcoeff ? vc(d) : coeff[d]->Eval(*transf, ip);
int ind = vdofs[fdof*d+j];
if ( ind < 0 )
{
val = -val, ind = -1-ind;
}
if (++values_counter[ind] == 1)
{
(*this)(ind) = val;
}
else
{
(*this)(ind) += val;
}
}
}
}
}
void GridFunction::AccumulateAndCountTraceTangentValues(
VectorCoefficient &vcoeff, Array<int> &values_counter)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()
->GetRangeType() == FiniteElement::VECTOR &&
fes->GetTypicalTraceElement()
->GetMapType() == FiniteElement::H_CURL,
"Not an ND FE space!");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetPhysRangeDim(
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
"vcoeff vdim != PhysRangeDim");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
Vector lvec;
values_counter.SetSize(Size());
values_counter = 0;
HostReadWrite();
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
fes->GetFaceVDofs(i, dofs);
lvec.SetSize(fe->GetDof());
fe->Project(vcoeff, *T, lvec);
accumulate_dofs(dofs, lvec, *this, values_counter);
}
}
void GridFunction::ComputeMeans(AvgType type, Array<int> &zones_per_vdof)
{
switch (type)
@@ -2796,74 +2698,6 @@ void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
}
}
void GridFunction::ProjectTraceCoefficient(Coefficient *coeff[])
{
Array<int> values_counter;
AccumulateAndCountTraceValues(coeff, NULL, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficient(Coefficient &coeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == 1, "ProjectTraceCoefficient(Coefficient&)"
"is only valid for scalar GridFunction");
Coefficient *coeff_p = &coeff;
ProjectTraceCoefficient(&coeff_p);
}
void GridFunction::ProjectTraceCoefficient(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(FESpace()->GetVDim() == vcoeff.GetVDim(),
"Incompatible vcoeff vdim and fes vdim");
Array<int> values_counter;
AccumulateAndCountTraceValues(NULL, &vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff)
{
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetRangeType() ==
FiniteElement::SCALAR &&
fes->GetTypicalTraceElement()->GetMapType() ==
FiniteElement::INTEGRAL, "Not an RT FE space!");
MFEM_VERIFY(vcoeff.GetVDim() == fes->GetMesh()->SpaceDimension(),
"vcoeff vdim (" << vcoeff.GetVDim()
<< ") != SpaceDimension ("
<< fes->GetMesh()->SpaceDimension() << ")");
const FiniteElement *fe;
ElementTransformation *T;
Array<int> dofs;
int dim = vcoeff.GetVDim();
Vector vc(dim), nor(dim), lvec;
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
{
fe = fes->GetFaceElement(i);
T = fes->GetMesh()->GetFaceTransformation(i);
const IntegrationRule &ir = fe->GetNodes();
lvec.SetSize(fe->GetDof());
for (int j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
T->SetIntPoint(&ip);
vcoeff.Eval(vc, *T, ip);
CalcOrtho(T->Jacobian(), nor);
lvec(j) = (vc * nor);
}
fes->GetFaceVDofs(i, dofs);
SetSubVector(dofs, lvec);
}
}
void GridFunction::ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff)
{
Array<int> values_counter;
AccumulateAndCountTraceTangentValues(vcoeff, values_counter);
ComputeMeans(ARITHMETIC, values_counter);
}
void GridFunction::ProjectCoefficientGlobalL2(VectorCoefficient &vcoeff,
real_t rtol, int iter)
{
@@ -5452,7 +5286,6 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
Vector lel, uel;
GetElementBounds(plb, lel, uel, vdim);
+3 -27
View File
@@ -578,13 +578,6 @@ protected:
const Array<int> &bdr_attr,
Array<int> &values_counter);
void AccumulateAndCountTraceValues(Coefficient *coeff[],
VectorCoefficient *vcoeff,
Array<int> &values_counter);
void AccumulateAndCountTraceTangentValues(VectorCoefficient &vcoeff,
Array<int> &values_counter);
// Complete the computation of averages; called e.g. after
// AccumulateAndCountZones().
void ComputeMeans(AvgType type, Array<int> &zones_per_vdof);
@@ -670,23 +663,6 @@ public:
ProjectBdrCoefficient(&coeff_p, attr);
}
/// Project a Coefficient on a GridFunction defined on H1 trace space
void ProjectTraceCoefficient(Coefficient *coeff[]);
void ProjectTraceCoefficient(Coefficient &coeff);
/** @brief Project a VectorCoefficient @a vcoeff on a GridFunction
defined on a Vector H1 trace space. Note that this also works
for a scalar H1 trace space, where only the first component of
@a vcoeff is used. */
void ProjectTraceCoefficient(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an RT trace space */
void ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on a GridFunction
defined on an ND trace space */
void ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff);
/** @brief Project a VectorCoefficient on the GridFunction, modifying only
DOFs on the boundary associated with the boundary attributes marked in
the @a attr array. */
@@ -1791,8 +1767,8 @@ public:
const int ref_factor=1, const int vdim=-1) const;
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each element
/// ordered byNODES:
/// points based on \p ref_factor, and returns the bounds for each element
/// ordered byNodes:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}. We also return the
/// PLBound object used to compute the bounds.
@@ -1826,7 +1802,7 @@ public:
const int vdim = -1) const;
/// Compute bounds on the grid function for all the elements. The bounds
/// are returned in @b lower and @b upper, ordered byNODES:
/// are returned in @b lower and @b upper, ordered byNodes:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
+390 -3130
View File
File diff suppressed because it is too large Load Diff
+103 -494
View File
@@ -12,9 +12,6 @@
#ifndef MFEM_GSLIB
#define MFEM_GSLIB
#include <map>
#include <vector>
#include "../config/config.hpp"
#ifdef MFEM_USE_MPI
#include "pgridfunc.hpp"
@@ -24,45 +21,6 @@
#ifdef MFEM_USE_GSLIB
/* gslib license and copyright statement for code adapted from gslib:
Copyright (c) 2008-2024, UCHICAGO ARGONNE, LLC.
The UChicago Argonne, LLC as Operator of Argonne National
Laboratory holds copyright in the Software. The copyright holder
reserves all rights except those expressly granted to licensees,
and U.S. Government license rights.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions
are met:
1. Redistributions of source code must retain the above copyright
notice, this list of conditions and the disclaimer below.
2. Redistributions in binary form must reproduce the above copyright
notice, this list of conditions and the disclaimer (as noted below)
in the documentation and/or other materials provided with the
distribution.
3. Neither the name of ANL nor the names of its contributors
may be used to endorse or promote products derived from this software
without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL
UCHICAGO ARGONNE, LLC, THE U.S. DEPARTMENT OF
ENERGY OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED
TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*/
namespace gslib
{
struct comm;
@@ -122,18 +80,13 @@ protected:
// IntegrationRules for simplex->Quad/Hex and to project to p_max in-case of
// p-refinement.
Array<IntegrationRule *> ir_split;
/// Integration rules built at the field polynomial order (only for surface
/// meshes when mesh order is not the same as gridfunction order).
Array<IntegrationRule *> ir_split_sol;
/// Order at which #ir_split_sol was built; -1 means not built.
int ir_split_sol_order = -1;
Array<FiniteElementSpace *> fes_rst_map; //FESpaces to map Quad/Hex->Simplex
Array<GridFunction *> gf_rst_map; // GridFunctions to map Quad/Hex->Simplex
FiniteElementCollection *fec_map_lin;
void *fdataD;
struct gslib::crystal *cr; // gslib's internal data
struct gslib::comm *gsl_comm; // gslib's internal data
int dim, spacedim, points_cnt; // mesh dimension and number of points
int dim, points_cnt; // mesh dimension and number of points
Array<unsigned int> gsl_code, gsl_proc, gsl_elem, gsl_mfem_elem;
Vector gsl_mesh, gsl_ref, gsl_dist, gsl_mfem_ref;
Array<unsigned int> recv_proc, recv_index; // data for custom interpolation
@@ -142,8 +95,6 @@ protected:
AvgType avgtype; // average type used for L2 functions
Array<int> split_element_map;
Array<int> split_element_index;
// Geometry::Type (as int) of the original element for each split quad.
Array<int> split_element_geom;
int NE_split_total; // total number of elements after mesh splitting
int mesh_points_cnt; // number of mesh nodes
// Tolerance to ignore points found beyond the mesh boundary.
@@ -151,235 +102,111 @@ protected:
double bdr_tol;
// Use CPU functions for Mesh/GridFunction on device for gslib1.0.7
bool gpu_to_cpu_fallback = false;
// Check if a point is inside the oriented bounding box of an
// element before the Newton iteration.
// Note: only used in MFEM implementation (not in gslib) which currently
// supports GPU kernels for area meshes in 2D, volume meshes in 3D,
// and surface meshes in 1D/2D/3D.
bool obb_check = true;
// Device specific data used for FindPoints
struct DEV_STRUCT
struct
{
bool setup_device = false;
bool find_device = false;
int local_hash_size, dof1d, dof1d_sol, lh_nx, gh_nx;
int local_hash_size, dof1d, dof1d_sol, h_o_size, h_nx;
double newt_tol; // Tolerance specified during setup for Newton solve
struct gslib::crystal *cr;
struct gslib::hash_data_3 *hash3;
struct gslib::hash_data_2 *hash2;
mutable Vector bb, wtend, gll1d, lagcoeff, gll1d_sol, lagcoeff_sol;
mutable Array<unsigned int> lh_offset, gh_offset;
mutable Vector lh_min, lh_fac, gh_min, gh_fac;
// Tolerance to mark points found on the surface as CODE_INTERNAL
// or CODE_BORDER. This is needed because we cannot only use reference
// space coordinates to determine if a point is located inside the
// element or not.
mutable double surf_dist_tol;
mutable Array<unsigned int> loc_hash_offset;
mutable Vector loc_hash_min, loc_hash_fac;
} DEV;
// Helper function to setup and free gslib's crystal router.
void SetupCrystal(); // Called inside Setup and SetupSurf_base
void FreeCrystal(); // Called inside FreeData
/// Use GSLIB for communication and interpolation. Updates field_out on
/// host.
/// Use GSLIB for communication and interpolation
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
/// Uses GSLIB Crystal Router for communication followed by MFEM's
/// interpolation functions. Updates field_out on host.
/// interpolation functions
virtual void InterpolateGeneral(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering);
/** @brief Since GSLIB is designed to work with quads/hexes, we split every
* triangle/tet/prism/pyramid element into quads/hexes. */
/// Since GSLIB is designed to work with quads/hexes, we split every
/// triangle/tet/prism/pyramid element into quads/hexes.
virtual void SetupSplitMeshes();
/** @brief Setup integration points that will be used to interpolate the
* nodal location at points expected by GSLIB. */
/// Setup integration points that will be used to interpolate the nodal
/// location at points expected by GSLIB.
virtual void SetupIntegrationRuleForSplitMesh(Mesh *mesh,
IntegrationRule *irule,
int order);
/** @brief Build integration rules at the given @a order for each split mesh
* and store them in @a ir_out. Requires that \ref SetupSplitMeshes has
* already been called. */
virtual void SetupIntegrationRules(const int order,
Array<IntegrationRule *> &ir_out);
/** @brief Helper function that calls \ref SetupSplitMeshes and
* \ref SetupIntegrationRules. */
/// Helper function that calls \ref SetupSplitMeshes and
/// \ref SetupIntegrationRuleForSplitMesh.
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
/** @brief Get GridFunction value at the points expected by GSLIB.
* @param[in] gf_in Grid function to evaluate.
* @param[out] node_vals Output values.
* @param[in] ir_in If non-null, use these rules instead of #ir_split.
* @param[in] by_element If true, output has element-major layout
* [nel][vdim][ndofs]; otherwise component-major
* layout [vdim][total_pts]. */
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals,
const Array<IntegrationRule *> *ir_in = nullptr,
bool by_element = false) const;
/// Get GridFunction value at the points expected by GSLIB.
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
* simplices, find the original element number (that was split into
* micro quads/hexes) during the setup phase. */
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
/// find the original element number (that was split into micro quads/hexes)
/// during the setup phase.
virtual void MapRefPosAndElemIndices();
/// FindPoints locally on device for 3D.
// Device functions
// FindPoints locally on device for 3D.
void FindPointsLocal3(const Vector &point_pos, int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
Vector &gsl_dist_l, int npt);
/// FindPoints locally on device for 2D.
// FindPoints locally on device for 2D.
void FindPointsLocal2(const Vector &point_pos, int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
Vector &gsl_dist_l, int npt);
/// FindPoints locally on device for 3D surface elements.
void FindPointsSurfLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// FindPoints locally on device for 3D edge elements.
void FindPointsEdgeLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// FindPoints locally on device for 2D edge elements.
void FindPointsEdgeLocal2(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &gsl_code_dev_l,
Array<unsigned int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &gsl_dist_l,
int npt);
/// Interpolate on device for 3D.
// Interpolate on device for 3D.
void InterpolateLocal3(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1dsol);
/// Interpolate on device for 2D.
int nel, int dof1dsol);
// Interpolate on device for 2D.
void InterpolateLocal2(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1dsol);
int nel, int dof1dsol);
/// Interpolate on device for 1D.
void InterpolateLocal1(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp, int dof1dsol);
/// Prepare data for device execution for volume meshes.
// Prepare data for device functions.
void SetupDevice();
/** @brief Searches positions given in physical space by @a point_pos.
/** Searches positions given in physical space by @a point_pos.
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
void FindPointsOnDevice(const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES);
/** @brief Interpolation of field values at prescribed reference space
* positions.
* @param[in] field_in_evec E-vector of grid function to be interpolated.
* Assumed ordering is NDOFSxVDIMxNEL
* @param[in] nel Number of elements in the mesh.
* @param[in] ncomp Number of components in the field.
* @param[in] dof1dsol Number of degrees of freedom in each reference
* space direction.
* @param[in] ordering Ordering of the out field values: byNodes/byVDIM
*
* @param[out] field_out Interpolated values. For points that are not
* found the value is set to
* #default_interp_value. */
/** Interpolation of field values at prescribed reference space positions.
@param[in] field_in_evec E-vector of grid function to be interpolated.
Assumed ordering is NDOFSxVDIMxNEL
@param[in] nel Number of elements in the mesh.
@param[in] ncomp Number of components in the field.
@param[in] dof1dsol Number of degrees of freedom in each reference
space direction.
@param[in] ordering Ordering of the out field values: byNodes/byVDIM
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value. */
void InterpolateOnDevice(const Vector &field_in_evec, Vector &field_out,
const int nel, const int ncomp,
const int dof1dsol, const int ordering);
/** @brief Interpolation of field values at prescribed reference space
* positions for surface meshes. */
void InterpolateSurfBase(const Vector &field_in, Vector &field_out,
const int nel, const int ncomp,
const int dof1dsol, const int field_out_ordering);
/// Preprocess 2D surface mesh needed for FindPoints.
void findptsedge_setup_2(DEV_STRUCT &devs,
const double *const elx[2],
const unsigned n,
const unsigned int nel,
const unsigned m,
const double bbox_rel_size_inc,
const unsigned int local_hash_size,
const unsigned int global_hash_size,
const Vector *aabb_sz_inc);
/// Preprocess 3D surface mesh needed for FindPoints.
void findptssurf_setup_3(DEV_STRUCT &devs,
const double *const elx[3],
const unsigned n,
const unsigned int nel,
const unsigned m,
const double bbox_rel_size_inc,
const unsigned int local_hash_size,
const unsigned int global_hash_size,
const int rD,
const Vector *aabb_sz_inc);
/** @brief Shared implementation for the public surface-setup methods.
*
* @details Initializes the surface-search data structures, builds the
* split-element representation expected by gslib, and constructs the
* element bounding boxes used by the MFEM surface kernels.
*
* If @a aabb_sz_inc is null, the setup stores the default oriented
* bounding boxes and uses @a bbox_rel_size_inc as their relative size
* increase factor.
*
* If @a aabb_sz_inc is non-null, the setup stores axis-aligned bounding
* boxes only, applies the requested absolute AABB expansion in each
* physical direction, and adjusts the tolerance @a bdr_tol so points
* found in the expanded region are classified as border points.
*
* @param[in] m Input surface mesh.
* @param[in] bbox_rel_size_inc Relative size increase applied when
* expanding each element bounding box during
* setup.
* @param[in] aabb_sz_inc Optional total absolute AABB expansion
* applied to the stored axis-aligned
* bounding boxes after construction.
* @param[in] newt_tol Newton tolerance for the point-search
* kernels.
*/
void SetupSurf_Base(Mesh &m,
const double bbox_rel_size_inc,
const Vector *aabb_sz_inc,
const double newt_tol);
public:
/// Serial constructor
FindPointsGSLIB();
/// Serial constructor + setup with given Mesh (see \ref Setup)
FindPointsGSLIB(Mesh &mesh_in, const double bbox_rel_size_inc = 0.1,
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
@@ -388,7 +215,7 @@ public:
FindPointsGSLIB(MPI_Comm comm_);
/// Constructor + setup with given ParMesh (see \ref Setup)
FindPointsGSLIB(ParMesh &mesh_in, const double bbox_rel_size_inc = 0.1,
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
const double newt_tol = 1.0e-12,
const int npt_max = 256);
#endif
@@ -397,72 +224,25 @@ public:
FindPointsGSLIB(const FindPointsGSLIB&) = delete;
FindPointsGSLIB& operator=(const FindPointsGSLIB&) = delete;
/** @brief Preprocess the internal mesh in gslib.
@details Initializes the internal mesh in gslib, by sending the
positions of the Gauss-Lobatto nodes of the input Mesh object \p m.
/** Initializes the internal mesh in gslib, by sending the positions of the
Gauss-Lobatto nodes of the input Mesh object \p m.
Note: not tested with periodic (L2).
Note: the input mesh \p m must have Nodes set.
@param[in] m Input mesh.
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
when expanding each element bounding box.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for
simultaneous iteration. This alters
performance and memory footprint.
*/
void Setup(Mesh &m, const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
@param[in] m Input mesh.
@param[in] bb_t (Optional) Relative size of bounding box around
each element.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for simultaneous
iteration. This alters performance and
memory footprint.*/
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/// Preprocess the surface mesh to compute data for FindPoints.
void SetupSurf(Mesh &m,
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12);
/** @brief Preprocess the surface mesh to compute data for FindPoints using
* absolute AABB expansion.
*
* @details This method computes only axis-aligned bounding boxes and
* increases their total length by a user-specified amount in each
* physical direction. The absolute AABB expansion is applied
* symmetrically to the lower and upper bounds.
*
* The size of @a aabb_sz_inc determines how the expansion values are
* interpreted:
* - `1`: one expansion value used in every direction for every element
* - `NElements`: one expansion value per element, reused in x/y/z
* directions
* - `SpaceDim`: one expansion value per physical direction, reused for
* every element
* - `NElements*SpaceDim`: one expansion value per element and direction,
* ordered as `(dx1,dy1,dz1, ... dxN,dyN,dzN)`
*
* This method disables the oriented bounding-box precheck because the
* stored boxes are modified only in their axis-aligned representation.
*
* @param[in] m Input surface mesh.
* @param[in] aabb_sz_inc Total absolute AABB expansion applied in
* each physical direction to the stored
* axis-aligned bounding boxes.
* @param[in] newt_tol Newton tolerance for the point-search
* kernels.
*
* @note We disable the oriented bounding box check with this setup.
* @a bdr_tol is also adjusted so that all points in the AABBs can
* be found.
*/
void SetupSurfWithAABBExpansion(Mesh &m, const Vector &aabb_sz_inc,
const double newt_tol = 1.0e-12);
/** @brief Searches positions given in physical space by \p point_pos.
@details These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
/** Searches positions given in physical space by \p point_pos.
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
byVDim: (XYZ,XYZ,....XYZ) specified by \p point_pos_ordering.
This function populates the following member variables:
#gsl_code Return codes for each point: inside element (0),
element boundary (1), not found (2).
@@ -481,77 +261,40 @@ public:
#gsl_dist Distance between the sought and the found point
in physical space. */
void FindPoints(const Vector &point_pos,
int point_pos_ordering = Ordering::byNODES);
const int point_pos_ordering = Ordering::byNODES);
/// Convenience function when point positions are in a ParticleVector
void FindPoints(const ParticleVector &point_pos)
{
FindPoints(point_pos, point_pos.GetOrdering());
}
/** @brief Searches positions given in physical space by \p point_pos on
* surface mesh. */
void FindPointsSurf(const Vector &point_pos,
int point_pos_ordering = Ordering::byNODES);
/// Convenience function when point positions are in a ParticleVector
void FindPointsSurf(const ParticleVector &point_pos)
{
FindPointsSurf(point_pos, point_pos.GetOrdering());
}
/// Setup FindPoints and search positions
void FindPoints(Mesh &m, const Vector &point_pos,
const int point_pos_ordering = Ordering::byNODES,
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** @brief Interpolation of field values at prescribed reference space
* positions.
/** Interpolation of field values at prescribed reference space positions.
@param[in] field_in Function values that will be interpolated on the
reference positions. Note: it is assumed that
\p field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value.
The output ordering is determined from field_in.
@note: field_out is moved to device if field_in is on device. Otherwise,
field_out memory allocation is not changed.
*/
The output ordering is determined from field_in.*/
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
/// Interpolation of field values, with output ordering specification.
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
const int field_out_ordering);
/** @brief Same as Interpolate but for surface meshes */
virtual void InterpolateSurf(const GridFunction &field_in,
Vector &field_out);
/** @brief Same as Interpolate but for surface meshes with specified output
ordering */
virtual void InterpolateSurf(const GridFunction &field_in,
Vector &field_out,
const int field_out_ordering);
/** @brief Search positions and interpolate.
*
* @details The ordering (byNODES or byVDIM) of the output values in
* \p field_out corresponds to the ordering used in the input
* GridFunction \p field_in.
*/
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
the output values in \p field_out corresponds to the ordering used
in the input GridFunction \p field_in. */
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out,
int point_pos_ordering = Ordering::byNODES);
const int point_pos_ordering = Ordering::byNODES);
/// Search positions and interpolate with given point and output ordering.
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
Vector &field_out, const int point_pos_ordering,
const int field_out_ordering);
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
or byVDIM) of the output values in \p field_out corresponds to the
ordering used in the input GridFunction \p field_in. */
@@ -559,41 +302,32 @@ public:
const GridFunction &field_in, Vector &field_out,
const int point_pos_ordering = Ordering::byNODES);
/** @brief Average type to be used for L2 functions in-case a point is
* located at an element boundary where the function might be multi-valued.
*/
/// Average type to be used for L2 functions in-case a point is located at
/// an element boundary where the function might be multi-valued.
virtual void SetL2AvgType(AvgType avgtype_) { avgtype = avgtype_; }
/** @brief Set the default interpolation value for points that are not found in the mesh. */
/// Set the default interpolation value for points that are not found in the
/// mesh.
virtual void SetDefaultInterpolationValue(double interp_value_)
{
default_interp_value = interp_value_;
}
/** @brief Tolerance for detecting points outside the 'curvilinear' boundary.
*
* @details When using FindPoints, gslib may return points as found on the
* boundary even when they are slightly outside the domain. This tolerance
* is used to filter such points based on the distance^2 value and mark them
* as not found.
*
* @note When the SetupSurfWithAABBExpansion method is used for surface
* meshes, this tolerance is automatically computed based on the size of
* expanded AABBs. Using this method will override that computed tolerance.
* */
/// Set the tolerance for detecting points outside the 'curvilinear' boundary
/// that gslib may return as found on the boundary. Points found on boundary
/// with distance greater than @ bdr_tol are marked as not found.
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
{
bdr_tol = bdr_tol_;
}
/** @brief Enable/Disable use of CPU functions for GPU data if the gslib
* version is older. */
/// Enable/Disable use of CPU functions for GPU data if the gslib version
/// is older.
virtual void SetGPUtoCPUFallback(bool mode) { gpu_to_cpu_fallback = mode; }
/** @brief Cleans up memory allocated internally by gslib.
@details Note that in parallel, this must be called before MPI_Finalize,
as it calls MPI_Comm_free() for internal gslib communicators. FreeData is
/** Cleans up memory allocated internally by gslib.
Note that in parallel, this must be called before MPI_Finalize(), as it
calls MPI_Comm_free() for internal gslib communicators. FreeData is
also called by the class destructor and there are no memory leaks if the
destructor is called before MPI_Finalize(). If the destructor is called
after MPI_Finalize(), there will be an error because gslib will try to
@@ -601,8 +335,8 @@ public:
*/
virtual void FreeData();
/** @brief Return code for each point searched by FindPoints:
* inside element (0), element boundary (1), or not found (2). */
/// Return code for each point searched by FindPoints: inside element (0), on
/// element boundary (1), or not found (2).
virtual const Array<unsigned int> &GetCode() const { return gsl_code; }
/// Return element number for each point found by FindPoints.
virtual const Array<unsigned int> &GetElem() const { return gsl_mfem_elem; }
@@ -610,15 +344,15 @@ public:
virtual const Array<unsigned int> &GetProc() const { return gsl_proc; }
/// Return reference coordinates for each point found by FindPoints.
virtual const Vector &GetReferencePosition() const { return gsl_mfem_ref; }
/// Return distance between the sought and the found point in physical space.
/// Return distance between the sought and the found point in physical space,
/// for each point found by FindPoints.
virtual const Vector &GetDist() const { return gsl_dist; }
/** @brief Return element number for each point found by FindPoints
* corresponding to GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with
* simplices. */
/// Return element number for each point found by FindPoints corresponding to
/// GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with simplices.
virtual const Array<unsigned int> &GetGSLIBElem() const { return gsl_elem; }
/** @brief Return reference coordinates in [-1,1] (internal range in GSLIB)
* for each point found by FindPoints. */
/// Return reference coordinates in [-1,1] (internal range in GSLIB) for each
/// point found by FindPoints.
virtual const Vector &GetGSLIBReferencePosition() const { return gsl_ref; }
/// Get array of indices of not-found points.
@@ -661,7 +395,7 @@ public:
/// Return the axis-aligned bounding boxes (AABB) computed during \ref Setup.
/// The size of the returned vector is (nel x nverts x dim), where nel is the
/// number of elements (after splitting for simplicies), nverts is number of
/// number of elements (after splitting for simplcies), nverts is number of
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
@@ -675,18 +409,6 @@ public:
/// \p obbV, a vector of size (nel x nverts x dim) .
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
Vector &obbV) const;
/** @brief Return the bounding boxes as a mesh on rank 0.
*
* @param[in] type Bounding-box type: 0 - AABB, 1 - OBB.
*
* @return On rank 0, returns a newly allocated mesh containing the
* bounding boxes. The caller owns the returned pointer and is responsible
* for deleting it. On other ranks, returns nullptr.
*/
Mesh *GetBoundingBoxMesh(int type);
virtual const Vector &GetGLLMesh() const { return gsl_mesh; }
};
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
@@ -715,28 +437,25 @@ public:
Note: not tested with periodic meshes (L2).
Note: the input mesh \p m must have Nodes set.
@param[in] m Input mesh.
@param[in] meshid A unique # for each overlapping mesh.
This id is used to make sure that points
being searched are not looked for in the
mesh that they belong to.
@param[in] gfmax (Optional) GridFunction in H1 that is used
as a discriminator when one point is
located in multiple meshes. The mesh that
maximizes gfmax is chosen. For example,
using the distance field based on the
overlapping boundaries is helpful for
convergence during Schwarz iterations.
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
when expanding each element bounding box.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for
simultaneous iteration. This alters
performance and memory footprint.*/
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = nullptr,
const double bbox_rel_size_inc = 0.1,
const double newt_tol = 1.0e-12,
@param[in] m Input mesh.
@param[in] meshid A unique # for each overlapping mesh. This id is
used to make sure that points being searched are not
looked for in the mesh that they belong to.
@param[in] gfmax (Optional) GridFunction in H1 that is used as a
discriminator when one point is located in multiple
meshes. The mesh that maximizes gfmax is chosen.
For example, using the distance field based on the
overlapping boundaries is helpful for convergence
during Schwarz iterations.
@param[in] bb_t (Optional) Relative size of bounding box around
each element.
@param[in] newt_tol (Optional) Newton tolerance for the gslib
search methods.
@param[in] npt_max (Optional) Number of points for simultaneous
iteration. This alters performance and
memory footprint.*/
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = NULL,
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** Searches positions given in physical space by \p point_pos. All output
@@ -792,7 +511,7 @@ class GSOPGSLIB
protected:
struct gslib::crystal *cr; // gslib's internal data
struct gslib::comm *gsl_comm; // gslib's internal data
struct gslib::gs_data *gsl_data = nullptr;
struct gslib::gs_data *gsl_data = NULL;
int num_ids;
public:
@@ -817,116 +536,6 @@ public:
void GS(Vector &senddata, GSOp op);
};
#if defined(MFEM_USE_MPI)
/** \brief Class to map a point in physical space to candidate ranks.
*
* This class builds a Cartesian-aligned tensor grid that covers the entire
* domain and precomputes which ranks have elements intersecting each
* grid cell. Given a point in physical space, the grid cell containing
* the point is determined, and the list of candidate ranks whose
* elements intersect that cell is returned. This yields a fast, conservative
* point-to-rank candidate query. This is used internally by FindPointsGSLIB
* to speed up point searches in parallel.
*
* See Mittal et al., "General Field Evaluation in High-Order Meshes on GPUs".
* (2025). Computers & Fluids. for technical details.
*
*/
class GlobalBBoxTensorGridMap
{
private:
struct gslib::crystal *cr = nullptr; // gslib's internal data
struct gslib::comm *gsl_comm = nullptr; // gslib's internal data
int sdim, n_local_cells, num_procs;
Array<int> gmap_n;
Vector gmap_bnd_min, gmap_bnd_max;
Vector gmap_fac;
Array<int> ggrid_map;
void SetupCrystal(const MPI_Comm &comm);
public:
/// Constructor for a given mesh and number of tensor grid divisions
GlobalBBoxTensorGridMap(ParMesh &pmesh, int nx);
/** @brief Constructor for given element bounds and spatial dimension.
*
* @details This constructor must be called collectively on \a comm.
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
*
* Assumes elmin, elmax Ordering::byNodes:
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
*
* When by_max_size=false, n gives the number of tensor-grid divisions in
* each direction. When by_max_size=true, n is a per-rank size hint used to
* derive a uniform global resolution. The communicator-wide sum of n is
* converted to nx = ceil(pow(sum(n), 1./sdim)) in each direction, so n is
* not a hard cap on ggrid_map.Size().
*/
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
Vector &elmax, int nel, int sdim, int n,
bool by_max_size);
/** @brief Constructor for given element bounds, spatial dimension, and
* tensor-grid divisions in each direction.
*
* @details This constructor must be called collectively on \a comm.
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
* Requires nx.Size() == sdim and positive entries in nx.
*
* Assumes elmin, elmax Ordering::byNodes:
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
*/
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
Vector &elmax, int nel, int sdim, Array<int> &nx);
~GlobalBBoxTensorGridMap();
/** @brief Get list of procs corresponding to the list of points.
*
* @details This method must be called collectively on the communicator
* used to construct the map. The input points can be ordered byNodes:
* (XXX...,YYY...,ZZZ) or byVDIM: (XYZ,XYZ,...), as specified by
* \a ordering.
*
* The output map contains one entry for each input point, keyed by the
* point's local index in \a xyz. Points with no candidate ranks, including
* points outside the global bounding box, have an empty list of candidate
* ranks.
*/
void MapPointsToProcs(Vector &xyz, int ordering,
std::map<int, std::vector<int>> &pt_to_procs) const;
// Some getters
const Array<int> &GetGridMap() const { return ggrid_map; }
const Vector &GetGridFac() const { return gmap_fac; }
const Vector &GetGridMin() const { return gmap_bnd_min; }
const Vector &GetGridMax() const { return gmap_bnd_max; }
const Array<int> &GetGridN() const { return gmap_n; }
private:
/// Setup the map given element bounds and number of tensor grid divisions.
void Setup(const MPI_Comm &comm, Vector &elmin, Vector &elmax,
int nel, Array<int> &nx);
/// Get global hash cell index for a given point.
int GetGlobalGridCellFromPoint(Vector &xyz) const;
/** @brief Get owning proc and local index on that proc for given global
* grid cell index. */
void GlobalGridCellToProcAndLocalIndex(int i, int &proc, int &idx) const;
/// Map a point to proc and local index of the corresponding grid cell
void GetProcAndLocalIndexFromPoint(Vector &xyz, int &proc, int &idx) const;
/// Given local cell index, return list of procs saved in the map
Array<int> MapCellToProcs(int l_idx) const;
};
#endif // MFEM_USE_MPI
} // namespace mfem
#endif // MFEM_USE_GSLIB
+177 -73
View File
@@ -11,7 +11,7 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#include "../../linalg/kernels.hpp"
#ifdef MFEM_USE_GSLIB
@@ -27,6 +27,8 @@
#pragma GCC diagnostic pop
#endif
#include <climits>
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
@@ -52,14 +54,127 @@ struct findptsElementGPT_t
double x[DIM], jac[DIM * DIM], hes[4];
};
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<DIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_first_der;
using gslib::lag_eval_second_der;
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[DIM], A[DIM * DIM];
dbl_range_t x[DIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[DIM];
double fac[DIM];
unsigned int *offset;
int max;
};
// Eval the ith Lagrange interpolant and its first derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x - z[j]);
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN+i] = 2.0 * lCoeff[i] * u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2 * (x - z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN+i] = 2.0 * lCoeff[i] * u1;
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
}
// Axis-aligned bounding box test.
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
const double x[2])
{
double test = 1;
for (int d = 0; d < 2; ++d)
{
double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
test = test < 0 ? test : b_d;
}
return test;
}
// Axis-aligned bounding box test followed by oriented bounding-box test.
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
const double x[2])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0)
{
return bxyz;
}
else
{
double dxyz[2];
for (int d = 0; d < 2; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1;
for (int d = 0; d < 2; ++d)
{
double rst = 0;
for (int e = 0; e < 2; ++e)
{
rst += b->A[d * 2 + e] * dxyz[e];
}
double brst = (rst + 1) * (1 - rst);
test = test < 0 ? test : brst;
}
return test;
}
}
// Element index corresponding to hash mesh that the point is located in.
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[2])
{
const int n = p->hash_n;
int sum = 0;
for (int d = 2 - 1; d >= 0; --d)
{
sum *= n;
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
}
return sum;
}
/*Solve Ax=y. A is row-major */
static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
@@ -70,6 +185,12 @@ static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
x[1] = idet*(A[0]*y[1] - A[2]*y[0]);
}
/* L2 norm squared. */
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
{
return x[0] * x[0] + x[1] * x[1];
}
/* the bit structure of flags is CSSRR
the C bit --- 1<<4 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
@@ -231,7 +352,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *res,
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2<2>(resid);
const double dist2 = l2norm2(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d = 0; d < 2; ++d)
@@ -441,7 +562,7 @@ newton_area_fin:
int f = flags >> (2 * dd) & 3u;
res->r[dd] = f == 0 ? r0[dd] + dr[dd] : (f == 1 ? -1 : 1);
}
res->flags = flags | ((p->flags & FLAG_MASK) << 5);
res->flags = flags | (p->flags << 5);
}
// Full Newton solve on the face. One of r/s/t is constrained.
@@ -514,8 +635,7 @@ newton_edge_fin:
res->r[de] = nr;
res->r[dn]=p->r[dn];
res->dist2p = -v;
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 5);
#undef EVAL
res->flags = flags | new_flags | (p->flags << 5);
}
// Find closest mesh node to the sought point.
@@ -574,26 +694,27 @@ static MFEM_HOST_DEVICE double tensor_ig2_j(double *g_partials,
}
template<int T_D1D = 0>
static void FindPointsLocal2DKernel(const int npt,
const double tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
static void FindPointsLocal2D_Kernel(const int npt,
const double tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
#define MAX_CONST(a, b) (((a) > (b)) ? (a) : (b))
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NE = D1D*D1D;
@@ -608,7 +729,7 @@ static void FindPointsLocal2DKernel(const int npt,
// 3D1D for seed, 10D1D+6 for area, 3D1D+9 for edge
constexpr int size1 = 10*MD1 + 6;
constexpr int size2 = MD1*4; // edge constraints
constexpr int size3 = MD1*MD1*DIM; // local element coordinates
constexpr int size3 = MD1*MD1*MD1*DIM; // local element coordinates
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED findptsElementPoint_t el_pts[2];
@@ -1041,9 +1162,9 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
auto pgslm = gsl_mesh.Read();
auto pwt = DEV.wtend.Read();
auto pbb = DEV.bb.Read();
auto plhm = DEV.lh_min.Read();
auto plhf = DEV.lh_fac.Read();
auto plho = DEV.lh_offset.ReadWrite();
auto plhm = DEV.loc_hash_min.Read();
auto plhf = DEV.loc_hash_fac.Read();
auto plho = DEV.loc_hash_offset.ReadWrite();
auto pcode = code.Write();
auto pelem = elem.Write();
auto pref = ref.Write();
@@ -1054,49 +1175,32 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
switch (DEV.dof1d)
{
case 2:
FindPointsLocal2DKernel<2>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
return FindPointsLocal2D_Kernel<2>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 3:
FindPointsLocal2DKernel<3>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
return FindPointsLocal2D_Kernel<3>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 4:
FindPointsLocal2DKernel<4>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
return FindPointsLocal2D_Kernel<4>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
case 5:
FindPointsLocal2DKernel<5>(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
return FindPointsLocal2D_Kernel<5>(
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
pgll1d, plc);
default:
FindPointsLocal2DKernel(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb, DEV.h_nx,
plhm, plhf, plho, pcode, pelem,
pref, pdist, pgll1d, plc, DEV.dof1d);
}
}
#undef DIM2
#undef DIM
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
+166 -38
View File
@@ -11,7 +11,9 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#include "../../linalg/kernels.hpp"
#include <climits>
#ifdef MFEM_USE_GSLIB
@@ -57,15 +59,128 @@ struct findptsElemPt
double x[DIM], jac[DIM * DIM], hes[18];
};
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<DIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_first_der;
using gslib::lag_eval_second_der;
using gslib::lin_solve_sym_2;
struct dbl_range_t
{
double min, max;
};
struct obbox_t
{
double c0[DIM], A[DIM * DIM];
dbl_range_t x[DIM];
};
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[DIM];
double fac[DIM];
unsigned int *offset;
// int max;
};
// Eval the ith Lagrange interpolant and its first derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2*(x-z[j]);
u1 = d_j*u1+u0;
u0 = d_j*u0;
}
}
p0[i] = lCoeff[i]*u0;
p0[pN+i] = 2.0*lCoeff[i]*u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
double d_j = 2*(x-z[j]);
u2 = d_j*u2+u1;
u1 = d_j*u1+u0;
u0 = d_j*u0;
}
}
p0[i] = lCoeff[i]*u0;
p0[pN+i] = 2.0*lCoeff[i]*u1;
p0[2*pN+i] = 8.0*lCoeff[i]*u2;
}
// Axis-aligned bounding box test.
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
const double x[3])
{
double b_d;
for (int d = 0; d < 3; ++d)
{
b_d = (x[d]-b->x[d].min)*(b->x[d].max-x[d]);
if (b_d < 0) { return b_d; }
}
return b_d;
}
// Axis-aligned bounding box test followed by oriented bounding-box test.
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
const double x[3])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0)
{
return bxyz;
}
else
{
double dxyz[3];
for (int d = 0; d < 3; ++d)
{
dxyz[d] = x[d]-b->c0[d];
}
double test = 1;
for (int d = 0; d < 3; ++d)
{
double rst = 0;
for (int e = 0; e < 3; ++e)
{
rst += b->A[d*3+e]*dxyz[e];
}
double brst = (rst+1)*(1-rst);
test = test < 0 ? test : brst;
}
return test;
}
}
// Element index corresponding to hash mesh that the point is located in.
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
const double x[3])
{
const int n = p->hash_n;
int sum = 0;
for (int d = 3-1; d >= 0; --d)
{
sum *= n;
int i = (int)floor((x[d]-p->bnd[d].min)*p->fac[d]);
sum += i < 0 ? 0 : (n-1 < i ? n-1 : i);
}
return sum;
}
// Solve Ax=y. A is row-major.
static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
@@ -84,6 +199,22 @@ static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
x[2] = idet*(inv6*y[0]+inv7*y[1]+inv8*y[2]);
}
// Solve Ax=y. A is a symmetric 2x2 matrix.
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
const double A[3],
const double y[2])
{
const double idet = 1 / (A[0]*A[2]-A[1]*A[1]);
x[0] = idet*(A[2]*y[0]-A[1]*y[1]);
x[1] = idet*(A[0]*y[1]-A[1]*y[0]);
}
// L2 norm.
static MFEM_HOST_DEVICE inline double l2norm2(const double x[3])
{
return x[0]*x[0]+x[1]*x[1]+x[2]*x[2];
}
/* the bit structure of flags is CTTSSRR
the C bit --- 1<<6 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
@@ -328,7 +459,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsPt *res,
const findptsPt *p,
const double tol)
{
const double dist2 = l2norm2<3>(resid);
const double dist2 = l2norm2(resid);
const double decr = p->dist2-dist2;
const double pred = p->dist2p;
for (int d = 0; d < 3; ++d)
@@ -575,7 +706,7 @@ newton_vol_fin:
int f = flags >> (2*dd) & 3u;
res->r[dd] = f == 0 ? r0[dd]+dr[dd] : (f == 1 ? -1 : 1);
}
res->flags = flags | ((p->flags & FLAG_MASK) << 7);
res->flags = flags | (p->flags << 7);
}
// Full Newton solve on the face. One of r/s/t is constrained.
@@ -758,7 +889,7 @@ newton_face_fin:
res->r[dn] = p->r[dn];
res->r[d1] = r[0];
res->r[d2] = r[1];
res->flags = new_flags | ((p->flags & FLAG_MASK) << 7);
res->flags = new_flags | (p->flags << 7);
}
// Full Newton solve on the edge. Two of r/s/t are constrained.
@@ -842,8 +973,7 @@ newton_edge_fin:
res->r[dn1] = p->r[dn1];
res->r[dn2] = p->r[dn2];
res->dist2p = -v;
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 7);
#undef EVAL
res->flags = flags | new_flags | (p->flags << 7);
}
// Find closest mesh node to the sought point.
@@ -1122,6 +1252,7 @@ static void FindPointsLocal3DKernel(const int npt,
case 0: // findpt_vol
{
double *wtr = r_workspace_ptr;
double *resid = wtr+6*D1D;
double *jac = resid+3;
double *resid_temp = jac+9;
@@ -1372,7 +1503,7 @@ static void FindPointsLocal3DKernel(const int npt,
// Hes_T is transposed version (i.e. in col major)
// n1*[2, 1, 1, 0, 0]
// j==1 => wt_j = wt+n1
double *wt_j = wt+D1D*(2 - (row+1)/2);
double *wt_j = wt+D1D*(2-(row+1) / 2);
const double *x = e_x[row+1][d];
hes_T[j] = 0.0;
for (int k = 0; k < D1D; ++k)
@@ -1391,6 +1522,7 @@ static void FindPointsLocal3DKernel(const int npt,
hes[j] += resid[d]*hes_T[j*3+d];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(l,x,1)
@@ -1648,7 +1780,6 @@ static void FindPointsLocal3DKernel(const int npt,
} //findpts_local
} //elp
});
#undef MAXC
}
void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
@@ -1665,9 +1796,9 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
auto pgslm = gsl_mesh.Read();
auto pwt = DEV.wtend.Read();
auto pbb = DEV.bb.Read();
auto plhm = DEV.lh_min.Read();
auto plhf = DEV.lh_fac.Read();
auto plho = DEV.lh_offset.ReadWrite();
auto plhm = DEV.loc_hash_min.Read();
auto plhf = DEV.loc_hash_fac.Read();
auto plho = DEV.loc_hash_offset.ReadWrite();
auto pcode = code.Write();
auto pelem = elem.Write();
auto pref = ref.Write();
@@ -1678,36 +1809,33 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
{
case 2:
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 3:
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 4:
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
case 5:
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
pgslm, NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc);
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
plc);
break;
default:
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist, pgll1d, plc,
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, DEV.h_nx, plhm, plhf,
plho, pcode, pelem, pref, pdist, pgll1d, plc,
DEV.dof1d);
break;
}
}
#undef pMax
-656
View File
@@ -1,656 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
#define sDIM 2
#define sDIM2 4
#define rDIM 1
struct findptsElementPoint_t
{
double x[sDIM], r, oldr, dist2, dist2p, tr;
int flags;
};
struct findptsElementGEdge_t
{
double *x[sDIM];
};
struct findptsElementGPT_t
{
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
};
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<sDIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
using gslib::AABB_test;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_second_der;
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
1 = 01b if r is constrained at -1, i.e., rmin
2 = 10b if r is constrained at +1, i.e., rmax
*/
#define CONVERGED_FLAG (1u<<2)
#define FLAG_MASK 0x07u // = 111b
/* returns 1 if r direction (the only free direction in 2D) is constrained.
returns 1 if either 1st or 2nd bit of flags is set.
*/
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
{
return ((flags | flags>>1) & 1u);
}
/* pi=0, r=-1; pi=1, r=+1 */
static MFEM_HOST_DEVICE inline int point_index(const int x)
{
return ((x>>1) & 1u);
}
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
const double resid[2],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2<2>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
out_pt->x[0] = p->x[0];
out_pt->x[1] = p->x[1];
out_pt->oldr = p->r;
out_pt->dist2 = dist2;
if (decr >= 0.01*pred)
{
if (decr >= 0.9*pred) // very good iteration
{
out_pt->tr = p->tr*2;
}
else // somewhat good iteration
{
out_pt->tr = p->tr;
}
return false;
}
else
{
/* reject step; note: the point will pass through this routine
again, and we set things up here so it gets classed as a
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out_pt->tr = v0/4.0;
out_pt->dist2 = p->dist2;
out_pt->r = p->oldr;
out_pt->flags = p->flags>>3;
out_pt->dist2p = -HUGE_VAL;
if (pred < dist2*tol)
{
out_pt->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
out_pt,
const double jac[2],
const double rhess,
const double resid[2],
int flags,
const findptsElementPoint_t *const p,
const double tol )
{
const double tr = p->tr;
const double A = jac[0] * jac[0] + jac[1] * jac[1] -
rhess; // A = J^T J - resid_d H_d
const double y = jac[0]*resid[0] + jac[1]*resid[1]; // y = J^T resid
const double oldr = p->r;
double dr, newr, tdr, tnewr, v, tv;
int new_flags=0, tnew_flags=0;
#define EVAL(dr) ( (dr*A - 2*y) * dr )
if (A>0)
{
dr = y/A;
if (fabs(dr)<tol)
{
dr=0.0;
newr = oldr;
}
else
{
newr = oldr+dr;
}
if (fabs(dr)<tr && fabs(newr)<1)
{
v = EVAL(dr);
goto newton_edge_fin;
}
}
if ((newr=oldr-tr) > -1)
{
dr = -tr;
}
else
{
newr = -1, dr = -1-oldr, new_flags = flags|1u;
}
v = EVAL(dr);
if ((tnewr=oldr+tr) < 1)
{
tdr = tr;
}
else
{
tnewr = 1, tdr = 1-oldr, tnew_flags = flags|2u;
}
tv = EVAL(tdr);
if (tv<v)
{
newr = tnewr, dr = tdr, v = tv, new_flags = tnew_flags;
}
#undef EVAL
newton_edge_fin:
// check convergence by testing if change in r is less than tol
if (fabs(dr)<tol)
{
new_flags |= CONVERGED_FLAG;
}
out_pt->r = newr;
out_pt->dist2p = -v;
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
}
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
const double x[sDIM],
const double *z,
double *dist2,
double *r,
const int ir,
const int pN )
{
double dx[sDIM];
for (int d=0; d<sDIM; ++d)
{
dx[d] = x[d] - elx[d][ir];
}
dist2[ir] = HUGE_VAL;
const double dist2_rs = l2norm2(dx);
if (dist2[ir]>dist2_rs)
{
dist2[ir] = dist2_rs;
r[ir] = z[ir];
}
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal2DKernel( const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const bool obb_check,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0 )
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NEL = nel*D1D;
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
const int nThreads = D1D*sDIM;
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
{
// 2D1D for seed, 3D1D + 7 for edge
constexpr int size1 = 3*MD1 + 7;
// edge coordinates = D1D*2
constexpr int size2 = 2*MD1;
// local element coordinates in shared memory
constexpr int size3 = MD1*sDIM;
MFEM_SHARED findptsElementPoint_t el_pts[2];
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED double constraint_workspace[size2];
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
double *r_workspace_ptr = r_workspace;
findptsElementPoint_t *fpt, *tmp;
fpt = el_pts + 0;
tmp = el_pts + 1;
// x and y coord index within point_pos for point i
int id_x = point_pos_ordering == 0 ? i : i*sDIM;
int id_y = point_pos_ordering == 0 ? i+npt : i*sDIM+1;
double x_i[2] = {x[id_x], x[id_y]};
unsigned int *code_i = code_base + i;
double *dist2_i = dist2_base + i;
//---------------- map_points_to_els --------------------
findptsLocalHashData_t hash;
for (int d=0; d<sDIM; ++d)
{
hash.bnd[d].min = hashMin[d];
hash.fac[d] = hashFac[d];
}
hash.hash_n = hash_n;
hash.offset = hashOffset;
const int hi = hash_index(&hash, x_i);
const unsigned int *elp = hash.offset + hash.offset[hi];
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
*code_i = CODE_NOT_FOUND;
*dist2_i = HUGE_VAL;
for (; elp!=ele; ++elp)
{
const unsigned int el = *elp;
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
bool pass_bb = true;
obbox_t box;
if (obb_check)
{
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
pass_bb = (bbox_test(&box, x_i) >= 0);
}
else
{
for (int d = 0; d < sDIM; ++d)
{
box.x[d].min = boxinfo[n_box_ents*el + d];
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
}
pass_bb = (AABB_test(&box, x_i) >= 0);
}
if (pass_bb)
{
//------------ findpts_local ------------------
{
if (MD1 <= 6)
{
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
{
const int qp = j % D1D;
const int d = j / D1D;
elem_coords[qp + d*D1D] =
xElemCoord[qp + el*D1D + d*p_NEL];
}
MFEM_SYNC_THREAD;
}
const double *elx[sDIM];
for (int d=0; d<sDIM; d++)
{
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
xElemCoord + d*p_NEL + el*D1D;
}
MFEM_SYNC_THREAD;
//// findpts_el ////
{
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
fpt->dist2p = 0;
fpt->tr = 1;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
fpt->x[j] = x_i[j];
}
MFEM_SYNC_THREAD;
{
double *dist2_temp = r_workspace_ptr;
double *r_temp = dist2_temp + D1D;
MFEM_FOREACH_THREAD(j,x,D1D)
{
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
for (int ir=0; ir<D1D; ++ir)
{
if (dist2_temp[ir]<fpt->dist2)
{
fpt->dist2 = dist2_temp[ir];
fpt->r = r_temp[ir];
}
}
}
MFEM_SYNC_THREAD;
} //seed done
// Initialize tmp struct with fpt values before starting Newton iterations
MFEM_FOREACH_THREAD(j,x,1)
{
tmp->dist2 = HUGE_VAL;
tmp->dist2p = 0;
tmp->tr = 1;
tmp->flags = 0;
tmp->r = fpt->r;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
tmp->x[j] = fpt->x[j];
}
MFEM_SYNC_THREAD;
for (int step=0; step<50; step++)
{
int nc = num_constrained(tmp->flags & FLAG_MASK);
switch (nc)
{
case 0:
{
double *wt = r_workspace_ptr;
double *resid = wt + 3*D1D;
double *jac = resid + sDIM;
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
}
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d][j] = elx[d][j];
}
}
MFEM_SYNC_THREAD;
// compute basis function info upto 2nd derivative
MFEM_FOREACH_THREAD(j,x,D1D)
{
lag_eval_second_der(wt, tmp->r, j, gll1D,
lagcoeff, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,sDIM)
{
resid[j] = tmp->x[j];
jac[j] = 0.0;
hess[j] = 0.0;
for (int k=0; k<D1D; ++k)
{
resid[j] -= wt[ k]*edge.x[j][k];
jac[j] += wt[D1D+k]*edge.x[j][k];
hess[j] += wt[2*D1D+k]*edge.x[j][k];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
hess[2] = resid[0]*hess[0] + resid[1]*hess[1];
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
if (!reject_prior_step_q(fpt, resid, tmp, tol))
{
newton_edge(fpt, jac, hess[2], resid,
tmp->flags & FLAG_MASK, tmp, tol);
}
}
MFEM_SYNC_THREAD;
break;
}
case 1: // r is constrained to either -1 or 1
{
MFEM_FOREACH_THREAD(j,x,1)
{
const int pi = point_index(tmp->flags &
FLAG_MASK);
const double *wt = wtend + pi*3*D1D;
findptsElementGPT_t gpt;
for (int d=0; d<sDIM; ++d)
{
gpt.x[d] = elx[d][pi*(D1D-1)];
gpt.jac[d] = 0.0;
gpt.hes[d] = 0.0;
for (int k=0; k<D1D; ++k)
{
gpt.jac[d] += wt[D1D +k]*elx[d][k];
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
}
}
const double *const pt_x = gpt.x;
const double *const jac = gpt.jac;
const double *const hes = gpt.hes;
double resid[sDIM], steep, sr;
resid[0] = fpt->x[0] - pt_x[0];
resid[1] = fpt->x[1] - pt_x[1];
steep = jac[0]*resid[0] + jac[1]*resid[1];
sr = steep*tmp->r;
if ( !reject_prior_step_q(fpt, resid, tmp, tol) )
{
if (sr<0)
{
const double rhess = resid[0]*hes[0] +
resid[1]*hes[1];
newton_edge(fpt, jac, rhess,
resid, 0, tmp, tol);
}
else // sr==0
{
fpt->r = tmp->r;
fpt->dist2p = 0;
fpt->flags = tmp->flags | CONVERGED_FLAG;
}
}
}
MFEM_SYNC_THREAD;
break;
} // case 1
} //switch
if (fpt->flags & CONVERGED_FLAG)
{
break;
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
*tmp = *fpt;
}
MFEM_SYNC_THREAD;
} //for int step<50
} //findpts_el
bool converged_internal =
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
(fpt->dist2<dist2tol);
if (*code_i == CODE_NOT_FOUND || converged_internal ||
fpt->dist2 < *dist2_i)
{
MFEM_FOREACH_THREAD(j,x,1)
{
*(el_base+i) = el;
*code_i = converged_internal ? CODE_INTERNAL : CODE_BORDER;
*dist2_i = fpt->dist2;
*(r_base+i) = fpt->r;
}
MFEM_SYNC_THREAD;
if (converged_internal)
{
break;
}
}
} //findpts_local
} //obbox_test
} //elp
});
}
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt )
{
if (npt==0)
{
return;
}
MFEM_VERIFY(dim==1 && spacedim==2,"Function for 2D edges only");
bool use_dev = point_pos.UseDevice();
auto pp = point_pos.Read(use_dev);
auto pgslm = gsl_mesh.Read(use_dev);
auto pwt = DEV.wtend.Read(use_dev);
auto pbb = DEV.bb.Read(use_dev);
auto plhm = DEV.lh_min.Read(use_dev);
auto plhf = DEV.lh_fac.Read(use_dev);
auto plho = DEV.lh_offset.ReadWrite(use_dev);
auto pcode = code.Write(use_dev);
auto pelem = elem.Write(use_dev);
auto pref = ref.Write(use_dev);
auto pdist = dist.Write(use_dev);
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
const bool obb_chk = obb_check;
switch (DEV.dof1d)
{
case 2:
FindPointsEdgeLocal2DKernel<2>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
FindPointsEdgeLocal2DKernel<3>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
FindPointsEdgeLocal2DKernel<4>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
FindPointsEdgeLocal2DKernel(npt, DEV.newt_tol, dist2tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
#undef sDIM
#undef rDIM
#undef sDIM2
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt ) {} ;
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
-661
View File
@@ -1,661 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
#define sDIM 3
#define rDIM 1
#define sDIM2 (sDIM*sDIM)
#define rDIM2 (rDIM*rDIM)
struct findptsElementPoint_t
{
double x[sDIM], r, oldr, dist2, dist2p, tr;
int flags;
};
struct findptsElementGEdge_t
{
double *x[sDIM], *dxdn[sDIM], *d2xdn[sDIM];
};
struct findptsElementGPT_t
{
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
};
using dbl_range_t = gslib::dbl_range_t;
using obbox_t = gslib::obbox_t<sDIM>;
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
using gslib::AABB_test;
using gslib::bbox_test;
using gslib::hash_index;
using gslib::l2norm2;
using gslib::lag_eval_second_der;
/* the bit structure of flags is CRR
the C bit --- 1<<2 --- is set when the point is converged
RR is 0 = 00b if r is unconstrained,
1 = 01b if r is constrained at -1, i.e., rmin
2 = 10b if r is constrained at +1, i.e., rmax
*/
#define CONVERGED_FLAG (1u<<2)
#define FLAG_MASK 0x07u
/* returns the number of constrained reference coordinates, max 1
*/
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
{
return ((flags | flags>>1) & 1u);
}
static MFEM_HOST_DEVICE inline int point_index(const int x)
{
return ((x>>1)&1u);
}
/* check reduction in objective against prediction, and adjust
trust region radius (p->tr) accordingly;
may reject the prior step, returning 1; otherwise returns 0
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
const double resid[3],
const findptsElementPoint_t *p,
const double tol)
{
const double dist2 = l2norm2<sDIM>(resid);
const double decr = p->dist2 - dist2;
const double pred = p->dist2p;
for (int d=0; d<sDIM; ++d)
{
out_pt->x[d] = p->x[d];
}
out_pt->oldr = p->r;
out_pt->dist2 = dist2;
if (decr>=0.01*pred)
{
if (decr>=0.9*pred) // very good iteration
{
out_pt->tr = 2*p->tr;
}
else // good iteration
{
out_pt->tr = p->tr;
}
return false;
}
else // if the iteration in not good
{
/* reject step; note: the point will pass through this routine
again, and we set things up here so it gets classed as a
"very good iteration" --- this doubles the trust radius,
which is why we divide by 4 below */
double v0 = fabs(p->r - p->oldr);
out_pt->tr = v0/4.0;
out_pt->dist2 = p->dist2;
out_pt->r = p->oldr;
out_pt->flags = p->flags>>3;
out_pt->dist2p = -HUGE_VAL;
if (pred<dist2*tol)
{
out_pt->flags |= CONVERGED_FLAG;
}
return true;
}
}
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
out_pt,
const double jac[sDIM*rDIM],
const double rhes,
const double resid[sDIM],
int flags,
const findptsElementPoint_t *const p,
const double tol)
{
const double tr = p->tr;
/* A = J^T J - resid_d H_d */
const double A = jac[0]*jac[0]+ jac[1] * jac[1] + jac[2] * jac[2]
- rhes;
/* y = J^T r */
const double y = jac[0]*resid[0] + jac[1]*resid[1] + jac[0+2]*resid[2];
const double oldr = p->r;
double dr, nr, tdr, tnr;
double v, tv;
int new_flags = 0, tnew_flags = 0;
#define EVAL(dr) (dr*A - 2*y)*dr
/* if A is not SPD, quadratic model has no minimum */
if (A>0)
{
dr = y/A;
if (fabs(dr)<tol)
{
dr=0.0;
nr = oldr;
}
else
{
nr = oldr+dr;
}
if ( fabs(dr)<tr && fabs(nr)<1 )
{
v = EVAL(dr);
goto newton_edge_fin;
}
}
if ( (nr=oldr-tr)>-1 )
{
dr = -tr;
}
else
{
nr = -1, dr = -1-oldr, new_flags = flags | 1u;
}
v = EVAL(dr);
if ( (tnr = oldr+tr)<1 )
{
tdr = tr;
}
else
{
tnr = 1, tdr = 1-oldr, tnew_flags = flags | 2u;
}
tv = EVAL(tdr);
if (tv<v)
{
nr = tnr, dr = tdr, v = tv, new_flags = tnew_flags;
}
newton_edge_fin:
/* check convergence */
if ( fabs(dr)<tol )
{
new_flags |= CONVERGED_FLAG;
}
out_pt->r = nr;
out_pt->dist2p = -v;
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
#undef EVAL
}
static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
const double x[sDIM],
const double *z,
double *dist2,
double *r,
const int ir,
const int pN)
{
if (ir>=pN)
{
return;
}
double dx[sDIM];
for (int d=0; d<sDIM; ++d)
{
dx[d] = x[d] - elx[d][ir];
}
dist2[ir] = l2norm2(dx);
r[ir] = z[ir];
}
template<int T_D1D = 0>
static void FindPointsEdgeLocal3DKernel(const int npt,
const double tol,
const double dist2tol,
const double *x,
const int point_pos_ordering,
const double *xElemCoord,
const int nel,
const double *wtend,
const double *boxinfo,
const bool obb_check,
const int hash_n,
const double *hashMin,
const double *hashFac,
unsigned int *hashOffset,
unsigned int *const code_base,
unsigned int *const el_base,
double *const r_base,
double *const dist2_base,
const double *gll1D,
const double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_NEL = nel*D1D;
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
const int nThreads = D1D*sDIM;
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
{
constexpr int size1 = 3*MD1 + 13;
constexpr int size2 = 3*MD1;
constexpr int size3 = MD1*sDIM;
MFEM_SHARED findptsElementPoint_t el_pts[2];
MFEM_SHARED double r_workspace[size1];
MFEM_SHARED double constraint_workspace[size2];
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
double *r_workspace_ptr = r_workspace;
findptsElementPoint_t *fpt, *tmp;
fpt = el_pts + 0;
tmp = el_pts + 1;
int id_x = point_pos_ordering==0 ? i : i*sDIM;
int id_y = point_pos_ordering==0 ? npt+i : 1+i*sDIM;
int id_z = point_pos_ordering==0 ? 2*npt+i : 2+i*sDIM;
double x_i[3] = {x[id_x], x[id_y], x[id_z]};
unsigned int *code_i = code_base + i;
double *dist2_i = dist2_base + i;
//// map_points_to_els ////
findptsLocalHashData_t hash;
for (int d=0; d<sDIM; ++d)
{
hash.bnd[d].min = hashMin[d];
hash.fac[d] = hashFac[d];
}
hash.hash_n = hash_n;
hash.offset = hashOffset;
const unsigned int hi = hash_index(&hash, x_i);
const unsigned int *elp = hash.offset + hash.offset[hi];
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
*code_i = CODE_NOT_FOUND;
*dist2_i = HUGE_VAL;
for (; elp!=ele; ++elp)
{
const unsigned int el = *elp;
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
bool pass_bb = true;
obbox_t box;
if (obb_check)
{
for (int idx = 0; idx < sDIM; ++idx)
{
box.c0[idx] = boxinfo[n_box_ents*el + idx];
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
}
for (int idx = 0; idx < sDIM2; ++idx)
{
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
}
pass_bb = (bbox_test(&box, x_i) >= 0);
}
else
{
for (int d = 0; d < sDIM; ++d)
{
box.x[d].min = boxinfo[n_box_ents*el + d];
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
}
pass_bb = (AABB_test(&box, x_i) >= 0);
}
if (pass_bb)
{
//// findpts_local ////
{
if (MD1 <= 6)
{
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
{
const int qp = j % D1D;
const int d = j / D1D;
elem_coords[qp + d*D1D] =
xElemCoord[qp + el*D1D + d*p_NEL];
}
MFEM_SYNC_THREAD;
}
const double *elx[sDIM];
for (int d=0; d<sDIM; d++)
{
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
xElemCoord + d*p_NEL + el*D1D;
}
MFEM_SYNC_THREAD;
//// findpts_el ////
{
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
fpt->dist2p = 0;
fpt->tr = 1.0;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
fpt->x[j] = x_i[j];
}
MFEM_SYNC_THREAD;
//// seed ////
{
double *dist2_temp = r_workspace_ptr;
double *r_temp = dist2_temp + D1D;
MFEM_FOREACH_THREAD(j,x,nThreads)
{
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
fpt->dist2 = HUGE_VAL;
for (int ir=0; ir<D1D; ++ir)
{
if (dist2_temp[ir] < fpt->dist2)
{
fpt->dist2 = dist2_temp[ir];
fpt->r = r_temp[ir];
}
}
}
MFEM_SYNC_THREAD;
} //seed done
MFEM_FOREACH_THREAD(j,x,1)
{
tmp->dist2 = HUGE_VAL;
tmp->dist2p = 0;
tmp->tr = 1;
tmp->flags = 0;
tmp->r = fpt->r;
}
MFEM_FOREACH_THREAD(j,x,sDIM)
{
tmp->x[j] = fpt->x[j];
}
MFEM_SYNC_THREAD;
for (int step=0; step<50; step++)
{
switch (num_constrained(tmp->flags & FLAG_MASK))
{
case 0:
{
double *wt = r_workspace_ptr;
double *resid = wt + 3*D1D;
double *jac = resid + sDIM;
double *hess = jac + sDIM*rDIM;
findptsElementGEdge_t edge;
for (int d=0; d<sDIM; ++d)
{
edge.x[d] = constraint_workspace + d*D1D;
}
MFEM_FOREACH_THREAD(j,x,D1D)
{
for (int d=0; d<sDIM; ++d)
{
edge.x[d][j] = elx[d][j];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,D1D)
{
lag_eval_second_der(wt, tmp->r, j, gll1D,
lagcoeff, D1D);
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,sDIM)
{
resid[j] = tmp->x[j];
jac[j] = 0.0;
hess[j] = 0.0;
for (int k=0; k<D1D; ++k)
{
resid[j] -= wt[ k]*edge.x[j][k];
jac[j] += wt[D1D+k]*edge.x[j][k];
hess[j] += wt[2*D1D+k]*edge.x[j][k];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
hess[3] = resid[0]*hess[0] + resid[1]*hess[1] +
resid[2]*hess[2];
}
MFEM_FOREACH_THREAD(l,x,1)
{
if (!reject_prior_step_q(fpt,resid,tmp,tol))
{
newton_edge(fpt,jac,hess[3],resid,
tmp->flags&FLAG_MASK,tmp,tol);
}
}
MFEM_SYNC_THREAD;
break;
}
case 1:
{
MFEM_FOREACH_THREAD(j,x,1)
{
const int pi = point_index(tmp->flags &
FLAG_MASK);
const double *wt = wtend + pi*3*D1D;
findptsElementGPT_t gpt;
for (int d=0; d<sDIM; ++d)
{
gpt.x[d] = elx[d][pi*(D1D-1)];
gpt.jac[d] = 0.0;
gpt.hes[d] = 0.0;
for (int k=0; k<D1D; ++k)
{
gpt.jac[d] += wt[D1D +k]*elx[d][k];
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
}
}
const double *const pt_x = gpt.x;
const double *const jac = gpt.jac;
const double *const hes = gpt.hes;
double resid[sDIM], steep, sr;
resid[0] = fpt->x[0] - pt_x[0];
resid[1] = fpt->x[1] - pt_x[1];
resid[2] = fpt->x[2] - pt_x[2];
steep = jac[0]*resid[0] + jac[1]*resid[1] +
jac[2]*resid[2];
sr = steep*tmp->r;
if (!reject_prior_step_q(fpt, resid, tmp, tol))
{
if (sr<0)
{
const double rhess = resid[0]*hes[0] +
resid[1]*hes[1] +
resid[2]*hes[2];
newton_edge(fpt, jac, rhess,
resid, 0, tmp, tol);
}
else // sr==0
{
fpt->r = tmp->r;
fpt->dist2p = 0;
fpt->flags = tmp->flags | CONVERGED_FLAG;
}
}
}
MFEM_SYNC_THREAD;
break;
} // case 1
} //switch
if (fpt->flags & CONVERGED_FLAG)
{
break;
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
*tmp = *fpt;
}
MFEM_SYNC_THREAD;
} // for step<50
} // findpts_el
bool converged_internal =
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
(fpt->dist2<dist2tol);
if (*code_i==CODE_NOT_FOUND || converged_internal ||
fpt->dist2<*dist2_i)
{
MFEM_FOREACH_THREAD(j,x,1)
{
*(el_base+i) = el;
*code_i = converged_internal?CODE_INTERNAL:CODE_BORDER;
*dist2_i = fpt->dist2;
*(r_base+i) = fpt->r;
}
MFEM_SYNC_THREAD;
if (converged_internal)
{
break;
}
}
} // findpts_local
} // obbox_test
} // elp
});
}
void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt)
{
if (npt == 0)
{
return;
}
MFEM_VERIFY(spacedim==3 && dim == 1,"Function for 3D edges only");
bool use_dev = point_pos.UseDevice();
auto pp = point_pos.Read(use_dev);
auto pgslm = gsl_mesh.Read(use_dev);
auto pwt = DEV.wtend.Read(use_dev);
auto pbb = DEV.bb.Read(use_dev);
auto plhm = DEV.lh_min.Read(use_dev);
auto plhf = DEV.lh_fac.Read(use_dev);
auto plho = DEV.lh_offset.ReadWrite(use_dev);
auto pcode = code.Write(use_dev);
auto pelem = elem.Write(use_dev);
auto pref = ref.Write(use_dev);
auto pdist = dist.Write(use_dev);
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
auto plc = DEV.lagcoeff.Read(use_dev);
double dist2tol = DEV.surf_dist_tol;
const bool obb_chk = obb_check;
switch (DEV.dof1d)
{
case 2:
FindPointsEdgeLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 3:
FindPointsEdgeLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
case 4:
FindPointsEdgeLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
pp, point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc);
break;
default:
FindPointsEdgeLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
point_pos_ordering, pgslm,
NE_split_total, pwt, pbb, obb_chk,
DEV.lh_nx, plhm, plhf, plho,
pcode, pelem, pref, pdist,
pgll1d, plc, DEV.dof1d);
break;
}
}
#undef rDIM2
#undef sDIM2
#undef rDIM
#undef sDIM
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::FindPointsEdgeLocal3( const Vector &point_pos,
int point_pos_ordering,
Array<unsigned int> &code,
Array<unsigned int> &elem,
Vector &ref,
Vector &dist,
int npt ) {} ;
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
File diff suppressed because it is too large Load Diff
-190
View File
@@ -1,190 +0,0 @@
#ifndef MFEM_GSLIB_KERNEL_HELPERS_HPP
#define MFEM_GSLIB_KERNEL_HELPERS_HPP
#include "../../config/config.hpp"
#include <cmath>
namespace mfem
{
namespace gslib
{
struct dbl_range_t
{
double min, max;
};
template <int SDIM>
struct obbox_t
{
double c0[SDIM], A[SDIM * SDIM];
dbl_range_t x[SDIM];
};
template <int SDIM>
struct findptsLocalHashData_t
{
int hash_n;
dbl_range_t bnd[SDIM];
double fac[SDIM];
unsigned int *offset;
};
// Eval the ith Lagrange interpolant at x.
MFEM_HOST_DEVICE inline void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
const double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
// Eval the ith Lagrange interpolant and its first derivative at x.
MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
const double d_j = 2 * (x - z[j]);
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN + i] = 2.0 * lCoeff[i] * u1;
}
// Eval the ith Lagrange interpolant and its first and second derivative at x.
MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
int i, const double *z,
const double *lCoeff,
int pN)
{
double u0 = 1, u1 = 0, u2 = 0;
for (int j = 0; j < pN; ++j)
{
if (i != j)
{
const double d_j = 2 * (x - z[j]);
u2 = d_j * u2 + u1;
u1 = d_j * u1 + u0;
u0 = d_j * u0;
}
}
p0[i] = lCoeff[i] * u0;
p0[pN + i] = 2.0 * lCoeff[i] * u1;
p0[2 * pN + i] = 8.0 * lCoeff[i] * u2;
}
// Solve Ax=y where A is a symmetric 2x2 matrix packed as {a00, a01, a11}.
MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
const double A[3],
const double y[2])
{
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
}
// Positive when the point is inside the axis-aligned bounding box.
template <int SDIM>
MFEM_HOST_DEVICE inline double AABB_test(const obbox_t<SDIM> *const b,
const double (&x)[SDIM])
{
double test = 1.0;
for (int d = 0; d < SDIM; ++d)
{
const double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
test = test < 0.0 ? test : b_d;
}
return test;
}
// Positive when the point is inside the oriented bounding box.
template <int SDIM>
MFEM_HOST_DEVICE inline double bbox_test(const obbox_t<SDIM> *const b,
const double (&x)[SDIM])
{
const double bxyz = AABB_test(b, x);
if (bxyz < 0.0)
{
return bxyz;
}
double dxyz[SDIM];
for (int d = 0; d < SDIM; ++d)
{
dxyz[d] = x[d] - b->c0[d];
}
double test = 1.0;
for (int d = 0; d < SDIM; ++d)
{
double rst = 0.0;
for (int e = 0; e < SDIM; ++e)
{
rst += b->A[d * SDIM + e] * dxyz[e];
}
const double brst = (rst + 1.0) * (1.0 - rst);
test = test < 0.0 ? test : brst;
}
return test;
}
// Hash index in the hash table for the point x.
template <int SDIM>
MFEM_HOST_DEVICE inline int hash_index(
const findptsLocalHashData_t<SDIM> *const p,
const double (&x)[SDIM])
{
const int n = p->hash_n;
int sum = 0;
for (int d = SDIM - 1; d >= 0; --d)
{
sum *= n;
const int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
}
return sum;
}
// Squared Euclidean norm.
template <int SDIM>
MFEM_HOST_DEVICE inline double l2norm2(const double (&x)[SDIM])
{
double sum = 0.0;
for (int d = 0; d < SDIM; ++d)
{
sum += x[d] * x[d];
}
return sum;
}
template <int SDIM>
MFEM_HOST_DEVICE inline double l2norm2(const double *x)
{
double sum = 0.0;
for (int d = 0; d < SDIM; ++d)
{
sum += x[d] * x[d];
}
return sum;
}
} // namespace gslib
} // namespace mfem
#endif
-152
View File
@@ -1,152 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-function"
#endif
#include "gslib.h"
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
#define GSLIB_RELEASE_VERSION 10007
#endif
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
#pragma GCC diagnostic pop
#endif
namespace mfem
{
#if GSLIB_RELEASE_VERSION >= 10009
#define CODE_INTERNAL 0
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
using gslib::lagrange_eval;
template<int T_D1D = 0>
static void InterpolateLocal1DKernel(const double *const gf_in,
int *const el,
double *const r,
double *const int_out,
const int npt,
const int nfields,
double *gll1D,
double *lagcoeff,
const int pN = 0)
{
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : pN;
const int p_Nq = D1D;
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
// for each point of the npt points, create a thread block of size dof1Dsol
mfem::forall_2D(npt, D1D, 1, [=] MFEM_HOST_DEVICE (int i)
{
MFEM_SHARED double wtr[MD1];
MFEM_SHARED double sums[MD1];
// Evaluate basis functions at the reference space coordinates
MFEM_FOREACH_THREAD(j,x,D1D)
{
lagrange_eval(wtr, r[i], j, p_Nq, gll1D, lagcoeff);
}
MFEM_SYNC_THREAD;
for (int fld=0; fld<nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Nq + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
const int elemOffset = el[i]*nfields*p_Nq + fld*p_Nq;
MFEM_FOREACH_THREAD(j,x,D1D)
{
sums[j] = wtr[j] * gf_in[elemOffset + j];
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(j,x,1)
{
double sumv = 0.0;
// sum the contributions of each lagrange polynomial
for (int jj=0; jj<D1D; ++jj)
{
sumv += sums[jj];
}
int_out[fld*npt + i] = sumv;
}
MFEM_SYNC_THREAD;
}
});
}
void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt,
int ncomp,
int dof1Dsol )
{
MFEM_VERIFY(dim == 1, "Kernel for edges only.");
if (npt == 0) { return; }
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
switch (dof1Dsol)
{
case 2:
InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
}
}
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
#else
void FindPointsGSLIB::InterpolateLocal1(const Vector &field_in,
Array<int> &gsl_elem_dev_l,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol) {};
#endif
} // namespace mfem
#endif //ifdef MFEM_USE_GSLIB
+41 -36
View File
@@ -11,7 +11,6 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -33,7 +32,18 @@ namespace mfem
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
using gslib::lagrange_eval;
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
template<int T_D1D = 0>
static void InterpolateLocal2DKernel(const double *const gf_in,
@@ -42,6 +52,8 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
double *const int_out,
const int npt,
const int ncomp,
const int nel,
const int gf_offset,
double *gll1D,
double *lagcoeff,
const int pN = 0)
@@ -52,8 +64,6 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
const int p_Np = D1D*D1D;
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
"Increase Max allowable polynomial order.");
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
mfem::forall_2D(npt, D1D, D1D, [=] MFEM_HOST_DEVICE (int i)
{
@@ -72,9 +82,9 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
for (int fld = 0; fld < Nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Np + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
//if using R->Mult for L -> E-Vec use below: NDOFSxVDIMxNEL
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
MFEM_FOREACH_THREAD(j,x,D1D)
{
@@ -110,38 +120,33 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol)
int nel, int dof1Dsol)
{
if (npt == 0) { return; }
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
const int gf_offset = field_in.Size()/ncomp;
auto pfin = field_in.Read();
auto pgsl = gsl_elem_dev_l.ReadWrite();
auto pgslr = gsl_ref_l.ReadWrite();
auto pfout = field_out.Write();
auto pgll = DEV.gll1d_sol.ReadWrite();
auto plcf = DEV.lagcoeff_sol.ReadWrite();
switch (dof1Dsol)
{
case 2:
InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf, dof1Dsol);
}
}
@@ -155,7 +160,7 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol) {};
int nel, int dof1Dsol) {};
#endif
} // namespace mfem
+41 -35
View File
@@ -11,7 +11,6 @@
#include "../gslib.hpp"
#include "../../general/forall.hpp"
#include "gslib_kernel_helpers.hpp"
#ifdef MFEM_USE_GSLIB
@@ -33,7 +32,18 @@ namespace mfem
#define CODE_BORDER 1
#define CODE_NOT_FOUND 2
using gslib::lagrange_eval;
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
int i, int p_Nq,
double *z, double *lagrangeCoeff)
{
double p_i = (1 << (p_Nq - 1));
for (int j = 0; j < p_Nq; ++j)
{
double d_j = x - z[j];
p_i *= j == i ? 1 : d_j;
}
p0[i] = lagrangeCoeff[i] * p_i;
}
template<int T_D1D = 0>
static void InterpolateLocal3DKernel(const double *const gf_in,
@@ -42,6 +52,8 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
double *const int_out,
const int npt,
const int ncomp,
const int nel,
const int gf_offset,
double *gll1D,
double *lagcoeff,
const int pN = 0)
@@ -72,9 +84,9 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
for (int fld = 0; fld < Nfields; ++fld)
{
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
// offset would be `el[i] * p_Np + fld * gf_offset`.
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
//if using R->Mult for L -> E-Vec use below.
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
MFEM_FOREACH_THREAD(j,x,D1D)
{
@@ -113,43 +125,37 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol)
int nel, int dof1Dsol)
{
if (npt == 0) { return; }
bool use_dev = field_in.UseDevice();
auto pfin = field_in.Read(use_dev);
auto pgsle = gsl_elem_dev_l.ReadWrite(use_dev);
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
auto pfout = field_out.Write(use_dev);
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
const int gf_offset = field_in.Size()/ncomp;
auto pfin = field_in.Read();
auto pgsle = gsl_elem_dev_l.ReadWrite();
auto pgslr = gsl_ref_l.ReadWrite();
auto pfout = field_out.Write();
auto pgll = DEV.gll1d_sol.ReadWrite();
auto plcf = DEV.lagcoeff_sol.ReadWrite();
switch (dof1Dsol)
{
case 2:
InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 3:
InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 4:
InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
case 5:
InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf);
break;
default:
InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
npt, ncomp, pgll, plcf, dof1Dsol);
break;
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf);
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
npt, ncomp, nel, gf_offset,
pgll, plcf, dof1Dsol);
}
}
#undef MAXC
#undef CODE_INTERNAL
#undef CODE_BORDER
#undef CODE_NOT_FOUND
@@ -159,7 +165,7 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
Vector &gsl_ref_l,
Vector &field_out,
int npt, int ncomp,
int dof1Dsol) {};
int nel, int dof1Dsol) {};
#endif
} // namespace mfem
@@ -10,7 +10,6 @@
// CONTRIBUTING.md for details.
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -20,13 +19,6 @@ namespace mfem
DiffusionIntegrator::Kernels::Kernels()
{
// 2D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<2,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<2,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<2,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<2,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,7,6>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<2,1,1>();
DiffusionIntegrator::AddSpecialization<2,2,2>();
@@ -48,18 +40,7 @@ DiffusionIntegrator::Kernels::Kernels()
DiffusionIntegrator::AddSpecialization<2,8,9>();
DiffusionIntegrator::AddSpecialization<2,9,10>();
// others
DiffusionIntegrator::AddSimplexSpecialization<2,2,5>();
DiffusionIntegrator::AddSimplexSpecialization<2,3,6>();
// 3D
// Q = P, only for simplex
DiffusionIntegrator::AddSimplexSpecialization<3,2,1>();
DiffusionIntegrator::AddSimplexSpecialization<3,3,2>();
DiffusionIntegrator::AddSimplexSpecialization<3,4,3>();
DiffusionIntegrator::AddSimplexSpecialization<3,5,4>();
DiffusionIntegrator::AddSimplexSpecialization<3,6,5>();
DiffusionIntegrator::AddSimplexSpecialization<3,7,6>();
DiffusionIntegrator::AddSimplexSpecialization<3,8,7>();
// Q = P+1
DiffusionIntegrator::AddSpecialization<3,1,1>();
DiffusionIntegrator::AddSpecialization<3,2,2>();
+32 -27
View File
@@ -12,6 +12,7 @@
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
#include "../kernel_dispatch.hpp"
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
@@ -19,8 +20,6 @@
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp"
namespace mfem
{
@@ -638,8 +637,8 @@ inline void SmemPADiffusionApply2D(const int NE,
const bool symmetric,
const Array<real_t> &b_,
const Array<real_t> &g_,
const Array<real_t> &,
const Array<real_t> &,
const Array<real_t> &bt_,
const Array<real_t> &gt_,
const Vector &d_,
const Vector &x_,
Vector &y_,
@@ -1065,6 +1064,8 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
// Grad X
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
@@ -1085,6 +1086,8 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
// Grad Y
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
@@ -1106,6 +1109,8 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
// Grad Z + Q-function
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
@@ -1218,48 +1223,48 @@ inline void SmemPADiffusionApply3D(const int NE,
namespace
{
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
using ApplySimplexKernelType = DiffusionIntegrator::ApplySimplexKernelType;
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
using DiffusionApplyKernelType =
DiffusionIntegrator::DiffusionApplyKernelType;
using DiffusionDiagonalKernelType =
DiffusionIntegrator::DiffusionDiagonalKernelType;
}
template<int DIM, int D1D, int Q1D>
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
template<int DIM, int T_D1D, int T_Q1D>
DiffusionApplyKernelType DiffusionIntegrator::DiffusionApplyPAKernel::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
}
inline
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
inline DiffusionApplyKernelType
DiffusionIntegrator::DiffusionApplyPAKernel::Fallback(int DIM, int, int)
{
if (dim == 2) { return internal::PADiffusionApply2D; }
else if (dim == 3) { return internal::PADiffusionApply3D; }
if (DIM == 2) { return internal::PADiffusionApply2D; }
else if (DIM == 3) { return internal::PADiffusionApply3D; }
else { MFEM_ABORT(""); }
}
template<int DIM, int D1D, int Q1D>
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
DiffusionDiagonalKernelType
DiffusionIntegrator::DiffusionDiagonalPAKernel::Kernel()
{
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D, Q1D>; }
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
MFEM_ABORT("");
}
inline DiagonalKernelType
DiffusionIntegrator::DiagonalPAKernels::Fallback(int dim, int, int)
inline DiffusionDiagonalKernelType
DiffusionIntegrator::DiffusionDiagonalPAKernel::Fallback(int DIM, int, int)
{
if (dim == 2) { return internal::PADiffusionDiagonal2D; }
else if (dim == 3) { return internal::PADiffusionDiagonal3D; }
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+9 -38
View File
@@ -15,7 +15,6 @@
#include "../../mesh/nurbs.hpp"
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "bilininteg_diffusion_kernels.hpp"
#include "bilininteg_diffusion_pa_simplices.hpp"
namespace mfem
{
@@ -32,8 +31,8 @@ void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
const Array<real_t> &B = maps->B;
const Array<real_t> &G = maps->G;
const Vector &Dv = pa_data;
DiagonalPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
diag, dofs1D, quad1D);
DiffusionDiagonalPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
diag, dofs1D, quad1D);
}
}
@@ -69,26 +68,8 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
#endif // MFEM_USE_OCCA
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
return ApplySimplexPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
rmaps->lex_map,
rmaps->forward_map2d_diff,
rmaps->inverse_map2d_diff,
rmaps->forward_map3d_diff,
rmaps->inverse_map3d_diff,
rmaps->Ga1,
rmaps->Ga2,
rmaps->Ga3,
rmaps->Ga1t,
rmaps->Ga2t,
rmaps->Ga3t,
Dv, x, y, dofs1D, quad1D);
}
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
Gt, Dv, x, y, dofs1D, quad1D);
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
Gt, Dv, x, y, dofs1D, quad1D);
}
}
@@ -113,8 +94,7 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
fespace = &fes;
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, stroud);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -139,22 +119,13 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
dim = mesh->Dimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
const int sdim = mesh->SpaceDimension();
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
if (MQ) { coeff.ProjectTranspose(*MQ); }
else if (VQ) { coeff.Project(*VQ); }
@@ -203,9 +174,9 @@ void DiffusionIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
abs_pa_data.Abs();
auto abs_maps = maps->Abs();
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
abs_pa_data, x, y, dofs1D, quad1D);
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric,
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
abs_pa_data, x, y, dofs1D, quad1D);
}
void DiffusionIntegrator::AddAbsMultTransposePA(const Vector &x,
File diff suppressed because it is too large Load Diff
-3
View File
@@ -10,7 +10,6 @@
// CONTRIBUTING.md for details.
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp" // IWYU pragma: keep
namespace mfem
{
@@ -40,10 +39,8 @@ MassIntegrator::Kernels::Kernels()
MassIntegrator::AddSpecialization<2,9,10>();
// others
MassIntegrator::AddSpecialization<2,2,4>();
MassIntegrator::AddSpecialization<2,2,5>();
MassIntegrator::AddSpecialization<2,3,6>();
MassIntegrator::AddSpecialization<2,4,6>();
// 3D
// Q=P+1
MassIntegrator::AddSpecialization<3,1,1>();
+63 -99
View File
@@ -19,8 +19,6 @@
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "bilininteg_mass_pa_simplices.hpp"
namespace mfem
{
@@ -183,12 +181,6 @@ constexpr int NBZ(int D1D)
{
return ipow(2, D(D1D) >= 0 ? D(D1D) : 0);
}
constexpr int NBZ3D(int MDQ)
{
return MDQ > 0 ? std::min<int>(
(128 + MDQ * MDQ * MDQ - 1) / (MDQ * MDQ * MDQ), 64)
: 1;
}
}
// Shared memory PA Mass Diagonal 2D kernel
@@ -812,23 +804,19 @@ void PAMassApply3D_Element(const int e,
}
}
template <int T_D1D, int T_Q1D, int TBATCH, bool ACCUMULATE = true>
MFEM_HOST_DEVICE inline void
SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
const real_t *d_, const real_t *x_, real_t *y_,
int d1d = 0, int q1d = 0)
template<int T_D1D, int T_Q1D, bool ACCUMULATE = true>
MFEM_HOST_DEVICE inline
void SmemPAMassApply3D_Element(const int e,
const int NE,
const real_t *b_,
const real_t *d_,
const real_t *x_,
real_t *y_,
const int d1d = 0,
const int q1d = 0)
{
static_assert(TBATCH > 0, "TBATCH must be positive");
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
constexpr int tbatch = TBATCH;
const int tidz = MFEM_THREAD_ID(z);
#else
// host always batch size 1
constexpr int tbatch = 1;
constexpr int tidz = 0;
#endif
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int D1D = T_D1D ? T_D1D : d1d;
constexpr int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
@@ -841,37 +829,33 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
MFEM_SHARED real_t sDQ[MQ1*MD1];
real_t (*B)[MD1] = (real_t (*)[MD1]) sDQ;
real_t (*Bt)[MQ1] = (real_t (*)[MQ1]) sDQ;
MFEM_SHARED real_t sm0[tbatch][MDQ*MDQ*MDQ];
MFEM_SHARED real_t sm1[tbatch][MDQ*MDQ*MDQ];
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+tidz);
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) (sm1+tidz);
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm0+tidz);
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm1+tidz);
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) (sm0+tidz);
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm1+tidz);
MFEM_FOREACH_THREAD(dy, y, D1D)
MFEM_SHARED real_t sm0[MDQ*MDQ*MDQ];
MFEM_SHARED real_t sm1[MDQ*MDQ*MDQ];
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm0;
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) sm1;
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm0;
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm1;
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) sm0;
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm1;
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
MFEM_UNROLL(MD1)
for (int dz = 0; dz < D1D; ++dz)
{
X[dz][dy][dx] = x(dx, dy, dz, e);
X[dz][dy][dx] = x(dx,dy,dz,e);
}
}
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
}
if (tidz == 0)
{
MFEM_FOREACH_THREAD(dy, y, D1D)
MFEM_FOREACH_THREAD(dx,x,Q1D)
{
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
B[dx][dy] = b(dx,dy);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy, y, D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -896,9 +880,9 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy, y, Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -923,9 +907,9 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy, y, Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx, x, Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -945,22 +929,22 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
MFEM_UNROLL(MQ1)
for (int qz = 0; qz < Q1D; qz++)
{
QQQ[qz][qy][qx] = u[qz] * d(qx, qy, qz, e);
QQQ[qz][qy][qx] = u[qz] * d(qx,qy,qz,e);
}
}
}
MFEM_SYNC_THREAD;
if (tidz == 0)
MFEM_FOREACH_THREAD(di,y,D1D)
{
MFEM_FOREACH_THREAD(di, y, D1D)
MFEM_FOREACH_THREAD(q,x,Q1D)
{
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[di][q] = b(q, di); }
Bt[di][q] = b(q,di);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qy, y, Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -985,9 +969,9 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy, y, D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u[Q1D];
MFEM_UNROLL(MQ1)
@@ -1012,9 +996,9 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dy, y, D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx, x, D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u[D1D];
MFEM_UNROLL(MD1)
@@ -1036,11 +1020,11 @@ SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
{
if (ACCUMULATE)
{
y(dx, dy, dz, e) += u[dz];
y(dx,dy,dz,e) += u[dz];
}
else
{
y(dx, dy, dz, e) = u[dz];
y(dx,dy,dz,e) = u[dz];
}
}
}
@@ -1131,8 +1115,8 @@ inline void PAMassApply3D(const int NE,
});
}
// Shared memory PA Mass Apply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0, int TBATCH=1>
// Shared memory PA Mass Apply 2D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPAMassApply3D(const int NE,
const Array<real_t> &b_,
const Array<real_t> &bt_,
@@ -1142,9 +1126,6 @@ inline void SmemPAMassApply3D(const int NE,
const int d1d = 0,
const int q1d = 0)
{
static_assert(T_D1D > 0, "T_D1D must be positive");
static_assert(T_Q1D > 0, "T_Q1D must be positive");
static_assert(TBATCH > 0, "TBATCH must be positive");
MFEM_CONTRACT_VAR(bt_);
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -1156,11 +1137,9 @@ inline void SmemPAMassApply3D(const int NE,
const auto d = d_.Read();
const auto x = x_.Read();
auto y = y_.ReadWrite();
mfem::forall_2D_batch<T_Q1D * T_Q1D * TBATCH>(NE, Q1D, Q1D, TBATCH,
[=] MFEM_HOST_DEVICE(int e)
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
internal::SmemPAMassApply3D_Element<T_D1D, T_Q1D, TBATCH>(e, NE, b, d, x,
y, d1d, q1d);
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
});
}
@@ -1410,57 +1389,42 @@ using ApplyKernelType = MassIntegrator::ApplyKernelType;
using DiagonalKernelType = MassIntegrator::DiagonalKernelType;
}
template<int DIM, int D1D, int Q1D>
template<int DIM, int T_D1D, int T_Q1D>
ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<D1D, Q1D>; }
else if constexpr (DIM == 3)
{
constexpr int MDQ = D1D >= Q1D ? D1D : Q1D;
// max 64 threads in z limit in cuda and hip
if constexpr (MDQ > 0)
{
return internal::SmemPAMassApply3D<D1D, Q1D,
internal::mass::NBZ3D(MDQ)>;
}
}
else { MFEM_ABORT(""); }
return nullptr;
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassApply3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
}
inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
int dim, int, int)
int DIM, int, int)
{
if (dim == 1) { return internal::PAMassApply1D; }
else if (dim == 2) { return internal::PAMassApply2D; }
else if (dim == 3) { return internal::PAMassApply3D; }
if (DIM == 1) { return internal::PAMassApply1D; }
else if (DIM == 2) { return internal::PAMassApply2D; }
else if (DIM == 3) { return internal::PAMassApply3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
template<int DIM, int D1D, int Q1D>
template<int DIM, int T_D1D, int T_Q1D>
DiagonalKernelType MassIntegrator::DiagonalPAKernels::Kernel()
{
if constexpr (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<D1D, Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
return nullptr;
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
MFEM_ABORT("");
}
inline DiagonalKernelType MassIntegrator::DiagonalPAKernels::Fallback(
int dim, int, int)
int DIM, int, int)
{
if (dim == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (dim == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (dim == 3) { return internal::PAMassAssembleDiagonal3D; }
if (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
else if (DIM == 2) { return internal::PAMassAssembleDiagonal2D; }
else if (DIM == 3) { return internal::PAMassAssembleDiagonal3D; }
else { MFEM_ABORT(""); }
return nullptr;
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+5 -43
View File
@@ -15,7 +15,6 @@
#include "../qfunction.hpp"
#include "../ceed/integrators/mass/mass.hpp"
#include "bilininteg_mass_kernels.hpp"
#include "bilininteg_mass_pa_simplices.hpp"
namespace mfem
{
@@ -30,11 +29,9 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
// Assuming the same element type
fespace = &fes;
Mesh *mesh = fes.GetMesh();
dim = mesh->Dimension();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation *T0 = mesh->GetTypicalElementTransformation();
const bool stroud = fes.UsesRaggedTensorBasis();
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0, stroud);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0);
if (DeviceCanUseCeed())
{
delete ceedOp;
@@ -51,25 +48,17 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
return;
}
int map_type = el.GetMapType();
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::DETERMINANTS, mt);
if (stroud)
{
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
}
else
{
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
}
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, mt);
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
// QuadratureSpace expects ir defined in reference simplex for Bernstein
// elements with partial assembly
{
const int NE = ne;
const int NQ = nq;
@@ -158,10 +147,9 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const int D1D = dofs1D;
const int Q1D = quad1D;
const Vector &D = pa_data;
const Array<real_t> &B = maps->B;
const Array<real_t> &Bt = maps->Bt;
const Vector &D = pa_data;
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
@@ -176,31 +164,7 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
}
#endif // MFEM_USE_OCCA
if (fespace->UsesRaggedTensorBasis())
{
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
const Array<real_t> &Ba1 = rmaps->Ba1;
const Array<real_t> &Ba2 = rmaps->Ba2;
const Array<real_t> &Ba3 = rmaps->Ba3;
const Array<real_t> &Ba1t = rmaps->Ba1t;
const Array<real_t> &Ba2t = rmaps->Ba2t;
const Array<real_t> &Ba3t = rmaps->Ba3t;
const Array<int> &lex_map = rmaps->lex_map;
const Array<int> &forward_map2d = rmaps->forward_map2d_mass;
const Array<int> &inverse_map2d = rmaps->inverse_map2d_mass;
const Array<int> &forward_map3d = rmaps->forward_map3d_mass;
const Array<int> &inverse_map3d = rmaps->inverse_map3d_mass;
ApplySimplexPAKernels::Run(dim, D1D, Q1D, ne, lex_map, forward_map2d,
inverse_map2d,
forward_map3d, inverse_map3d, Ba1, Ba2, Ba3, Ba1t, Ba2t, Ba3t,
D, x, y, D1D, Q1D);
}
else
{
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
}
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
}
}
@@ -213,8 +177,6 @@ void MassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
}
else
{
MFEM_VERIFY(!fespace->UsesRaggedTensorBasis(),
"AbsMultPA not implemented for ragged tensor basis");
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absB(maps->B);
File diff suppressed because it is too large Load Diff
-500
View File
@@ -307,506 +307,6 @@ DomainLFIntegrator::AssembleKernels::Kernel()
MFEM_ABORT("");
}
template <int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble2D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 2 * (d - 1) * d * ne, "");
constexpr int vdim = 2;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1) : Reshape(F,vdim,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int vdim = 2;
if (M[e] == 0) { return; } // ignore
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
MFEM_SHARED real_t sQQ[vdim*Q*Q];
MFEM_SHARED real_t sQD[vdim*Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
const DeviceCube QQ(sQQ, q, q, vdim);
const DeviceCube QD(sQD, q, d, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const real_t cst_val_0 = C(0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
const real_t J0 = J(x,y,0,vd,e);
const real_t J1 = J(x,y,1,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,e);
QQ(x,y,vd) = W(x,y)*(J0*C0 + J1*C1);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t qd = 0.0;
for (int qx = 0; qx < q; ++qx)
{
qd += QQ(qx,qy,vd) * Btx(dx,qx);
}
QD(dx,qy,vd) = qd;
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
DeviceTensor<4> Yxy(Y, nx, ny, vdim, ne);
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t dd = 0.0;
for (int qy = 0; qy < q; ++qy)
{
dd += QD(dx,qy,vd) * Bty(dy,qy);
}
Yxy(dx,dy,vd,e) += dd;
}
}
}
MFEM_SYNC_THREAD;
});
}
template <int T_D1D = 0, int T_Q1D = 0>
static void HdivDLFAssemble3D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HDIV_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HDIV_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 3 * (d - 1) * (d - 1) * d * ne, "y wrong length");
constexpr int vdim = 3;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int vdim = 3;
if (M[e] == 0) { return; } // ignore
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HDIV_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HDIV_MAX_D1D;
MFEM_SHARED real_t sBot[Q*D];
MFEM_SHARED real_t sBct[Q*D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d-1, q);
kernels::internal::LoadB<D,Q>(d-1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D,Q>(d, q, BC, sBct);
MFEM_SHARED real_t sm0[vdim*Q*Q*Q];
MFEM_SHARED real_t sm1[vdim*Q*Q*Q];
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const real_t cst_val_0 = C(0,0,0,0,0);
const real_t cst_val_1 = C(1,0,0,0,0);
const real_t cst_val_2 = C(2,0,0,0,0);
MFEM_FOREACH_THREAD(y,y,q)
{
MFEM_FOREACH_THREAD(x,x,q)
{
for (int z = 0; z < q; ++z)
{
const real_t J0 = J(x,y,z,0,vd,e);
const real_t J1 = J(x,y,z,1,vd,e);
const real_t J2 = J(x,y,z,2,vd,e);
const real_t C0 = cst ? cst_val_0 : C(0,x,y,z,e);
const real_t C1 = cst ? cst_val_1 : C(1,x,y,z,e);
const real_t C2 = cst ? cst_val_2 : C(2,x,y,z,e);
QQQ(x,y,z,vd) = W(x,y,z)*(J0*C0 + J1*C1 + J2*C2);
}
}
}
}
MFEM_SYNC_THREAD;
// Apply Bt operator
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
DeviceMatrix Btx = (vd == 0) ? Bct : Bot;
MFEM_FOREACH_THREAD(qy,y,q)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qx = 0; qx < q; ++qx)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += QQQ(qx,qy,qz,vd) * Btx(dx,qx);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DQQ(dx,qy,qz,vd) = u[qz]; }
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
DeviceMatrix Bty = (vd == 1) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { u[qz] = 0.0; }
MFEM_UNROLL(Q)
for (int qy = 0; qy < q; ++qy)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += DQQ(dx,qy,qz,vd) * Bty(dy,qy);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz) { DDQ(dx,dy,qz,vd) = u[qz]; }
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd,z,vdim)
{
const int nx = (vd == 0) ? d : d-1;
const int ny = (vd == 1) ? d : d-1;
const int nz = (vd == 2) ? d : d-1;
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
DeviceMatrix Btz = (vd == 2) ? Bct : Bot;
MFEM_FOREACH_THREAD(dy,y,ny)
{
MFEM_FOREACH_THREAD(dx,x,nx)
{
real_t u[D];
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { u[dz] = 0.0; }
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] += DDQ(dx,dy,qz,vd) * Btz(dz,qz);
}
}
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz) { Yxyz(dx,dy,dz,vd,e) += u[dz]; }
}
}
}
MFEM_SYNC_THREAD;
});
}
/// @param ne number of elements
/// @param markers array where entry markers[e] == 0 to skip assembly over
/// element e element
/// @param jac Spatial Jacobians evaluated at all quadrature points
/// @param weights 1D quadrature weights
/// @param testBO 1D open basis test functions
/// @param testBC 1D closed basis test functions
/// @param coeff coefficient values evaluated at quadrature points, possibly
/// compressed.
/// @param d number of 1D closed dofs
/// @param q number of 1D quadrature points
/// @tparam T_D1D maximum number of dofs along any direction, or 0
/// @tparam T_Q1D maximum number of quadrature points along any direction, or 0
template <int T_D1D = 0, int T_Q1D = 0>
static void HcurlDLFAssemble3D(const int ne, const Array<int> &markers,
const Vector &jac, const Array<real_t> &weights,
const Array<real_t> &testBO,
const Array<real_t> &testBC, const Vector &coeff,
Vector &y, const int d, const int q)
{
MFEM_VERIFY(T_D1D || d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Problem size too large.");
MFEM_VERIFY(T_Q1D || q <= DeviceDofQuadLimits::Get().HCURL_MAX_Q1D,
"Problem size too large.");
MFEM_VERIFY(y.Size() == 3 * (d - 1) * d * d * ne, "y wrong length");
constexpr int vdim = 3;
const auto F = coeff.Read();
const auto M = markers.Read();
const auto BO = Reshape(testBO.Read(), q, d-1);
const auto BC = Reshape(testBC.Read(), q, d);
const auto J = Reshape(jac.Read(), q, q, q, vdim, vdim, ne);
const auto W = Reshape(weights.Read(), q, q, q);
const bool cst = coeff.Size() == vdim;
const auto C = cst ? Reshape(F,vdim,1,1,1,1) : Reshape(F,vdim,q,q,q,ne);
auto Y = y.ReadWrite();
mfem::forall_3D(ne, q, q, vdim, [=] MFEM_HOST_DEVICE(int e)
{
if (M[e] == 0)
{
// ignore
return;
}
constexpr int vdim = 3;
constexpr int Q = T_Q1D ? T_Q1D : DofQuadLimits::HCURL_MAX_Q1D;
constexpr int D = T_D1D ? T_D1D : DofQuadLimits::HCURL_MAX_D1D;
MFEM_SHARED real_t sBot[Q * D];
MFEM_SHARED real_t sBct[Q * D];
// Bo and Bc into shared memory
const DeviceMatrix Bot(sBot, d - 1, q);
kernels::internal::LoadB<D, Q>(d - 1, q, BO, sBot);
const DeviceMatrix Bct(sBct, d, q);
kernels::internal::LoadB<D, Q>(d, q, BC, sBct);
MFEM_SHARED real_t sm0[vdim * Q * Q * Q];
MFEM_SHARED real_t sm1[vdim * Q * Q * Q];
DeviceTensor<4> QQQ(sm1, q, q, q, vdim);
DeviceTensor<4> DQQ(sm0, d, q, q, vdim);
DeviceTensor<4> DDQ(sm1, d, d, q, vdim);
const real_t cst_val_0 = C(0, 0, 0, 0, 0);
const real_t cst_val_1 = C(1, 0, 0, 0, 0);
const real_t cst_val_2 = C(2, 0, 0, 0, 0);
MFEM_FOREACH_THREAD(vd, z, vdim)
{
MFEM_FOREACH_THREAD(y, y, q)
{
MFEM_FOREACH_THREAD(x, x, q)
{
for (int z = 0; z < q; ++z)
{
real_t curr[3];
curr[0] = cst ? cst_val_0 : C(0, x, y, z, e);
curr[1] = cst ? cst_val_1 : C(1, x, y, z, e);
curr[2] = cst ? cst_val_2 : C(2, x, y, z, e);
const real_t J11 = J(x, y, z, 0, 0, e);
const real_t J21 = J(x, y, z, 1, 0, e);
const real_t J31 = J(x, y, z, 2, 0, e);
const real_t J12 = J(x, y, z, 0, 1, e);
const real_t J22 = J(x, y, z, 1, 1, e);
const real_t J32 = J(x, y, z, 2, 1, e);
const real_t J13 = J(x, y, z, 0, 2, e);
const real_t J23 = J(x, y, z, 1, 2, e);
const real_t J33 = J(x, y, z, 2, 2, e);
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
const real_t A[9] = {A11, A12, A13, A21, A22,
A23, A31, A32, A33
};
QQQ(x, y, z, vd) = W(x, y, z) * (A[vd * vdim] * curr[0] +
A[vd * vdim + 1] * curr[1] +
A[vd * vdim + 2] * curr[2]);
}
}
}
}
MFEM_SYNC_THREAD;
// Apply Bt operator
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
DeviceMatrix Btx = (vd == 0) ? Bot : Bct;
MFEM_FOREACH_THREAD(qy, y, q)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qx = 0; qx < q; ++qx)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += QQQ(qx, qy, qz, vd) * Btx(dx, qx);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
DQQ(dx, qy, qz, vd) = u[qz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
const int ny = (vd == 1) ? d - 1 : d;
DeviceMatrix Bty = (vd == 1) ? Bot : Bct;
MFEM_FOREACH_THREAD(dy, y, ny)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[Q];
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qy = 0; qy < q; ++qy)
{
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
u[qz] += DQQ(dx, qy, qz, vd) * Bty(dy, qy);
}
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
DDQ(dx, dy, qz, vd) = u[qz];
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(vd, z, vdim)
{
const int nx = (vd == 0) ? d - 1 : d;
const int ny = (vd == 1) ? d - 1 : d;
const int nz = (vd == 2) ? d - 1 : d;
DeviceTensor<5> Yxyz(Y, nx, ny, nz, vdim, ne);
DeviceMatrix Btz = (vd == 2) ? Bot : Bct;
MFEM_FOREACH_THREAD(dy, y, ny)
{
MFEM_FOREACH_THREAD(dx, x, nx)
{
real_t u[D];
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] = 0.0;
}
MFEM_UNROLL(Q)
for (int qz = 0; qz < q; ++qz)
{
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
u[dz] += DDQ(dx, dy, qz, vd) * Btz(dz, qz);
}
}
MFEM_UNROLL(D)
for (int dz = 0; dz < nz; ++dz)
{
Yxyz(dx, dy, dz, vd, e) += u[dz];
}
}
}
}
MFEM_SYNC_THREAD;
});
}
template <FiniteElement::DerivType TestType, int DIM, int TEST_D1D, int Q1D>
VectorFEDomainLFIntegrator::AssembleKernelType
VectorFEDomainLFIntegrator::AssembleKernels::Kernel()
{
if constexpr (TestType == FiniteElement::DIV)
{
if constexpr (DIM == 2)
{
return HdivDLFAssemble2D<TEST_D1D, Q1D>;
}
if constexpr (DIM == 3)
{
return HdivDLFAssemble3D<TEST_D1D, Q1D>;
}
}
if constexpr (TestType == FiniteElement::CURL)
{
if constexpr (DIM == 3)
{
return HcurlDLFAssemble3D<TEST_D1D, Q1D>;
}
}
MFEM_ABORT("");
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem

Some files were not shown because too many files have changed in this diff Show More