Compare commits

...
Author SHA1 Message Date
Stowell, Mark L. c968516f36 Merge branch 'complex-coef-dev' of github.com:mfem/mfem into complex-coef-dev 2025-07-22 10:46:29 -07:00
Stowell, Mark L. 9a9d1ea967 Switching to complex_t 2025-07-22 10:45:56 -07:00
Stowell, Mark L. 46ee2ab5dc Adding complex_t type using code by @camierjs 2025-07-22 10:43:59 -07:00
adam-sim-dev fb672667cc Merge branch 'master' into complex-coef-dev 2025-07-15 09:22:34 +08:00
Stowell, Mark L. 1d35fafd21 Removing complex<int> unit test due to a "Build Analysis" error 2025-07-14 13:44:17 -07:00
Stowell, Mark L. a0ed1bfbca Adding ComplexVector unit test 2025-07-14 13:18:58 -07:00
Stowell, Mark L. 0c08279225 Adding coefficient as template argument in SesquilinearForm 2025-07-14 10:40:53 -07:00
Stowell, Mark L. bc84ce3b47 Adding ComplexMatrixConstantCoefficient 2025-07-14 10:40:02 -07:00
Stowell, Mark L. 36abe386e0 Adding DenseMatrix constructor to StdComplexDenseMatrix 2025-07-14 10:39:31 -07:00
Stowell, Mark L. a4868f2a98 Adding ComplexMatrixCoefficient 2025-07-12 16:05:19 -07:00
Stowell, Mark L. 21321b3abc Adding new type of complex dense matrix 2025-07-12 16:05:03 -07:00
Stowell, Mark L. bb4f39c3d7 Avoiding circular dependency 2025-07-12 11:49:55 -07:00
Stowell, Mark L. b605a29988 Making Real/Imag part coefficients publicly available 2025-07-12 11:49:35 -07:00
Stowell, Mark L. 7967e13f1d Moving complex coefficient code into separate files 2025-07-12 10:30:00 -07:00
Stowell, Mark L. b615f22b66 Adding complex coefficient support to ComplexLinearForm classes 2025-07-12 10:20:34 -07:00
Stowell, Mark L. e0fe515f21 Adding ComplexCoefficient methods to ComplexLinearForm classes 2025-07-12 07:42:11 -07:00
Stowell, Mark L. 269ee766db Updating ex22p 2025-07-11 17:14:17 -07:00
Stowell, Mark L. da8a221097 Adding complex coefficients to parallel classes 2025-07-11 17:14:04 -07:00
Stowell, Mark L. 20d6e63df0 Ubuntu portability 2025-07-11 14:48:12 -07:00
Stowell, Mark L. 2288cdcb7f Ubuntu portability 2025-07-11 14:32:25 -07:00
Stowell, Mark L. 233337c9d1 Adding UseDevice to extraction of real and imaginary parts of ComplexVector 2025-07-11 14:26:26 -07:00
Stowell, Mark L. 4d9cd853b7 Fix for Ubuntu portability 2025-07-11 14:18:25 -07:00
Stowell, Mark L. 27352658c3 Using the new complex coefficient classes in ex22 2025-07-11 14:00:22 -07:00
Stowell, Mark L. b8a303a07a Adding Complex coefficient classes 2025-07-11 14:00:02 -07:00
Stowell, Mark L. cee9bf3bb2 Adding a ComplexVector class 2025-07-11 13:59:23 -07:00
Tzanio Kolev a901754de5 Merge pull request #4841 from mfem/nurbs-surf
NURBS surface interpolation minapp
2025-07-11 09:18:02 -07:00
Tzanio Kolev 03da41c0f5 Merge pull request #4924 from mfem/dfem-elem-restriction-fix
Multiple actions with one dfem `DifferentiableOperator`
2025-07-11 09:17:25 -07:00
Tzanio Kolev c6ec74db41 Merge pull request #4921 from mfem/submesh-small-simplification
Small simplification in `SubMeshUtils::AddBoundaryElements`
2025-07-10 08:55:28 -07:00
Eric B. Chin 8854247f86 Merge branch 'master' into dfem-elem-restriction-fix 2025-07-09 11:17:01 -07:00
Veselin Dobrev 755e4501e1 Merge pull request #4899 from mfem/ubsan
[Github] Sanitizers action
2025-07-08 23:02:02 -07:00
Veselin Dobrev 1ec73c3bf6 Merge branch 'master' into ubsan 2025-07-08 19:36:26 -07:00
E. B. Chin 3a15fe3d96 sum into residual_l 2025-07-08 15:31:36 -07:00
Tzanio Kolev 8b000dd222 Merge pull request #4898 from mfem/dfem-bugfixes
Fix UB in dFEM
2025-07-06 10:51:12 -07:00
Tzanio Kolev a4d6889332 Merge pull request #4913 from mfem/tmop-memory-warning-fix
TMOP uninitialized memory warning fix
2025-07-06 10:50:38 -07:00
Tzanio Kolev e0fbc5e3aa Merge branch 'master' into nurbs-surf 2025-07-06 10:50:17 -07:00
Veselin Dobrev 03da9d7789 Small simplification in SubMeshUtils::AddBoundaryElements 2025-07-05 13:46:56 -07:00
John Camier 9c8874a38c Merge branch 'master' into ubsan 2025-07-03 16:06:57 -07:00
Tzanio Kolev bb2460cbd0 Merge branch 'master' into tmop-memory-warning-fix 2025-07-02 18:50:36 -07:00
Tzanio Kolev 0648e50e70 Merge pull request #4843 from mfem/netcdf-single
Write Exodus meshes with real_t instead of double
2025-07-02 11:45:02 -07:00
camierjs 45e8125fd6 Merge branch 'master' into ubsan 2025-07-02 07:33:56 -07:00
Tzanio Kolev 25056defeb Merge branch 'master' into nurbs-surf 2025-07-01 14:27:00 -07:00
Tzanio Kolev 20cd965ed8 Merge branch 'master' into tmop-memory-warning-fix 2025-07-01 12:54:18 -07:00
Tzanio Kolev 1fda9c2391 Merge branch 'master' into netcdf-single 2025-07-01 12:46:55 -07:00
Tzanio Kolev 2cec0353b1 Merge pull request #4757 from mfem/hip-unit-tests
Added general GPU and CUDA/HIP-specific unit tests
2025-07-01 12:46:18 -07:00
Tzanio Kolev 9d22775395 Merge pull request #4822 from mfem/gpu-thread-direct
Direct threadblock loops
2025-07-01 12:43:38 -07:00
Tzanio Kolev 94625fad8f Merge pull request #4870 from mfem/vtkhdf-chunk-fix
Improve chunking in VTKHDF writer
2025-07-01 12:42:45 -07:00
Tzanio Kolev c9a9c71ff5 Merge pull request #4866 from rfhaque/master
Decompose det.cpp into separate source and header files
2025-07-01 12:41:57 -07:00
Tzanio Kolev 764d9919b5 Merge pull request #4892 from mfem/hughcars/missing-host-read-write-fix
Fix missing HostReadWrite for Nodes
2025-07-01 12:40:40 -07:00
Tzanio Kolev 8efbd4e46f Merge branch 'master' into netcdf-single 2025-07-01 12:00:59 -07:00
John Camier b94ac358e4 Merge branch 'master' into hip-unit-tests 2025-07-01 09:51:43 -07:00
John Camier d4d4b79522 Merge branch 'master' into vtkhdf-chunk-fix 2025-07-01 09:51:12 -07:00
John Camier 120f4cb043 Merge branch 'master' into hughcars/missing-host-read-write-fix 2025-07-01 09:50:46 -07:00
camierjs 4f1597c1bc CHANGELOG: MFEM_FOREACH_THREAD_DIRECT in GPU computing 2025-06-30 18:16:56 -07:00
John Camier 56beedbdcb Merge branch 'master' into gpu-thread-direct 2025-06-30 18:06:23 -07:00
camierjs 520a9c5125 On branch: master, next & remove debug mode 2025-06-29 15:39:12 -07:00
camierjs ed521022cd Fine-grained sanitizer jobs 2025-06-29 15:17:44 -07:00
John Camier 7b35a47626 Merge branch 'master' into ubsan 2025-06-28 21:34:36 -07:00
Veselin Dobrev def35c8a15 In HypreParMatrix::EliminateBC, use stream synchronization instead of
device synchronization and do it only when hypre is using GPU-aware MPI.

Remove device synchronization before CUDA/HIP free which implicitly
perform the same synchronization.
2025-06-28 08:05:30 -07:00
Veselin Dobrev c53a016d06 Merge pull request #4818 from mfem/mesh-transform-dev
Adding an affine transformation to mesh-explorer
2025-06-28 07:58:55 -07:00
Veselin Dobrev ea2b42c49d Merge pull request #4903 from mfem/fix-umpire-introspection-off
better umpire device deallocation
2025-06-28 07:57:55 -07:00
John Camier 4e35b3d8f1 Merge branch 'master' into hughcars/missing-host-read-write-fix 2025-06-28 06:26:57 -07:00
John Camier c7e066a0d3 Merge branch 'master' into fix-umpire-introspection-off 2025-06-28 06:26:17 -07:00
Andrew Ho 29681677a1 Merge branch 'master' into hip-unit-tests 2025-06-27 23:01:04 -07:00
Mittal, Ketan 9d72af995f remove unneeded code 2025-06-27 12:50:27 -07:00
Mittal, Ketan b099252dcf fix 2025-06-27 11:50:06 -07:00
camierjs d696fc2cea Merge branch 'master' into ubsan 2025-06-26 17:23:44 -07:00
Andrew Ho 8dcd0d6349 add device syncs in EliminateBC
This should fix potential race conditions in some situations
2025-06-26 15:57:18 -07:00
Veselin Dobrev 2951d5f98e Fix false positives for testing errors in debug mode 2025-06-26 13:51:34 -07:00
Andrew Ho 4b27589abf missing hostread 2025-06-26 11:38:50 -07:00
Riyaz Haque 8338aa85e6 Merge branch 'master' into master 2025-06-26 10:33:31 -07:00
Veselin Dobrev 0a5730eac3 Merge pull request #4769 from mfem/dev/abs-diag-smoothers
Abs-Val-Jacobi-type of preconditions/smoothers for PA operators
2025-06-26 09:36:30 -07:00
Veselin Dobrev e20bb381ca In Device::Print, show the GPU-aware MPI configuration only when MPI is
being used.
2025-06-26 08:26:20 -07:00
camierjs 89e23a93f5 Re-wrap Miscellaneous at 80 characters 2025-06-26 06:51:30 -07:00
Veselin Dobrev 4816fa0849 Added the option to enable GPU-aware MPI in MFEM using the environment
variable 'MFEM_GPU_AWARE_MPI' set to any value. Setting this environment
variable is an alternative to calling 'Device::SetGPUAwareMPI(true)'.

In Device::Print, show the GPU-aware MPI usage status when using a device
backend.

Update some unit tests to better handle failures.
2025-06-25 11:47:31 -07:00
Riyaz Haque 6fc7a8ca5a Merge branch 'master' into master 2025-06-25 10:33:57 -07:00
Tzanio Kolev 22c8b607dc Merge branch 'master' into hughcars/missing-host-read-write-fix 2025-06-25 08:02:02 -07:00
camierjs 91deab3c75 CHANGELOG w/o blanks-around-headings 2025-06-25 05:48:39 -07:00
Tzanio Kolev 87cee25894 Merge branch 'master' into dfem-bugfixes 2025-06-25 03:58:42 -07:00
Tzanio Kolev 1bcddddfdb Merge branch 'master' into gpu-thread-direct 2025-06-25 03:41:32 -07:00
camierjs da68672955 Rename 'unit' to 'tests' 2025-06-24 17:12:07 -07:00
camierjs a5dd4b862b Rename matrix sanitizer 2025-06-24 17:05:33 -07:00
Veselin Dobrev a3fcb89049 Small doxygen and CHANGELOG updates 2025-06-24 16:59:38 -07:00
Veselin Dobrev bd6f3d51c8 Remove checks for c++17 which is now required 2025-06-24 16:05:47 -07:00
camierjs 8d95f71305 Merge branch 'master' into ubsan 2025-06-24 14:33:07 -07:00
camierjs aee0cb1dc6 Split tests: [unit, examples, miniapps] 2025-06-24 14:33:00 -07:00
Dylan Copeland 7dded1fdcf Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-06-24 13:28:06 -07:00
Veselin Dobrev 477c475c1a Merge pull request #4893 from mfem/gpu-omp
Use device priorities for vector dot, min and max
2025-06-24 12:33:16 -07:00
Veselin Dobrev 34fdcddb8d Merge pull request #4810 from mfem/plbound
Bounding high-order FEM functions
2025-06-24 12:31:47 -07:00
Veselin Dobrev 91fa9936ef Merge pull request #4835 from mfem/fix-nurbsext-constructor
NURBSExtension Constructor and edge_to_knot mapping
2025-06-24 12:29:14 -07:00
Veselin Dobrev d95e6e0adc Merge pull request #4854 from mfem/pref-mixed-serial
Restore p-refinement on serial mixed meshes
2025-06-24 12:27:52 -07:00
John Camier cc0dcaad09 Merge branch 'master' into vtkhdf-chunk-fix 2025-06-23 16:11:40 -07:00
John Camier 82d3b8b92e Merge branch 'master' into gpu-omp 2025-06-23 16:11:10 -07:00
John Camier 960379a43d Merge branch 'master' into ubsan 2025-06-23 16:11:00 -07:00
Hugh Carson 3782ece6b3 Merge branch 'master' into hughcars/missing-host-read-write-fix 2025-06-23 16:14:44 -04:00
camierjs 823c7a952d Asan strdup for Hypre 2.19.0 2025-06-23 09:03:27 -07:00
camierjs 2bc553633e MFEM ex1p ctest 2025-06-23 08:37:47 -07:00
camierjs 8be00d1115 Reuse actions Hypre & Metis caches 2025-06-23 07:45:57 -07:00
Riyaz Haque bdd2bfbb79 Merge with master 2025-06-22 22:35:34 -07:00
John Camier 1ea4cd14da Merge branch 'master' into hip-unit-tests 2025-06-22 20:26:16 -07:00
John Camier 63b72c4153 Merge branch 'master' into gpu-thread-direct 2025-06-22 20:26:08 -07:00
John Camier 054593bd4d Merge branch 'master' into vtkhdf-chunk-fix 2025-06-22 20:25:57 -07:00
John Camier 9c90e3830a Merge branch 'master' into gpu-omp 2025-06-22 20:25:27 -07:00
John Camier a8e83d2f0d Merge branch 'master' into fix-umpire-introspection-off 2025-06-22 20:23:40 -07:00
camierjs 298a4bc32a Avoid CopyFrom in DeviceConformingProlongationOperator 2025-06-22 18:18:03 -07:00
camierjs 65866edd70 Force MPICXX 2025-06-22 13:52:40 -07:00
camierjs 85b96208f0 Env CTEST fix 2025-06-22 13:22:40 -07:00
camierjs b825a46061 CTEST fix 2025-06-22 13:20:11 -07:00
camierjs d6cbd4f99a ctests options fix 2025-06-22 13:06:35 -07:00
camierjs c691658232 Switch to ctests 2025-06-22 13:05:08 -07:00
camierjs 157ffad537 Simplify, meld back & ninja default nproc 2025-06-22 12:07:07 -07:00
camierjs cfdb7d2a03 Cleanup 2025-06-22 11:10:43 -07:00
camierjs a95a2dc251 Ninja verbose builds & jobs 2025-06-22 10:12:51 -07:00
camierjs a7737e65ab MPI_LIB to LDFLAGS 2025-06-22 09:56:53 -07:00
camierjs dda669bd70 Cleanup & ninja tests/unit/test 2025-06-22 09:54:25 -07:00
camierjs 802345aa91 config-options force CMAKE_CXX_FLAGS_RELEASE 2025-06-22 09:21:59 -07:00
camierjs 1e84d8a9c5 config-options redundant FLAGS 2025-06-22 09:19:43 -07:00
camierjs ce30630f5f config-options to Release 2025-06-22 09:14:21 -07:00
camierjs 0055de1734 yamllint fix 2025-06-22 09:10:28 -07:00
camierjs 954757f7de Ninja test 2025-06-22 09:08:53 -07:00
camierjs 7ad1da790c config-options strip new line 2025-06-22 08:33:11 -07:00
camierjs 4290365459 CMAKE_CXX_FLAGS escapes 2025-06-22 08:25:07 -07:00
camierjs 04d6fa900d CMAKE_CXX_FLAGS escapes 2025-06-22 08:23:42 -07:00
camierjs 9fec57261b CMAKE_CXX_FLAGS escapes 2025-06-22 08:20:48 -07:00
camierjs 9e8a710c92 CLANG_VER 2025-06-22 08:18:26 -07:00
camierjs c5d2f364ac Clang Local ENV 2025-06-22 08:09:04 -07:00
camierjs fca154fbd4 Clang Local 2025-06-22 08:07:56 -07:00
camierjs 804bdb498a CMAKE_EXE_LINKER_FLAGS 2025-06-22 08:06:48 -07:00
camierjs 1546398ee8 Clang Local 2025-06-22 08:04:30 -07:00
camierjs 9237b4cc2e config-options escapes 2025-06-22 08:00:49 -07:00
camierjs a02927005d config-options fix 2025-06-22 07:59:34 -07:00
camierjs 6256741216 config-options addons 2025-06-22 07:41:04 -07:00
camierjs a5fcfa02e8 CMake config-options 2025-06-22 07:28:40 -07:00
camierjs 2fc9b97cb4 .github/workflows/mfem-sanitizers.yml Hypre version 2.19.0 2025-06-22 07:25:01 -07:00
camierjs a5ca81806b .github/workflows/mfem-sanitizers.yml CMake try 2025-06-22 07:19:01 -07:00
camierjs d4513550f9 [asan] miniapps/solvers/bramble_pasciak fix 2025-06-21 20:33:10 -07:00
camierjs d1befb2ea6 yaml lint & re-enable examples & miniapps 2025-06-21 15:33:15 -07:00
camierjs e11a093e72 Setup DofToQuad information mode 2025-06-21 14:36:25 -07:00
camierjs 943234617b hypre-dir metis-dir fix 2025-06-21 13:41:21 -07:00
camierjs 36389366ac Use dirs 2025-06-21 13:19:04 -07:00
camierjs dbd2b5556b Fix ASan suppression file 2025-06-21 13:12:32 -07:00
Gabriel Pinochet-Soto 288657ebbd Use SLI + Mass integrator as example 2025-06-21 13:11:25 -07:00
camierjs 6570ca9c7a mkdir ASAN_DIR 2025-06-21 13:09:28 -07:00
camierjs 71943e120c Split GITHUB_ENV setup 2025-06-21 13:06:03 -07:00
Gabriel Pinochet-Soto 5a26bd936c Merge branch 'master' of github.com:mfem/mfem into dev/abs-diag-smoothers 2025-06-21 13:05:49 -07:00
camierjs ef65351cc9 Use LLVM_DIR 2025-06-21 12:34:36 -07:00
camierjs 5a7807055b GITHUB_WORKSPACE 2025-06-21 12:33:47 -07:00
camierjs 08b46a12dc LLVM_DIR w/o env 2025-06-21 12:32:55 -07:00
camierjs 78124a649d env fix 2025-06-21 12:28:57 -07:00
camierjs 0e9d10c53d LLVM_DIR fix 2025-06-21 12:27:55 -07:00
camierjs 911eb07565 Cleanup workflows mfem-sanitizers.yml 2025-06-21 12:26:52 -07:00
camierjs 2efec6390b Fix mfem-sanitizers.yml env (bis) 2025-06-21 12:08:21 -07:00
camierjs 756bc524ab Fix mfem-sanitizers.yml env 2025-06-21 12:06:01 -07:00
camierjs bade193d79 [asan] tests/unit/mesh/test_ncmesh: TetMemory MFEM_USE_MEMALLOC=OFF 2025-06-21 11:56:40 -07:00
camierjs 4363cd2dc2 [asan] tests/unit/linalg/test_hypre_prec 2025-06-21 10:06:20 -07:00
camierjs cbbe609ff8 Merge branch 'master' into ubsan 2025-06-21 08:51:08 -07:00
camierjs 23caac9573 Cleanup 2025-06-21 08:51:01 -07:00
camierjs 3eb0e321f4 [asan] tests/unit/fem/test_var_order.cpp 2025-06-20 20:38:11 -07:00
camierjs 58c276b1d9 Fix HYPRE_TOP_DIR 2025-06-20 17:45:30 -07:00
camierjs b81c67a061 Switch to LLVM 19.1.7 2025-06-20 17:43:57 -07:00
camierjs fbd6aa17e0 Avoid examples 2025-06-20 17:03:16 -07:00
camierjs 199192c0f6 Bump LLVM and HYPRE versions 2025-06-20 17:02:37 -07:00
Gabriel Pinochet-Soto 05034b8917 Correct examples; vis is true 2025-06-20 16:58:45 -07:00
camierjs 27989c68bd [ubsan] miniapps/solvers/bramble_pasciak seed 2025-06-20 16:57:31 -07:00
camierjs 8ba6b88ee0 CHANGELOG Miscellaneous update 2025-06-20 14:54:12 -07:00
camierjs df03268c01 [asan] miniapps/solvers/block-solvers and div_free_solver leaks 2025-06-20 14:45:08 -07:00
Tom Stitt 32aea6ed2d deallocate via the allocator instead of the resource manager. fixes use of device allocators with introspection off 2025-06-20 14:15:31 -07:00
camierjs 7e7322cb86 WIP non coupled DivFreeSolver 2025-06-20 13:43:22 -07:00
Dylan Copeland d8b9c7881b Transpose argument for banded factorization. 2025-06-20 10:22:47 -07:00
Dylan Copeland 58f0e28453 Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-06-20 10:13:14 -07:00
camierjs 9d48f23dca [asan] miniapps/solvers/bramble_pasciak.cpp leaks 2025-06-20 09:33:22 -07:00
camierjs 1947748db6 [undefined] miniapps/spde/generate_random_field seed 2025-06-20 06:39:41 -07:00
camierjs 7aca441524 Exclude MemorySanitizer w/ MPI 2025-06-19 20:55:34 -07:00
camierjs 689522beb5 Fix parallel UB and Memory 2025-06-19 20:33:31 -07:00
camierjs c0254e3835 Hardcoded asan.supp path 2025-06-19 19:27:42 -07:00
camierjs 85e1855d1b ASan suppression file path fix 2025-06-19 19:02:12 -07:00
camierjs 51eb3c0026 GITHUB_WORKSPACE asan.supp 2025-06-19 18:59:52 -07:00
camierjs c1c348f24f Add ASan suppression file 2025-06-19 18:53:58 -07:00
camierjs 043c6ee860 LLVM_LDFLAGS for get_hypre_version 2025-06-19 18:34:26 -07:00
camierjs 790b4b9bce MPI_LIB for get_hypre_version 2025-06-19 18:24:14 -07:00
camierjs 69aea4ce75 MPI_INC/LIB 2025-06-19 18:10:23 -07:00
camierjs d7d173a215 MPICXX to c++ 2025-06-19 18:02:19 -07:00
camierjs d7ac021db2 MPI workflow debug, env and compiler 2025-06-19 17:54:17 -07:00
camierjs 2b1d4eb30c Parallel sanitizer action 2025-06-19 17:02:26 -07:00
camierjs c134457322 Revert md style for CHANGELOG 2025-06-19 15:50:00 -07:00
camierjs 33bd60f7c7 Updates changelog with sanitizer details 2025-06-19 15:27:38 -07:00
camierjs 8f3e61a4d3 Meld back to master 2025-06-19 11:02:07 -07:00
camierjs 351806ce77 [MemorySanitizer] use-of-uninitialized-value InvTNewtonSolverBase 2025-06-19 10:59:27 -07:00
camierjs d865ac444e [undefined] mesh_readers ReadHeaderEntry 2025-06-19 10:51:25 -07:00
camierjs b14a9b4663 Meld back toward master 2025-06-19 09:42:35 -07:00
camierjs cdd560e6a2 [undefined] lissajous null pointer socketstream
Meld back toward master
2025-06-19 09:23:57 -07:00
camierjs 4c9ddf93e4 [undefined] KnotVector fix 2025-06-19 08:31:46 -07:00
camierjs 8410c205a3 [undefined] test_sedov DeltaCoefficient SetWeight 2025-06-19 07:48:01 -07:00
camierjs f714dcfd57 Hcurl/Hdiv PA Coefficient coeff2 fix 2025-06-19 07:20:12 -07:00
camierjs 53fb6e3977 [MemorySanitizer] TestFDCalcDivShape dim 2 pt.z 2025-06-19 07:06:10 -07:00
Gabriel Pinochet-Soto 9407050e6a Revert "Update to use maps->Abs"
This reverts commit 0515209ffd.
2025-06-18 23:09:29 -07:00
Gabriel Pinochet-Soto 0515209ffd Update to use maps->Abs 2025-06-18 20:05:11 -07:00
Gabriel Pinochet-Soto 3446b46420 Merge branch 'master' of github.com:mfem/mfem into dev/abs-diag-smoothers 2025-06-18 19:37:54 -07:00
Gabriel Pinochet-Soto f8f18f8722 Remove warning 2025-06-18 19:33:37 -07:00
Gabriel Pinochet-Soto e7eabeb5e2 Remove TODOs bilinearform_ext.cpp 2025-06-18 18:55:22 -07:00
camierjs 678a9db016 [MemorySanitizer] TestCalcDivShape dim 2 pt.z
Avoid overflow in dot product test
2025-06-18 16:46:10 -07:00
camierjs c09246351d [MemorySanitizer] dim 2 pt.z 2025-06-18 16:16:45 -07:00
camierjs 86a8d39e54 Cleanup and allow all tests 2025-06-18 15:23:22 -07:00
camierjs 54854b0908 Cleanup, LLVM flags w/o AddressSanitizer 2025-06-18 15:06:24 -07:00
camierjs 8fa48c2425 Cleanup LLVM LIB & INC 2025-06-18 14:45:12 -07:00
camierjs 3e9b8605f9 llvm-project/runtimes fix 2025-06-18 14:18:17 -07:00
camierjs d84aa5a355 Cleanup 2025-06-18 14:12:49 -07:00
camierjs c11c5cf654 Run LLVM Clone 2025-06-18 13:58:33 -07:00
camierjs 1c8d25c6ed LLVM libcxx Build Steps 2025-06-18 13:54:49 -07:00
camierjs be6d5e2b01 memory_sanitizer 2025-06-18 13:30:38 -07:00
camierjs 8abdc6500e MemorySanitizer code fix 2025-06-18 13:18:18 -07:00
camierjs 21df84f320 address_sanitizer 2025-06-18 13:03:02 -07:00
camierjs 9f5860fda2 Add LLVM_VERSION and ex1 trigger tests 2025-06-18 12:53:48 -07:00
camierjs c5a03405cf Rename mfem-sanitizers 2025-06-18 12:20:03 -07:00
camierjs 33d0d7dbfe Fix stdlib 2025-06-18 12:19:12 -07:00
camierjs c085ec6544 Setup LLVM_SANITIZER 2025-06-18 12:14:24 -07:00
camierjs 1f860fbfaf Cleanup GITHUB_WORKSPACE 2025-06-18 12:01:09 -07:00
camierjs c43420c375 MFEM Checkout 2025-06-18 11:57:42 -07:00
camierjs ecbdd73c54 Dump workspaces 2025-06-18 11:53:23 -07:00
camierjs d30e13c543 github.workspace paths 2025-06-18 11:50:17 -07:00
camierjs ba2b8aea52 Build libc++ 2025-06-18 11:39:06 -07:00
camierjs f1353bd6e9 Change mfem-sanitizers.sh path 2025-06-18 11:33:51 -07:00
camierjs 3cb0bee255 Fix both uses and run keys 2025-06-18 10:46:49 -07:00
camierjs bda0b9aba5 Fix github mfem-sanitizers paths 2025-06-18 10:43:21 -07:00
camierjs e85af79c16 Try MemorySanitizer 2025-06-18 10:39:50 -07:00
John Camier 06c4da64af Merge branch 'master' into hip-unit-tests 2025-06-18 10:22:55 -07:00
John Camier b841c9df71 Merge branch 'master' into gpu-thread-direct 2025-06-18 10:22:43 -07:00
John Camier 31a6329964 Merge branch 'master' into vtkhdf-chunk-fix 2025-06-18 10:22:24 -07:00
John Camier 57b5d23e4b Merge branch 'master' into gpu-omp 2025-06-18 10:21:40 -07:00
camierjs 9f83167010 make test 2025-06-18 10:11:32 -07:00
camierjs 2b4085e2dd Try ubuntu-latest & Setup clang 2025-06-18 10:01:06 -07:00
camierjs a69ea6d698 Merge branch 'master' into ubsan 2025-06-18 08:04:59 -07:00
camierjs a617205ee1 Revert check & make test 2025-06-17 21:11:03 -07:00
camierjs 620f906765 make check 2025-06-17 21:02:20 -07:00
camierjs 2620effa65 Avoid optimizations 2025-06-17 21:00:45 -07:00
camierjs 3da9bdc39a global env OPTIONS 2025-06-17 20:46:18 -07:00
camierjs 2ab00899fd Merge branch 'master' into ubsan 2025-06-17 20:23:38 -07:00
camierjs 7009449ef5 make check 2025-06-17 20:22:57 -07:00
camierjs 34482860b0 ASAN_OPTIONS w/o spaces 2025-06-17 20:22:18 -07:00
camierjs 8f4aafdebc matrix env make test 2025-06-17 19:00:01 -07:00
camierjs 26eea4f2fd mfem-sanitizers run tweak 2025-06-17 18:20:14 -07:00
camierjs 313f6856e8 Remove var-tracking-assignments 2025-06-17 17:58:37 -07:00
camierjs bf97e92be2 Use config-options for CXX 2025-06-17 17:57:38 -07:00
camierjs d7a04ca4ac Adjust options and remove symbolizer 2025-06-17 17:29:18 -07:00
camierjs 28b1ac0c9d Try mfem-sanitizers.yml style 2025-06-17 17:18:34 -07:00
John Camier 4f3c73b3eb Merge branch 'master' into gpu-omp 2025-06-17 17:01:31 -07:00
camierjs 0eb2d21602 Remove from MemorySanitizer 'include' 2025-06-17 16:35:15 -07:00
camierjs efcf608a5d Remove MemorySanitizer 2025-06-17 16:32:20 -07:00
camierjs f1d56c4068 env matrix 2025-06-17 16:23:45 -07:00
camierjs 94698f27d7 Remove matrix name 2025-06-17 16:19:27 -07:00
camierjs 30dac8986c Sanitizers matrix 2025-06-17 16:18:13 -07:00
John Camier 0691354c84 Update mfem-sanitizers.yml matrix 2025-06-17 16:01:09 -07:00
camierjs 5c0b2a6b62 Add LDFLAGS for all runs
Fix float-conversion warnings
2025-06-17 15:43:41 -07:00
camierjs c25b84fbd8 Add undefined behavior and uninitialized memory use detectors 2025-06-17 14:52:43 -07:00
Mittal, Ketan 8be13975b0 Merge branch 'plbound' of https://github.com/mfem/mfem into plbound 2025-06-17 11:49:40 -07:00
Mittal, Ketan 7fce328f12 fix spacing before sample run 2025-06-17 11:43:42 -07:00
Tzanio Kolev d6c1edc9a3 Merge branch 'master' into netcdf-single 2025-06-17 08:14:28 -07:00
Veselin Dobrev 56b1a1715a Use DofToQuad::Abs in a few places. Remove unused methods AddAbsMult*.
Small doxygen tweaks.
2025-06-17 01:16:34 -07:00
Veselin Dobrev 3514c0f0d4 Merge branch 'master' into dev/abs-diag-smoothers
Resolved conflicts:
   CHANGELOG
   makefile
   miniapps/CMakeLists.txt
2025-06-16 22:14:00 -07:00
John Camier 74498373c9 Merge branch 'master' into gpu-thread-direct 2025-06-16 08:42:17 -07:00
John Camier 1cc3d81866 Merge branch 'master' into vtkhdf-chunk-fix 2025-06-16 08:33:56 -07:00
Julian Andrej 233316269a ub fix attempt 2025-06-16 07:50:14 -07:00
Andrew Ho e34c6b6013 Merge branch 'master' into hip-unit-tests 2025-06-14 16:22:47 -07:00
Dylan Copeland ea593def25 Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-06-13 14:15:39 -07:00
camierjs 5791aa4629 Merge branch 'master' into gpu-omp 2025-06-13 13:43:40 -07:00
camierjs 18cff41dac Merge branch 'master' into gpu-omp 2025-06-13 11:51:56 -07:00
Dylan Copeland f316ec7d5e Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-06-13 11:08:05 -07:00
Tzanio Kolev b5366e3ad3 Merge branch 'master' into dev/abs-diag-smoothers 2025-06-13 10:56:52 -07:00
John Camier 800be7971a Merge branch 'master' into hip-unit-tests 2025-06-13 08:49:35 -07:00
John Camier ee06c0eb44 Merge branch 'master' into gpu-thread-direct 2025-06-13 08:49:18 -07:00
camierjs fb9449c47d Adjust use_dev code path, make style & const 2025-06-13 08:31:55 -07:00
camierjs 1555bfe3a5 Move reduction internals to reducers header 2025-06-12 11:56:50 -07:00
camierjs a3b12b6f97 Use device priorities for vector dot, min and max 2025-06-12 11:27:06 -07:00
Hugh Carson f0e8e6e19a Add in missing HostReadWrite in DoNodeReorder for operator() usage 2025-06-12 11:38:00 -04:00
Tzanio Kolev c82d9bd8a0 Merge branch 'master' into dev/abs-diag-smoothers 2025-06-11 20:09:17 -07:00
Tzanio Kolev 1e6ee60790 Merge branch 'master' into dev/abs-diag-smoothers 2025-06-11 10:14:46 -07:00
Veselin Dobrev b7ff8749a3 Fix out-of-source testing with GNU make.
Adjust a tolerance in miniapps/nurbs/nurbs_solenoidal.cpp for macOS.

Fix typos in the miniapps/nurbs/makefile in the nurbs_solenoidal tests.

Re-formatting some long lines.
2025-06-09 18:19:09 -07:00
dylan-copeland ff11a6b572 Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-06-09 13:03:05 -07:00
Ketan Mittalandtarikdzanic a72d85fa6c Update fem/bounds.cpp
Co-authored-by: tarikdzanic <41964501+tarikdzanic@users.noreply.github.com>
2025-06-03 16:47:49 -07:00
Mittal, Ketan 0fd9dcceb3 change some raw ptrs to unique ptrs 2025-06-03 12:55:04 -07:00
Andrew Ho a8bbdf4fd4 Merge branch 'master' into hip-unit-tests 2025-06-02 11:33:00 -07:00
Mittal, Ketan b66d56ee9b Merge branch 'plbound' of https://github.com/mfem/mfem into plbound 2025-06-02 10:35:55 -07:00
Mittal, Ketan 9852e93449 reviewer comment 2025-06-02 10:35:46 -07:00
Will Pazner 2421b48f56 Enable compression by default in ParaViewHDFDataCollection 2025-05-30 07:19:55 -07:00
Will Pazner 2cd2d11215 In VTKHDF, avoid resizing on initial dataset creation 2025-05-30 07:19:28 -07:00
Ketan Mittal 27db3a4121 Merge branch 'master' into plbound 2025-05-29 10:49:30 -07:00
John Camier e11d19e3a9 Merge branch 'master' into gpu-thread-direct 2025-05-29 07:43:53 -07:00
Will PaznerandJohn Camier b0c30784b6 Address some implicit conversion warnings
Co-authored-by: John Camier <camierjs@gmail.com>
2025-05-28 20:42:16 -07:00
Gabriel Pinochet-Soto a35ef68c2a Style correction... 2025-05-28 11:35:12 -07:00
Gabriel Pinochet-Soto 2fca844393 Implement L(p,q) elementwise, update CHANGELOG 2025-05-28 11:32:50 -07:00
John Camier a7aa6c5a7c Merge branch 'master' into vtkhdf-chunk-fix 2025-05-28 11:15:42 -07:00
Gabriel Pinochet-Soto 698183a8db Correct examples 2025-05-28 10:42:20 -07:00
Riyaz Haque f6c2f10dee Merge branch 'master' into master 2025-05-28 07:47:25 -07:00
Gabriel Pinochet-Soto fa61508248 Merge branch 'master' of github.com:mfem/mfem into dev/abs-diag-smoothers 2025-05-27 22:44:20 -07:00
Gabriel Pinochet-Soto d2baadad26 Update descriptions, add --device cuda example 2025-05-27 17:30:03 -07:00
Mittal, Ketan 0aa68f431f Merge branch 'master' of https://github.com/mfem/mfem into plbound 2025-05-27 11:58:58 -07:00
Mittal, Ketan d2288ef6fd another reviewer comment 2025-05-27 11:58:34 -07:00
Ketan MittalandWill Pazner 6f882ed87e Update fem/pgridfunc.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-27 11:23:26 -07:00
Mittal, Ketan 813a7de323 reviewer comments 2025-05-27 11:18:28 -07:00
Justin Laughlin cc4f5625e2 Merge branch 'master' into fix-nurbsext-constructor 2025-05-27 09:27:59 -07:00
John Camier 5dcf6f7ead Merge branch 'master' into gpu-thread-direct 2025-05-26 20:39:03 -07:00
Mittal, Ketan f4baef8b5d fix shadow declaration 2025-05-26 17:06:59 -07:00
Mittal, Ketan 0e4b0f7bda minor 2025-05-26 16:44:40 -07:00
Mittal, Ketan 7317d4d139 minor 2025-05-26 16:44:28 -07:00
John Camier e8ed1a4c02 Merge branch 'master' into vtkhdf-chunk-fix 2025-05-26 16:11:23 -07:00
Mittal, Ketan e9e684599f Merge branch 'plbound' of https://github.com/mfem/mfem into plbound 2025-05-26 14:12:44 -07:00
Mittal, Ketan 6c994fea99 minor 2025-05-26 14:12:31 -07:00
Gabriel Pinochet-Soto 3b05995fd1 Update some tests 2025-05-26 08:36:18 -07:00
Riyaz Haque 3cfca882af Merge remote-tracking branch 'CEED/master' 2025-05-26 07:40:01 -07:00
Gabriel Pinochet-Soto 5e99ffc9a3 Relabel PC enum, other typos 2025-05-24 22:23:44 -07:00
Gabriel Pinochet-Soto bcbb24dce0 Remove redundant LEGACYFULL option 2025-05-24 19:06:51 -07:00
Gabriel Pinochet-Soto 797113ff71 Remove MG for HCurl 2025-05-24 18:53:44 -07:00
Gabriel Pinochet-Soto 41af83b10c Remove elast from miniapp 2025-05-24 18:42:22 -07:00
Gabriel Pinochet-Soto 4da2ea6f08 Remove abs elasticity integrators from source code 2025-05-24 18:41:43 -07:00
Gabriel Pinochet-Soto 9eb0f5c0c3 Merge branch 'master' of github.com:mfem/mfem into dev/abs-diag-smoothers 2025-05-24 16:06:51 -07:00
Ketan Mittal 4a1a5dfa55 Merge branch 'master' into plbound 2025-05-23 15:58:16 -07:00
Mittal, Ketan 59edd7255c separate out PLBound from gridfunc.hpp 2025-05-23 15:56:46 -07:00
Mittal, Ketan 6d01e152de reviewer comments 2025-05-23 12:26:06 -07:00
Will Pazner 70cb8fcc04 Fix shadow warning in VTKHDF 2025-05-23 09:13:05 -07:00
Will Pazner 0366ad2468 Improve chunking in VTKHDF writer
The data arrays grow only in the first dimension, and their other
dimensions are fixed. Therefore, the chunk size is those dimensions
should be equal to the dataset size. This can greatly reduce the
size of saved datasets.
2025-05-23 09:12:56 -07:00
Will Pazner 0e22b182a6 Rename MultUnsigned to AbsMult in restriction classes
Deprecate 'MultUnsigned' and related functions
2025-05-22 10:49:28 -07:00
Justin Laughlin 49eb2715ba typos 2025-05-21 18:26:38 -07:00
Justin Laughlin 5d3be16590 Fix check for empty nodes/weights 2025-05-21 17:05:17 -07:00
Justin Laughlin 899a2fc7cb Fix typos, improve documentation of new nurbs meshes, add size != 0 check in unit test 2025-05-21 13:55:03 -07:00
Ketan MittalandWill Pazner bd0fa51539 Update fem/pgridfunc.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-21 13:40:59 -07:00
Ketan MittalandWill Pazner b151c909f3 Update miniapps/tools/gridfunction-bounds.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-21 13:40:29 -07:00
Ketan MittalandWill Pazner ce7f94ec1f Update miniapps/tools/gridfunction-bounds.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-21 13:40:18 -07:00
Ketan MittalandWill Pazner ee6b9fdc2f Update miniapps/meshing/mesh-bounding-boxes.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-21 13:40:06 -07:00
Justin Laughlin c418868c83 Fix typo 2025-05-21 11:24:24 -07:00
Mark L. Stowell 7a0344a6bb Merge branch 'master' into mesh-transform-dev 2025-05-21 10:19:18 -07:00
Justin Laughlin 51812480bd Merge branch 'master' into fix-nurbsext-constructor 2025-05-20 15:36:39 -07:00
Justin Laughlin 61cc19ca22 One final cleanup 2025-05-20 15:34:57 -07:00
Justin Laughlin 5e48080f3d Minor cleanup/comments 2025-05-20 15:24:42 -07:00
Justin Laughlin af73851cb2 Comment 2025-05-20 15:13:28 -07:00
Justin Laughlin 338e4288ca Add some comments and update nurbs getters 2025-05-20 15:04:58 -07:00
Justin Laughlin dca4cd510a Add inverse map rpkv_to_ukv for efficiency 2025-05-20 13:16:25 -07:00
Justin Laughlin 2b23f35ec7 Copy miniapps/nurbs/meshes with cmake 2025-05-20 12:48:26 -07:00
Justin Laughlin ea569c5806 Add documentation and fix name 2025-05-20 11:20:34 -07:00
Riyaz Haque a355f28eae Decompose det.cpp into separate source and header files 2025-05-19 19:36:05 -07:00
Tzanio Kolev e2efc259b6 Merge branch 'master' into pref-mixed-serial 2025-05-19 16:51:16 -07:00
John Camier a75b1ca9c0 Merge branch 'master' into hip-unit-tests 2025-05-19 15:32:41 -07:00
John Camier 8ff6d69f74 Merge branch 'master' into gpu-thread-direct 2025-05-19 15:32:18 -07:00
Tzanio Kolev b1468c14fd Merge pull request #4862 from mfem/dev/abs-diag-smoothers-mods
Additional proposed modification for PR #4769 (`dev/abs-diag-smoothers`)
2025-05-18 10:32:17 -07:00
Justin Laughlin d3281a8a86 Address some comments from Dylan - update path on nurbs test 2025-05-16 22:12:17 -07:00
Justin Laughlin a7d59d35e0 Fix pkv_map in GetEdgeToUniqueKnotvector 2025-05-16 21:39:50 -07:00
Justin Laughlin 70680d6187 Add more nurbs meshes to test with 2025-05-16 20:04:01 -07:00
Dylan Copeland 35aeecb5c0 Minor fixes. 2025-05-16 19:15:26 -07:00
Dylan Copeland 44f2a63f16 Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-05-16 19:07:45 -07:00
Justin Laughlin 948462da6e minor cleanup 2025-05-16 18:07:12 -07:00
Justin Laughlin fddefa7838 Update GetEdgeToUniqueKnotvector - should handle edge cases better 2025-05-16 18:04:21 -07:00
Veselin Dobrev 62a01ee670 Re-format the sample runs in miniapps/diag-smoothers so that they follow the
formatting used in other places.
2025-05-16 16:54:53 -07:00
Veselin Dobrev e23768f5f7 Additional proposed modification for PR #4769 (dev/abs-diag-smoothers) 2025-05-16 16:13:36 -07:00
Justin Laughlin 53d4f78fdb Merge branch 'master' into fix-nurbsext-constructor 2025-05-14 18:57:40 -07:00
Justin Laughlin eeb71eee37 Add test 2025-05-14 18:10:30 -07:00
Justin Laughlin 3630a8f8a1 Add getters for testing; minor fix/formatting 2025-05-14 17:01:03 -07:00
Justin Laughlin 048904b731 Cleanup/formatting + implement Dylan's comments 2025-05-14 14:31:26 -07:00
Justin Laughlin 2496b33699 Fix Mesh::GetEdgeToUniqueKnotvector so it works for 1d 2025-05-14 11:39:04 -07:00
Andrew Ho eb9022540e Merge branch 'master' into hip-unit-tests 2025-05-12 13:04:43 -07:00
Dylan Copeland 1ff8b6811d Empty line 2025-05-09 11:03:09 -07:00
Dylan Copeland cf269700a8 Enable and add sample runs for p-refinement on serial mixed meshes. 2025-05-09 10:56:46 -07:00
John Camier cf127c8b14 Merge branch 'master' into gpu-thread-direct 2025-05-09 08:05:36 -07:00
Mark L. Stowell cb0c205bd1 Merge branch 'master' into mesh-transform-dev 2025-05-07 13:58:59 -07:00
Stowell, Mark L. 2ed6fdc85a Cleaning up compiler warnings 2025-05-07 13:12:53 -07:00
Justin Laughlin 15b35a01ba Merge branch 'master' into fix-nurbsext-constructor 2025-05-06 10:45:29 -07:00
Justin Laughlin ff746a8af6 Remove some getters that are unnecessary for this PR 2025-05-05 14:47:13 -07:00
Tom Stitt df36d0f352 add check for D1D size 2025-05-05 14:06:44 -07:00
Tom Stitt 765ebcecaa Merge branch 'gpu-thread-direct' of github.com:mfem/mfem into gpu-thread-direct 2025-05-05 13:49:09 -07:00
John Camier b3a08b91d6 Merge branch 'master' into hip-unit-tests 2025-05-04 07:38:47 -07:00
Tzanio Kolev 726b5f99ff Added jittering option (off by default) 2025-05-03 23:27:20 -07:00
Dylan Copeland a943683063 Removed optional nodes argument to Mesh::Print. Refactored miniapp. 2025-05-03 19:47:54 -07:00
Tzanio Kolev b6fb45f384 CI fixes 2025-05-03 18:22:51 -07:00
Tzanio Kolev d119fa7636 CI fixes 2025-05-03 17:43:50 -07:00
Tzanio Kolev 4a7c643f99 Fixed, hacks and improvements in the NURBS Surface miniapp 2025-05-03 17:34:45 -07:00
Tzanio Kolev 41ea219782 Merge branch 'master' into plbound 2025-05-03 13:41:55 -07:00
Tzanio Kolev 2dfd2ccfc5 Merge branch 'master' into gpu-thread-direct 2025-05-03 13:27:33 -07:00
Tzanio Kolev ae9a8b2897 Merge branch 'master' into nurbs-surf 2025-05-02 14:02:23 -07:00
Veselin Dobrev b54ee3537f In a few places, use std::fabs instead of std::abs or just abs 2025-05-02 10:21:04 -07:00
Dylan Copeland 2d3aba5d87 Generalized machine epsilon in KnotVector::FindMaxima. 2025-05-02 10:05:58 -07:00
Andrew Ho f4c66c56d6 updated changelog 2025-05-02 09:56:15 -07:00
Andrew Ho 3c3face72c Merge branch 'master' into hip-unit-tests 2025-05-02 10:31:33 -06:00
dylan-copeland 762551da72 Mac fix. 2025-05-01 21:08:35 -07:00
Dylan Copeland 293a374a74 Minor fixes. 2025-05-01 21:00:07 -07:00
Dylan Copeland 39be93547e Refactoring to simplify the API. 2025-05-01 20:49:50 -07:00
Dylan Copeland 7559d37c58 Label glvis windows. 2025-05-01 20:13:23 -07:00
dylan-copeland 90eed63144 Remove unused variables. 2025-05-01 18:23:39 -07:00
Gabriel Pinochet-Soto d46b421417 !!useAbs 2025-05-01 17:28:39 -07:00
Dylan Copeland e6bc4e5a0e Remove no-vis in tests. 2025-05-01 17:25:08 -07:00
Gabriel Pinochet-Soto 1d7029c5d6 Add Abs to DofToQuad, needs testing 2025-05-01 17:12:14 -07:00
Gabriel Pinochet-Soto 5596d38532 Docstring AbsPhyDer 2025-05-01 17:05:14 -07:00
Gabriel Pinochet-Soto 389580efef Replace mfem_error with MFEM_ABORT 2025-05-01 16:53:12 -07:00
Gabriel Pinochet-SotoandWill Pazner 2029636109 Update fem/bilinearform_ext.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-01 16:52:22 -07:00
Gabriel Pinochet-SotoandWill Pazner 1ca38f826c Update fem/bilinearform_ext.cpp
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-01 16:49:46 -07:00
Dylan Copeland 03ec8d78e2 Documentation. New miniapp checklist. 2025-05-01 16:46:32 -07:00
Dylan Copeland 3fbeff1db7 Fix visualization. 2025-05-01 11:50:07 -07:00
Dylan Copeland 3d7ac596da Revert a previous change. Reduce output to 3 meshes. 2025-05-01 11:32:34 -07:00
Dylan Copeland 57e1693fc7 CHANGELOG and some minor edits. 2025-05-01 10:54:42 -07:00
Gabriel Pinochet-SotoandWill Pazner 4745e062f2 Docstring Array<T>::Abs
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-05-01 09:56:57 -07:00
Tzanio Kolev a631ab7e77 Merge branch 'master' into dev/abs-diag-smoothers 2025-05-01 09:46:00 -07:00
Dylan Copeland 67ca28a501 Merge branch 'nurbs-surf' of github.com:mfem/mfem into nurbs-surf 2025-04-30 12:40:46 -07:00
Dylan Copeland 6d334a925a Merge branch 'master' of github.com:mfem/mfem into nurbs-surf 2025-04-30 12:40:31 -07:00
Dylan Copeland 8183e1729d More optimization by reusing banded matrix factorization. 2025-04-30 12:40:16 -07:00
Tzanio Kolev 30f3e42232 Merge branch 'master' into fix-nurbsext-constructor 2025-04-30 10:49:45 -07:00
Tzanio Kolev ef1e0caed1 Merge branch 'master' into nurbs-surf 2025-04-30 09:01:34 -07:00
Will Pazner adbe1bfe3a Write Exodus meshes with real_t instead of double
Also pass std::string by const reference instead of value
2025-04-30 08:59:07 -07:00
John Camier 5590b87f5e Merge branch 'master' into hip-unit-tests 2025-04-30 07:53:13 -07:00
Dylan Copeland f28cd12995 ifdef lapack for banded solver 2025-04-29 22:32:36 -07:00
Dylan Copeland e0aba0647d Banded solver for 1D KnotVector interpolation. 2025-04-29 22:28:49 -07:00
camierjs cd4593bf8e Fix MFEM_TMOP_(PA_)DEVICE and update gitignore 2025-04-29 16:09:23 -07:00
Justin Laughlin 4d92694fc5 Update GetNURBSPatches 2025-04-29 13:41:11 -07:00
Dylan Copeland 0cc5280e34 New miniapp to fit a NURBS surface to a structured grid of 3D point data. 2025-04-29 11:14:28 -07:00
Justin Laughlin a4214d22f4 fix typo 2025-04-28 18:40:02 -07:00
Tzanio Kolev f12f0efb31 Merge branch 'master' into gpu-thread-direct 2025-04-28 18:35:46 -07:00
Justin Laughlin ce1576411c Cleanup demo script 2025-04-28 18:31:22 -07:00
Justin Laughlin 1c4f617fd5 Cleaup 2025-04-28 18:30:14 -07:00
Justin Laughlin b234475774 minor cleanup 2025-04-28 18:25:18 -07:00
Justin Laughlin 5da1d7ddf3 Update edge_to_knot (renamed edge_to_ukv) generator so it maps to unique knotvector indices 2025-04-28 18:18:18 -07:00
Justin Laughlin cff3c6cb6f Cleanup Mesh::LoadPatchTopo and Mesh::GetEdgeToKnotMapping 2025-04-28 16:32:56 -07:00
Justin Laughlin 7ef863b731 Add new algorithm to compute edge_to_knot map 2025-04-28 15:42:27 -07:00
Tzanio Kolev 09b31f0526 Merge branch 'master' into plbound 2025-04-26 12:32:38 -07:00
Gabriel Pinochet-Soto 5044d9cd45 Remove *Abs* functions in operator.xpp and related 2025-04-25 12:35:02 -07:00
Gabriel Pinochet-Soto 2edddb700e Remove unused functions in bilininteg.xpp 2025-04-25 12:21:08 -07:00
Gabriel Pinochet-Soto 1d79ab00ba Remove todo; cf. 3b49d70 2025-04-25 12:03:21 -07:00
Gabriel Pinochet-Soto 85e08b67a0 Update CHANGELOG 2025-04-25 11:55:04 -07:00
Tom Stitt e3c3150958 Merge remote-tracking branch 'origin/master' into gpu-thread-direct 2025-04-25 11:39:44 -07:00
Gabriel Pinochet-Soto c513bb1276 Merge branch 'master' of github.com:mfem/mfem into dev/abs-diag-smoothers 2025-04-25 11:00:03 -07:00
Justin Laughlin 7559524573 separate GetEdgeToKnotMapping. Updating constructor WIP 2025-04-24 22:46:33 -07:00
Justin Laughlin 7ac34aab22 More testing - turns out mapping is more complicated because it is edges to unique knotvectors 2025-04-24 21:30:55 -07:00
Justin Laughlin 17edacb630 Fix orientation in edge_to_knot map of NURBSExtension constructor 2025-04-24 18:36:50 -07:00
Justin Laughlin 703ba47151 demo 2025-04-24 18:06:42 -07:00
Justin Laughlin 57caab145a Getters for NURBS patch data 2025-04-24 17:08:54 -07:00
Andrew Ho 911fbfbe82 updated testing readme 2025-04-24 11:36:19 -07:00
Andrew Ho 0b3b21dbb2 Merge remote-tracking branch 'base/hip-unit-tests' into hip-unit-tests 2025-04-24 11:21:52 -07:00
Andrew Ho 823fd86a87 forgot about cmake 2025-04-24 11:18:47 -07:00
Andrew Ho 2d3c1bc79a Merge remote-tracking branch 'base/hip-unit-tests' into hip-unit-tests 2025-04-24 10:48:26 -07:00
Andrew Ho 1d556b93e7 more gpu unit tests 2025-04-24 10:47:57 -07:00
Andrew Ho 00fcb1b37f removed cuda/hip-specific unit test executables 2025-04-24 09:55:55 -07:00
Andrew Ho 9445358bc9 Merge branch 'master' into hip-unit-tests 2025-04-24 09:32:32 -07:00
Andrew Ho e4becc6e02 Merge branch 'master' into hip-unit-tests 2025-04-22 15:28:02 -07:00
Veselin Dobrev b294248b06 Update .gitignore 2025-04-22 13:08:44 -07:00
Tom Stitt 9b1c3b718d remove check 2025-04-22 12:59:25 -07:00
Tom Stitt e9826f9c69 Adds MFEM_FOREACH_THREAD_DIRECT which uses a conditional instead of a loop for faster GPU kernels when the thread loop bound is less-than-or-equal-to the corresponding block size 2025-04-22 12:53:25 -07:00
Stowell, Mark L. 83ca977d71 Adding an affine transformation option to mesh-explorer 2025-04-21 16:38:16 -07:00
Stowell, Mark L. 6739574668 Adding an affine mesh transformation coefficient 2025-04-21 16:37:54 -07:00
Ketan Mittal 9f153f9fcb Merge branch 'master' into plbound 2025-04-20 18:59:08 +12:00
Veselin Dobrev 00d35316b9 Merge branch 'master' into dev/abs-diag-smoothers 2025-04-18 13:43:42 -07:00
Mittal, Ketan 84a3fe000b fix gitignore 2025-04-18 13:32:57 -07:00
Stowell, Mark L. 7354e1ce6c Adding option to increase the space dimension of the mesh 2025-04-18 12:00:58 -07:00
Mittal, Ketan d5089996cc fix AddVertex usage 2025-04-18 11:55:13 -07:00
Mittal, Ketan 9787ad0d3b minor 2025-04-18 11:37:31 -07:00
Mittal, Ketan 26fb49960e add miniapp description 2025-04-18 11:12:32 -07:00
Mittal, Ketan 5cb2c82d59 fix makefile 2025-04-18 11:03:33 -07:00
Mittal, Ketan 9e7f8ce838 minor 2025-04-17 15:38:39 -07:00
Veselin Dobrev 41e4360d2e Fix the CUDA build and the out-of-source build with GNU make 2025-04-17 13:43:37 -07:00
Mittal, Ketan 263dd0f019 fix make clean 2025-04-17 12:11:02 -07:00
Mittal, Ketan 433f9a4c51 minor 2025-04-17 11:36:37 -07:00
Mittal, Ketan 08856158ce fix for some static constexpr definitions in hpp 2025-04-17 11:18:20 -07:00
Veselin Dobrev b21bf9a2d9 Merge branch 'master' into dev/abs-diag-smoothers 2025-04-16 22:22:56 -07:00
Veselin Dobrev 8d705f4230 Various formatting edits and other small tweaks 2025-04-16 22:21:33 -07:00
Mittal, Ketan 0a6c72c52d doxygen fix 2025-04-16 16:12:36 -07:00
Mittal, Ketan 8cefee8799 fix ParFESpace definition 2025-04-16 13:52:40 -07:00
Mittal, Ketan e9b631116f fix typo 2025-04-16 13:41:00 -07:00
Mittal, Ketan 6468880b2a minor 2025-04-16 12:58:07 -07:00
Mittal, Ketan 89a0e08f29 remove some unused methods 2025-04-16 12:57:50 -07:00
Mittal, Ketan 7471a22505 resolve conflicts 2025-04-16 12:56:40 -07:00
Mittal, Ketan 779ce337f4 miniapps 2025-04-16 12:45:55 -07:00
Mittal, Ketan 54b24da610 missing files 2025-04-16 12:45:30 -07:00
Mittal, Ketan 1a332939a0 initial capability 2025-04-16 12:45:11 -07:00
Veselin Dobrev bbd26e1835 In miniapps/diag-smoothers, fix the CMake tests and adjust mesh refinements 2025-04-14 16:41:24 -07:00
Veselin Dobrev e8842c506a Fix the build in miniapps/smoothers 2025-04-14 14:54:25 -07:00
Veselin Dobrev 214313291b Merge branch 'master' into dev/abs-diag-smoothers
Fixed conflicts:
  CHANGELOG
  fem/bilinearform_ext.cpp
  fem/bilinearform_ext.hpp
2025-04-14 13:56:38 -07:00
Andrew Ho 783136234a Merge branch 'master' into hip-unit-tests 2025-04-13 19:35:44 -07:00
Andrew Ho dd85ae3384 Merge remote-tracking branch 'base/master' into hip-unit-tests 2025-04-07 11:54:06 -07:00
Andrew Ho 294a71c705 Merge remote-tracking branch 'base/hip-unit-tests' into hip-unit-tests 2025-04-04 11:58:14 -07:00
Andrew Ho 0b632bf3d4 Merge remote-tracking branch 'base/master' into hip-unit-tests 2025-04-04 11:39:35 -07:00
Gabriel Pinochet-Soto cfbaf5a6bf Header renaming 2025-03-29 11:02:27 -07:00
Gabriel Pinochet-Soto 2e2b8faba5 Modify abort msg in MMA::MMASubSvanberg 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto bf9ce2c6a4 Update gitignore and CHANGELOG 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 9f858378ca Add Examples 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto c2b948a036 AbsMult for Mass Integrs 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 0771f904c0 AbsMult for Diffusion 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 3b49d70f35 AbsMult for ElasticityInteg 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 7c077e656d CurlCurl kernels AbsApply 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto b9f146b46f Add AbsMult to QuadInterp 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 07438814ce Add interface in pfespace.hpp 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 328c914481 Add AbsMult for restriction operators; address code duplication comment 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 1766007d8a Add AbsMult interface to Integrators 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 6311548ac6 Use constexpr on (Abs)Mult cases 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 9d84c17e0b Add AbsMult to base class Operator 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 23c1fc6452 Add Hypre AbsMult 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 3117d0e8ab Implement Mult/AbsMult for DenseMatrix 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto a9b720b79d Update Mult Kernels
- Observation: This could be replaced with a lambda function of the
type `useAbs ? [](TA a) { return std::abs(a); } : ...`
2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 3fe878efaa Ommisions...
- Ommit hypre_parcsr.xpp implementation of L(p,q)
- Ommit solvers.xpp implementation of L(p,q)
- Ommit tests implementation of L(p,q)
2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 4c0def024c Monitor SLI (akin to CG) 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto bf7c26ebf9 Vector::Abs 2025-03-28 22:13:22 -07:00
Gabriel Pinochet-Soto 73014e34c5 Add Array<T>::Abs()
- static assert of arithmetic type of T
2025-03-28 22:13:22 -07:00
Andrew Ho fabce12b76 Merge branch 'master' into hip-unit-tests 2025-03-24 23:55:09 -07:00
Andrew Ho 8fbfdb19fe added general GPU and CUDA/HIP-specific unit tests
added gpu, raja-gpu, etc. for generic GPU device configuration
2025-03-24 13:23:43 -07:00
181 changed files with 12790 additions and 2325 deletions
+154
View File
@@ -0,0 +1,154 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitizer Config
description: Sets up environment variables for MFEM sanitizer workflow
inputs:
DEBUG:
description: If true, use intermediate caches to speed up the workflow
by reusing previous builds.
default: false
REPOSITORY:
description: Repository to checkout
default: mfem/mfem
BRANCH:
description: Branch to checkout
default: ubsan
CLANG_VER:
description: CLANG version to use
default: 18
# https://github.com/llvm/llvm-project/releases
LLVM_VER:
description: LLVM version to use
default: 19.1.7
# https://github.com/hypre-space/hypre/releases
HYPRE_VER:
description: HYPRE version to use
default: 2.19.0
METIS_VER:
description: METIS version to use
default: 4.0.3
CTEST:
description: CTest command to use
default: ctest -j --test-load $(nproc)
--schedule-random
--stop-on-failure --output-on-failure
--test-dir
# https://clang.llvm.org/docs/AddressSanitizer.html
ASAN_OPTIONS:
default: detect_leaks=1,
strict_init_order=1,
strict_string_checks=1,
check_initialization_order=1,
detect_stack_use_after_return=1
ASAN_CXXFLAGS:
default: -fsanitize=address
-fsanitize-address-use-after-scope
ASAN_LDFLAGS:
default: -fsanitize=address
# https://clang.llvm.org/docs/UndefinedBehaviorSanitizer.html
UBSAN_OPTIONS:
default: halt_on_error=1, print_stacktrace=1
UBSAN_CXXFLAGS:
default: -fsanitize=undefined
UBSAN_LDFLAGS:
default: -fsanitize=undefined
# https://clang.llvm.org/docs/MemorySanitizer.html
MSAN_OPTIONS:
default: "poison_in_dtor=1"
MSAN_CXXFLAGS:
default: -fsanitize=memory
-fsanitize-memory-track-origins
-fsanitize-memory-use-after-dtor
MSAN_LDFLAGS:
default: -fsanitize=memory
LSAN_DIR:
description: LSAN suppression directory
default: lsan
LSAN_FILE:
description: LSAN suppression file
default: lsan.supp
NO_FLAGS:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
runs:
using: 'composite'
steps:
- name: Env (Inputs)
run: |
echo DEBUG=${{inputs.DEBUG}} >> $GITHUB_ENV
echo REPOSITORY=${{inputs.REPOSITORY}} >> $GITHUB_ENV
echo BRANCH=${{inputs.BRANCH}} >> $GITHUB_ENV
echo CLANG_VER=${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo LLVM_VER=${{inputs.LLVM_VER}} >> $GITHUB_ENV
echo HYPRE_VER=${{inputs.HYPRE_VER}} >> $GITHUB_ENV
echo METIS_VER=${{inputs.METIS_VER}} >> $GITHUB_ENV
echo CTEST=${{inputs.CTEST}} >> $GITHUB_ENV
echo ASAN_OPTIONS=${{inputs.ASAN_OPTIONS}} >> $GITHUB_ENV
echo UBSAN_OPTIONS=${{inputs.UBSAN_OPTIONS}} >> $GITHUB_ENV
echo MSAN_OPTIONS=${{inputs.MSAN_OPTIONS}} >> $GITHUB_ENV
echo LSAN_DIR=${{inputs.LSAN_DIR}} >> $GITHUB_ENV
echo LSAN_FILE=${{inputs.LSAN_FILE}} >> $GITHUB_ENV
echo ASAN_CXXFLAGS=${{inputs.ASAN_CXXFLAGS}} >> $GITHUB_ENV
echo ASAN_LDFLAGS=${{inputs.ASAN_LDFLAGS}} >> $GITHUB_ENV
echo UBSAN_CXXFLAGS=${{inputs.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Env (dir)
run: |
echo LLVM_DIR=${{github.workspace}}/llvm >> $GITHUB_ENV
echo HYPRE_DIR=hypre-${{inputs.HYPRE_VER}} >> $GITHUB_ENV
echo METIS_DIR=metis-${{inputs.METIS_VER}} >> $GITHUB_ENV
shell: bash
- name: Env (bis)
run: |
echo CC=clang-${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo CXX=clang++-${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo LLVM_INC=${{env.LLVM_DIR}}/include/c++/v1 >> $GITHUB_ENV
echo LLVM_LIB=${{env.LLVM_DIR}}/lib >> $GITHUB_ENV
echo HYPRE_TGZ=v${{inputs.HYPRE_VER}}.tar.gz >> $GITHUB_ENV
echo METIS_TGZ=metis-${{inputs.METIS_VER}}.tar.gz >> $GITHUB_ENV
LSAN_SUPPRESSIONS="${{github.workspace}}/${{inputs.LSAN_DIR}}/${{inputs.LSAN_FILE}}"
echo "LSAN_OPTIONS=suppressions=$LSAN_SUPPRESSIONS" >> $GITHUB_ENV
shell: bash
- name: Env (ter)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo LLVM_CXXFLAGS=-stdlib=libc++ -I${{env.LLVM_INC}} -Isystem${{env.LLVM_INC}} >> $GITHUB_ENV
echo LLVM_LDFLAGS=-L${{env.LLVM_LIB}} -lc++abi -Wl,-rpath,${{env.LLVM_LIB}} >> $GITHUB_ENV
shell: bash
- name: Env (quater)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo CXXFLAGS=${{env.LLVM_CXXFLAGS}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LLVM_LDFLAGS}} >> $GITHUB_ENV
shell: bash
+91
View File
@@ -0,0 +1,91 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'MFEM Compilation'
description: 'MFEM Compilation'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
runs:
using: 'composite'
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
if: ${{env.DEBUG == 'true'}}
id: debug
with:
path: mfem/build
key: build-${{inputs.par}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/setup
if: ${{steps.debug.outputs.cache-hit != 'true'}}
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- name: Build with ASAN
if: inputs.sanitizer == 'asan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.ASAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Build with MSAN
if: inputs.sanitizer == 'msan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Build with UBSAN
if: inputs.sanitizer == 'ubsan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- uses: mfem/github-actions/build-mfem@v2.5
if: ${{steps.debug.outputs.cache-hit != 'true'}}
env:
CXXFLAGS: ${{env.CXXFLAGS}}
LDFLAGS: ${{env.LDFLAGS}}
with:
mpi: ${{inputs.par == 'false' && 'seq' || 'par'}}
mfem-dir: mfem
os: ${{runner.os}}
library-only: true
build-system: cmake
hypre-dir: ${{env.HYPRE_DIR}}
metis-dir: ${{env.METIS_DIR}}
config-options: >-
-GNinja
-DMPICXX=${{env.CXX}}
-DCMAKE_CXX_STANDARD=17
-DMFEM_USE_MEMALLOC=OFF
-DCMAKE_BUILD_TYPE=Release
-DCMAKE_VERBOSE_MAKEFILE=ON
-DCMAKE_CXX_COMPILER=${{env.CXX}}
-DCMAKE_CXX_FLAGS_RELEASE='-g -O1 -fno-omit-frame-pointer'
- name: Delete object files
if: ${{steps.debug.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: find . -type f -name '*.o' -delete
shell: bash
- uses: actions/upload-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
if-no-files-found: error
retention-days: 1
overwrite: false
+33
View File
@@ -0,0 +1,33 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Install MPI'
description: 'Installs MPI and set up its environment variables'
runs:
using: 'composite'
steps:
- name: Install
run: sudo apt-get install openmpi-bin libopenmpi-dev
shell: bash
- name: Env
run: |
echo PRTE_MCA_rmaps_default_mapping_policy=:oversubscribe >> $GITHUB_ENV
echo MPI_INC=$(mpicxx --showme:compile) >> $GITHUB_ENV
echo MPI_LIB=$(mpicxx --showme:link) >> $GITHUB_ENV
shell: bash
- name: Env (bis)
run: |
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
shell: bash
@@ -0,0 +1,71 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Restore state'
description: 'Restore state to be able to run checks, tests'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
cache-path:
description: 'path to what needs to be restored'
default: none
cache-skip:
description: 'Skip cache restoration'
default: false
outputs:
cache-hit:
description: 'Output from a specific step'
value: ${{steps.debug.outputs.cache-hit}}
runs:
using: 'composite'
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
if: ${{env.DEBUG == 'true' && inputs.cache-skip != 'true'}}
id: debug
with:
path: ${{inputs.cache-path}}
key: ${{github.job}}-${{inputs.par}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/setup
if: ${{steps.debug.outputs.cache-hit != 'true'}}
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- uses: actions/download-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
- name: Ninja Patch
working-directory: mfem/build
run: |
sed -i -e 's/CXX_STATIC_LIBRARY_LINKER__mfem_Release.*/CUSTOM_COMMAND/' build.ninja
sed -i -e '/build tests\/unit\/all:/ s/tests\/unit\/[^ ]*unit_tests[^ ]*//g' build.ninja
sed -i -e '/^add_test(\[=\[\(unit_tests\|punit_tests\)\]=\]/ s/)/ "--input-file .\/list-test-names-${{matrix.tag}}" "--min-duration 1")/' tests/unit/CTestTestfile.cmake
shell: bash
- name: Copy Data
if: ${{steps.debug.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
ninja cmake_object_order_depends_target_unit_tests
cp -pR ../tests/unit/data tests/unit
shell: bash
+64
View File
@@ -0,0 +1,64 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Setup state'
description: 'Sets up the state to be able to run build & run'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
runs:
using: 'composite'
steps:
- uses: actions/cache/restore@v4 # Cache for LLVM libcxx
with:
path: ${{env.LLVM_DIR}}
fail-on-cache-miss: true
key: build-libcxx-${{env.LLVM_VER}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/mpi
if: ${{inputs.par == 'true'}}
- uses: actions/cache/restore@v4 # Cache for Hypre
if: ${{inputs.par == 'true'}}
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- uses: actions/cache/restore@v4 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Hypre/Metis links
if: ${{inputs.par == 'true'}}
run: ln -s -f ${{env.HYPRE_DIR}} hypre && ln -s -f ${{env.METIS_DIR}} metis-4.0
shell: bash
- uses: actions/cache/restore@v4 # Cache for LSAN suppression file
with:
path: ${{env.LSAN_DIR}}
fail-on-cache-miss: true
key: build-lsan-suppression-file
- uses: actions/checkout@v4 # Checkout the repository
with:
path: mfem
# ref: ${{env.BRANCH}}
# repository: ${{env.REPOSITORY}}
+26 -7
View File
@@ -7,18 +7,17 @@
https://mfem.org
This directory contains the GitHub CI scripts for MFEM.
Note that some of these scripts use the shared MFEM GitHub Actions from the external mfem/github-actions repository:
https://github.com/mfem/github-actions
<https://github.com/mfem/github-actions>
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.1`, the `v2.1` suffix denotes the branch in the above from which the action is taken.
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
The current CI workflows are:
### `repo-check.yml`
## `repo-check.yml`
Runs a number of static repository-level sanity checks.
@@ -30,19 +29,39 @@ Runs a number of static repository-level sanity checks.
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
### `mfem-analysis.yml` (`build-analysis`)
## `mfem-analysis.yml` (`build-analysis`)
Checks if the code builds and satisfies minimal requirements.
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
### `builds-and-tests.yml`
## `builds-and-tests.yml`
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
Uses the following GitHub Actions from https://github.com/mfem/github-actions:
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
- `mfem/github-actions/build-hypre`
- `mfem/github-actions/build-metis`
- `mfem/github-actions/build-mfem`
- `mfem/github-actions/upload-coverage`
## Sanitizer Workflow for MFEM Verification
This workflow validates MFEM unit tests, examples, and miniapps using sanitizer tools.
- `sanitizers.yml` orchestrates:
- Building and caching dependencies: HYPRE, METIS, LSAN suppression file, and LLVM libcxx.
- Launching fine-grained jobs for serial (ASAN, MSAN, UBSAN) and parallel (ASAN, UBSAN) sanitizers.
- `sanitize-tests.yml` is a reusable workflow accepting `par` mode (`true` for parallel) and `sanitizer` (ASAN, MSAN, or UBSAN) as inputs. It executes the following jobs:
- **Build**: Compiles the MFEM library with specified parallel and sanitizer settings.
- **Check**: Runs verification checks.
- Parallel jobs to test the following: **Examples**, **Miniapps** and **Unit tests**
The workflow leverages composite actions in `.github/actions/sanitize/`:
- `config`: Centralizes settings for the sanitizer workflow.
- `mfem`: Manages the MFEM library build process.
- `mpi`: Installs MPI and applies additional compilation flags.
- `restore`: Restores the testing environment state.
- `setup`: Builds or restores cached dependencies.
-69
View File
@@ -1,69 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
name: "Sanitizer"
permissions:
actions: write
on:
push:
branches:
- master
- next
pull_request:
workflow_dispatch:
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
Serial:
runs-on: ubuntu-24.04
steps:
- name: MFEM Checkout
uses: actions/checkout@v4
with:
path: mfem
- name: MFEM Build
uses: mfem/github-actions/build-mfem@v2.5
with:
os: ${{ runner.os }}
target: opt
mpi: seq
hypre-dir: unused-hypre-dir
metis-dir: unused-metis-dir
mfem-dir: mfem
build-system: make
library-only: false
config-options:
CXX="clang++-18"
CXXFLAGS="-g -O1 -std=c++17
-fsanitize=address
-fno-omit-frame-pointer
-fsanitize-address-use-after-scope"
- name: MFEM Info
working-directory: mfem
run: make info
- name: MFEM Sanitize
working-directory: mfem
run:
ASAN_OPTIONS="detect_leaks=1,
strict_init_order=1,
strict_string_checks=1,
check_initialization_order=1,
detect_stack_use_after_return=1"
make test
@@ -0,0 +1,39 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-hypre
on:
workflow_call:
jobs:
build-hypre:
runs-on: ubuntu-latest
name: 2.19.0
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{env.HYPRE_TGZ}}
dir: ${{env.HYPRE_DIR}}
target: int32
precision: fp64
build-system: make
@@ -0,0 +1,76 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-libcxx
on:
workflow_call:
jobs:
build-llvm-libcxx:
runs-on: ubuntu-latest
strategy:
matrix:
sanitizer: [asan, msan, ubsan]
include:
- sanitizer: asan
llvm_use_sanitizer: "Address"
- sanitizer: msan
llvm_use_sanitizer: "MemoryWithOrigins"
- sanitizer: ubsan
llvm_use_sanitizer: "Undefined"
name: ${{matrix.sanitizer}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.LLVM_DIR}}
key: build-libcxx-${{env.LLVM_VER}}-${{matrix.sanitizer}}
- name: Clone
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
run: >
git clone --filter=blob:none --depth=1
--branch llvmorg-${{env.LLVM_VER}}
--no-checkout https://github.com/llvm/llvm-project.git llvm-project
- name: Checkout
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: llvm-project
run: |
git sparse-checkout set --cone
git checkout llvmorg-${{env.LLVM_VER}}
git sparse-checkout set cmake llvm/cmake runtimes libcxx libcxxabi
- name: Mkdir
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
run: mkdir ${{env.LLVM_DIR}}
- name: CMake
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: ${{env.LLVM_DIR}}
run: >
VERBOSE=1
cmake -GNinja ../llvm-project/runtimes/
-DCMAKE_C_COMPILER=${{env.CC}}
-DCMAKE_CXX_COMPILER=${{env.CXX}}
-DCMAKE_BUILD_TYPE=RelWithDebInfo
-DCMAKE_INSTALL_PREFIX=/usr
-DLLVM_USE_SANITIZER=${{matrix.llvm_use_sanitizer}}
-DLLVM_BUILD_32_BITS=OFF
-DLIBCXXABI_USE_LLVM_UNWINDER=OFF
-DLLVM_INCLUDE_TESTS=OFF
-DLIBCXX_INCLUDE_TESTS=OFF
-DLIBCXX_INCLUDE_BENCHMARKS=OFF
-DLLVM_ENABLE_RUNTIMES='libcxx;libcxxabi'
- name: Build
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: ${{env.LLVM_DIR}}
run: cmake --build . -- cxx cxxabi
+38
View File
@@ -0,0 +1,38 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-file-lsan
on:
workflow_call:
jobs:
build-file-lsan:
runs-on: ubuntu-latest
name: lsan.supp
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.LSAN_DIR}}
key: build-lsan-suppression-file
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
run: |
mkdir -p ${{env.LSAN_DIR}}
cat << EOF > ${{env.LSAN_DIR}}/${{env.LSAN_FILE}}
leak:libevent_core-2.1.so
leak:ompi_mpi_finalize
leak:ompi_mpi_init
leak:PMPI_Init
leak:strdup
EOF
@@ -0,0 +1,36 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-metis
on:
workflow_call:
jobs:
build-metis:
runs-on: ubuntu-latest
name: 4.0.3
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{env.METIS_TGZ}}
dir: ${{env.METIS_DIR}}
+197
View File
@@ -0,0 +1,197 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitize
on:
workflow_call:
inputs:
par:
description: 'Whether to build for parallel (true/false)'
required: false
default: false
type: boolean
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
required: true
default: asan
type: string
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/mfem
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
check:
needs: [build]
runs-on: ubuntu-latest
env:
ex: ${{inputs.par && 'ex1p' || 'ex1'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/examples/${{env.ex}}
- name: MFEM Check
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v check
examples:
needs: [check]
runs-on: ubuntu-latest
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/examples/ex1
- name: Build Examples
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v examples
- name: Test Examples
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} examples ${{env.exclude}} --show-only
${{env.CTEST}} examples ${{env.exclude}}
miniapps:
needs: [check]
runs-on: ubuntu-latest
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/miniapps/meshing/minimal-surface
- name: Build Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v miniapps
- name: Test Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} miniapps ${{env.exclude}} --show-only
${{env.CTEST}} miniapps ${{env.exclude}}
tests-miniapps:
needs: [check]
runs-on: ubuntu-latest
env:
run: ${{inputs.par && '-R "_cpu_np"' || ''}}
exclude: ${{inputs.par && '"unit_tests|debug"' || '"^unit_tests$|debug"'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/sedov_tests_cpu
- name: Build Tests Unit Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v tests/unit/all
- name: Run Tests Unit Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} tests/unit -E ${{env.exclude}} ${{env.run}} --show-only
${{env.CTEST}} tests/unit -E ${{env.exclude}} ${{env.run}}
tests-unit-build:
needs: [check]
runs-on: ubuntu-latest
env:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- name: Build Unit Tests
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v ${{env.unit_tests}}
- name: Delete object files
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: find . -type f -name '*.o' -delete
- uses: actions/upload-artifact@v4
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit/${{env.unit_tests}}
if-no-files-found: error
retention-days: 1
overwrite: false
tests-unit-run:
needs: [tests-unit-build]
runs-on: ubuntu-latest
strategy:
matrix:
tag: [0, 1, 2, 3]
name: tests-unit-run-${{matrix.tag}}
env:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
np: ${{inputs.par && '_np=2' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- uses: actions/download-artifact@v4
if: ${{steps.restore.outputs.cache-hit != 'true'}}
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit
- name: Split Unit Tests
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: |
chmod 755 ${{env.unit_tests}}
./${{env.unit_tests}} --list-test-names-only | tail -n +2 > list-test-names
shuf list-test-names -o list-test-names
split --verbose -n l/4 -d -a 1 list-test-names list-test-names-
- name: Cat Unit Tests ${{matrix.tag}}
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: cat list-test-names-${{matrix.tag}}
- name: Run Unit Tests ${{matrix.tag}}
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} tests/unit -R "${{env.unit_tests}}${{env.np}}" --show-only
${{env.CTEST}} tests/unit -R "${{env.unit_tests}}${{env.np}}"
+73
View File
@@ -0,0 +1,73 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitizers
permissions:
actions: write
on:
push:
branches: ["master", "next"]
pull_request:
workflow_dispatch:
concurrency:
group: ${{github.workflow}}-${{github.ref}}
cancel-in-progress: true
jobs:
# Build steps for dependencies
build-hypre:
uses: ./.github/workflows/sanitize-build-hypre.yml
build-metis:
uses: ./.github/workflows/sanitize-build-metis.yml
build-lsan:
uses: ./.github/workflows/sanitize-build-lsan.yml
build-libcxx:
uses: ./.github/workflows/sanitize-build-libcxx.yml
# Serial sanitizers: asan, msan, ubsan
seq-asan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: asan
seq-msan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: msan
seq-ubsan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: ubsan
# Parallel sanitizers: asan, ubsan
par-asan:
needs: [build-libcxx, build-hypre, build-metis]
uses: ./.github/workflows/sanitize-tests.yml
with:
par: true
sanitizer: asan
par-ubsan:
needs: [build-libcxx, build-hypre, build-metis]
uses: ./.github/workflows/sanitize-tests.yml
with:
par: true
sanitizer: ubsan
+11 -2
View File
@@ -232,6 +232,7 @@ miniapps/meshing/fit-node-position
miniapps/meshing/trimmer
miniapps/meshing/reflector
miniapps/meshing/ref321
miniapps/meshing/mesh-bounding-boxes
miniapps/meshing/mesh-optimizer
miniapps/meshing/pmesh-optimizer
miniapps/meshing/pmesh-fitting
@@ -262,6 +263,8 @@ miniapps/meshing/mesh.*
miniapps/meshing/order.*
miniapps/meshing/sol.*
miniapps/meshing/refined.mesh
miniapps/meshing/bounding-box*
miniapps/meshing/jacobian-determinant*
miniapps/mtop/parheat
miniapps/mtop/ParHeat*
@@ -297,6 +300,7 @@ miniapps/nurbs/nurbs_solenoidal
miniapps/nurbs/nurbs_printfunc
miniapps/nurbs/nurbs_patch_ex1
miniapps/nurbs/nurbs_curveint
miniapps/nurbs/nurbs_surface
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/sol_?.gf
@@ -315,6 +319,7 @@ miniapps/nurbs/nurbs_naca_cmesh
miniapps/nurbs/naca-cmesh.mesh
miniapps/nurbs/glvis_naca-cmesh.mesh
miniapps/nurbs/Naca_cmesh
miniapps/nurbs/*-Surface.mesh
miniapps/performance/ex1
miniapps/performance/ex1p
@@ -336,6 +341,7 @@ miniapps/shifted/lsf_integral
miniapps/tools/display-basis
miniapps/tools/load-dc
miniapps/tools/convert-dc
miniapps/tools/gridfunction-bounds
miniapps/tools/lor-transfer
miniapps/tools/plor-transfer
miniapps/tools/get-values
@@ -402,12 +408,15 @@ miniapps/spde/ParaView
miniapps/tribol/contact-patch-test
miniapps/diag-smoothers/abs-l1-jacobi
miniapps/diag-smoothers/mg-abs-l1-jacobi
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
tests/unit/punit_tests
tests/unit/cunit_tests
tests/unit/pcunit_tests
tests/unit/gpu_unit_tests
tests/unit/pgpu_unit_tests
tests/unit/sedov_tests_*
tests/unit/psedov_tests_*
tests/unit/tmop_pa_tests_*
+31 -1
View File
@@ -29,9 +29,14 @@ Discretization improvements
Meshing improvements
--------------------
- Added support for higher order meshes in Mesh::MakeSimplicial and
ParMesh::MakeSimplicial.
- Added a new miniapp for interpolating a surface grid of points in 3D using a
smooth NURBS surface, that can then be sampled at arbitrary resolution while
staying close to the original geometry. See miniapps/nurbs/nurbs_surface.
GPU computing
-------------
- The function Vector::SetSubVector(const Array<int> &, const real_t) now
@@ -39,12 +44,37 @@ GPU computing
set. This is most often used for setting constant essential boundary
conditions. A new function Vector::SetSubVectorHost has been added in cases
where host execution is always needed (e.g. when the DOFs array is small).
- Introduced MFEM_FOREACH_THREAD_DIRECT, which directly maps loop tasks to GPU
threads, assigning one task per thread.
API changes:
New and updated examples and miniapps
-------------------------------------
- Added miniapps to demonstrate an implementation of the absolute-value
L(1)-Jacobi preconditioners in partially assembled operators. This includes
Multigrid wrapper to demonstrate the effectiveness of these Jacobi-type
operators as smoothers.
These miniapps can be found in `miniapps/diag-smoothers`.
API changes
-----------
- mfem::internal::tensor and mfem::internal::dual have been moved to
mfem::future::tensor and mfem::future::dual.
- API addition: in class `Operator`, added virtual functions: `AbsMult`, and
`AbsMultTranspose`; in class `Vector`, added `Abs` and `Pow`.
Miscellaneous
-------------
- Added the "gpu", "raja-gpu", and "ceed-gpu" backend aliases/shortcuts which
automatically select between CUDA or HIP.
- The CUDA-specific names used by some of the unit tests like 'cunit_tests' and
'pcunit_tests' were replaced by names using 'gpu' instead of 'c' (short for
CUDA) or 'cuda'. These tests automatically run the CUDA/HIP tests based on the
MFEM build configuration.
- Added the option to enable GPU-aware MPI in MFEM using the environment
variable 'MFEM_GPU_AWARE_MPI' set to any value. Setting this environment
variable is an alternative to calling 'Device::SetGPUAwareMPI(true)'.
- Added parallel Address Sanitizer, serial and parallel Undefined Behavior
Sanitizer and serial Memory Sanitizer GitHub actions tests on Ubuntu.
Version 4.8, released on Apr 9, 2025
====================================
+1 -1
View File
@@ -115,7 +115,7 @@ vertices
nodes
FiniteElementSpace
FiniteElementCollection: Quadratic
FiniteElementCollection: H1_3D_P2
VDim: 3
Ordering: 0
+1 -1
View File
@@ -56,7 +56,7 @@ vertices
nodes
FiniteElementSpace
FiniteElementCollection: Quadratic
FiniteElementCollection: H1_3D_P2
VDim: 3
Ordering: 0
+1 -1
View File
@@ -227,7 +227,7 @@ vertices
nodes
FiniteElementSpace
FiniteElementCollection: Quadratic
FiniteElementCollection: H1_2D_P2
VDim: 2
Ordering: 0
+1 -1
View File
@@ -65,7 +65,7 @@ vertices
nodes
FiniteElementSpace
FiniteElementCollection: Quadratic
FiniteElementCollection: H1_2D_P2
VDim: 2
Ordering: 0
+35 -76
View File
@@ -62,14 +62,9 @@ static real_t epsilon_ = 1.0;
static real_t sigma_ = 20.0;
static real_t omega_ = 10.0;
real_t u0_real_exact(const Vector &);
real_t u0_imag_exact(const Vector &);
void u1_real_exact(const Vector &, Vector &);
void u1_imag_exact(const Vector &, Vector &);
void u2_real_exact(const Vector &, Vector &);
void u2_imag_exact(const Vector &, Vector &);
complex<real_t> u0_exact(const Vector &x);
void u1_exact(const Vector &, ComplexVector &);
void u2_exact(const Vector &, ComplexVector &);
bool check_for_inline_mesh(const char * mesh_file);
@@ -215,54 +210,48 @@ int main(int argc, char *argv[])
ComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ComplexGridFunction(fespace); }
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
ComplexFunctionCoefficient u0(u0_exact);
ComplexVectorFunctionCoefficient u1(dim, u1_exact);
ComplexVectorFunctionCoefficient u2(dim, u2_exact);
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
ComplexConstantCoefficient oneCoef(1.0);
Vector zeroVec(dim); zeroVec = 0.0;
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
ComplexVectorConstantCoefficient oneVecCoef(oneVec);
switch (prob)
{
case 0:
if (exact_sol)
{
u.ProjectBdrCoefficient(u0_r, u0_i, ess_bdr);
u_exact->ProjectCoefficient(u0_r, u0_i);
u.ProjectBdrCoefficient(u0, ess_bdr);
u_exact->ProjectCoefficient(u0);
}
else
{
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
u.ProjectBdrCoefficient(oneCoef, ess_bdr);
}
break;
case 1:
if (exact_sol)
{
u.ProjectBdrCoefficientTangent(u1_r, u1_i, ess_bdr);
u_exact->ProjectCoefficient(u1_r, u1_i);
u.ProjectBdrCoefficientTangent(u1, ess_bdr);
u_exact->ProjectCoefficient(u1);
}
else
{
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
u.ProjectBdrCoefficientTangent(oneVecCoef, ess_bdr);
}
break;
case 2:
if (exact_sol)
{
u.ProjectBdrCoefficientNormal(u2_r, u2_i, ess_bdr);
u_exact->ProjectCoefficient(u2_r, u2_i);
u.ProjectBdrCoefficientNormal(u2, ess_bdr);
u_exact->ProjectCoefficient(u2);
}
else
{
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
u.ProjectBdrCoefficientNormal(oneVecCoef, ess_bdr);
}
break;
default: break; // This should be unreachable
@@ -300,27 +289,24 @@ int main(int argc, char *argv[])
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ComplexConstantCoefficient complexMassCoef(-omega_ * omega_ * epsilon_,
omega_ * sigma_);
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
a->AddDomainIntegrator<DiffusionIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<MassIntegrator>(complexMassCoef);
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
a->AddDomainIntegrator<CurlCurlIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<VectorFEMassIntegrator>(complexMassCoef);
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
a->AddDomainIntegrator<DivDivIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<VectorFEMassIntegrator>(complexMassCoef);
break;
default: break; // This should be unreachable
}
@@ -436,29 +422,24 @@ int main(int argc, char *argv[])
if (exact_sol)
{
real_t err_r = -1.0;
real_t err_i = -1.0;
real_t err_u = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
err_u = u.ComputeL2Error(u0);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
err_u = u.ComputeL2Error(u1);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
err_u = u.ComputeL2Error(u2);
break;
default: break; // This should be unreachable
}
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << "|| u_h - u ||_{L^2} = " << err_u << endl;
cout << endl;
}
@@ -564,36 +545,14 @@ complex<real_t> u0_exact(const Vector &x)
return std::exp(-i * kappa * x[dim - 1]);
}
real_t u0_real_exact(const Vector &x)
{
return u0_exact(x).real();
}
real_t u0_imag_exact(const Vector &x)
{
return u0_exact(x).imag();
}
void u1_real_exact(const Vector &x, Vector &v)
void u1_exact(const Vector &x, ComplexVector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_real_exact(x);
v.SetSize(dim); v = 0.0; v[0] = u0_exact(x);
}
void u1_imag_exact(const Vector &x, Vector &v)
void u2_exact(const Vector &x, ComplexVector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_real_exact(x);
}
void u2_imag_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_imag_exact(x);
v.SetSize(dim); v = 0.0; v[dim-1] = u0_exact(x);
}
+50 -33
View File
@@ -62,6 +62,10 @@ static real_t epsilon_ = 1.0;
static real_t sigma_ = 20.0;
static real_t omega_ = 10.0;
complex<real_t> u0_exact(const Vector &x);
void u1_exact(const Vector &, ComplexVector &);
void u2_exact(const Vector &, ComplexVector &);
real_t u0_real_exact(const Vector &);
real_t u0_imag_exact(const Vector &);
@@ -244,13 +248,22 @@ int main(int argc, char *argv[])
ParComplexGridFunction * u_exact = NULL;
if (exact_sol) { u_exact = new ParComplexGridFunction(fespace); }
ComplexFunctionCoefficient u0(u0_exact);
ComplexVectorFunctionCoefficient u1(dim, u1_exact);
ComplexVectorFunctionCoefficient u2(dim, u2_exact);
ComplexConstantCoefficient oneCoef(1.0);
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
ComplexVectorConstantCoefficient oneVecCoef(oneVec);
FunctionCoefficient u0_r(u0_real_exact);
FunctionCoefficient u0_i(u0_imag_exact);
VectorFunctionCoefficient u1_r(dim, u1_real_exact);
VectorFunctionCoefficient u1_i(dim, u1_imag_exact);
VectorFunctionCoefficient u2_r(dim, u2_real_exact);
VectorFunctionCoefficient u2_i(dim, u2_imag_exact);
/*
ConstantCoefficient zeroCoef(0.0);
ConstantCoefficient oneCoef(1.0);
@@ -258,40 +271,40 @@ int main(int argc, char *argv[])
Vector oneVec(dim); oneVec = 0.0; oneVec[(prob==2)?(dim-1):0] = 1.0;
VectorConstantCoefficient zeroVecCoef(zeroVec);
VectorConstantCoefficient oneVecCoef(oneVec);
*/
switch (prob)
{
case 0:
if (exact_sol)
{
u.ProjectBdrCoefficient(u0_r, u0_i, ess_bdr);
u_exact->ProjectCoefficient(u0_r, u0_i);
u.ProjectBdrCoefficient(u0, ess_bdr);
u_exact->ProjectCoefficient(u0);
}
else
{
u.ProjectBdrCoefficient(oneCoef, zeroCoef, ess_bdr);
u.ProjectBdrCoefficient(oneCoef, ess_bdr);
}
break;
case 1:
if (exact_sol)
{
u.ProjectBdrCoefficientTangent(u1_r, u1_i, ess_bdr);
u_exact->ProjectCoefficient(u1_r, u1_i);
u.ProjectBdrCoefficientTangent(u1, ess_bdr);
u_exact->ProjectCoefficient(u1);
}
else
{
u.ProjectBdrCoefficientTangent(oneVecCoef, zeroVecCoef, ess_bdr);
u.ProjectBdrCoefficientTangent(oneVecCoef, ess_bdr);
}
break;
case 2:
if (exact_sol)
{
u.ProjectBdrCoefficientNormal(u2_r, u2_i, ess_bdr);
u_exact->ProjectCoefficient(u2_r, u2_i);
u.ProjectBdrCoefficientNormal(u2, ess_bdr);
u_exact->ProjectCoefficient(u2);
}
else
{
u.ProjectBdrCoefficientNormal(oneVecCoef, zeroVecCoef, ess_bdr);
u.ProjectBdrCoefficientNormal(oneVecCoef, ess_bdr);
}
break;
default: break; // This should be unreachable
@@ -331,27 +344,24 @@ int main(int argc, char *argv[])
ConstantCoefficient lossCoef(omega_ * sigma_);
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
ComplexConstantCoefficient complexMassCoef(-omega_ * omega_ * epsilon_,
omega_ * sigma_);
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
switch (prob)
{
case 0:
a->AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new MassIntegrator(massCoef),
new MassIntegrator(lossCoef));
a->AddDomainIntegrator<DiffusionIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<MassIntegrator>(complexMassCoef);
break;
case 1:
a->AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
a->AddDomainIntegrator<CurlCurlIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<VectorFEMassIntegrator>(complexMassCoef);
break;
case 2:
a->AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
NULL);
a->AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
new VectorFEMassIntegrator(lossCoef));
a->AddDomainIntegrator<DivDivIntegrator>(stiffnessCoef);
a->AddDomainIntegrator<VectorFEMassIntegrator>(complexMassCoef);
break;
default: break; // This should be unreachable
}
@@ -475,22 +485,18 @@ int main(int argc, char *argv[])
if (exact_sol)
{
real_t err_r = -1.0;
real_t err_i = -1.0;
real_t err_u = -1.0;
switch (prob)
{
case 0:
err_r = u.real().ComputeL2Error(u0_r);
err_i = u.imag().ComputeL2Error(u0_i);
err_u = u.ComputeL2Error(u0);
break;
case 1:
err_r = u.real().ComputeL2Error(u1_r);
err_i = u.imag().ComputeL2Error(u1_i);
err_u = u.ComputeL2Error(u1);
break;
case 2:
err_r = u.real().ComputeL2Error(u2_r);
err_i = u.imag().ComputeL2Error(u2_i);
err_u = u.ComputeL2Error(u2);
break;
default: break; // This should be unreachable
}
@@ -498,8 +504,7 @@ int main(int argc, char *argv[])
if ( myid == 0 )
{
cout << endl;
cout << "|| Re (u_h - u) ||_{L^2} = " << err_r << endl;
cout << "|| Im (u_h - u) ||_{L^2} = " << err_i << endl;
cout << "|| u_h - u ||_{L^2} = " << err_u << endl;
cout << endl;
}
}
@@ -627,6 +632,12 @@ real_t u0_imag_exact(const Vector &x)
return u0_exact(x).imag();
}
void u1_exact(const Vector &x, ComplexVector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[0] = u0_exact(x);
}
void u1_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
@@ -639,6 +650,12 @@ void u1_imag_exact(const Vector &x, Vector &v)
v.SetSize(dim); v = 0.0; v[0] = u0_imag_exact(x);
}
void u2_exact(const Vector &x, ComplexVector &v)
{
int dim = x.Size();
v.SetSize(dim); v = 0.0; v[dim-1] = u0_exact(x);
}
void u2_real_exact(const Vector &x, Vector &v)
{
int dim = x.Size();
+5
View File
@@ -59,6 +59,7 @@ set(SRCS
integ/nonlininteg_vecconvection_pa.cpp
integ/nonlininteg_vecconvection_mf.cpp
coefficient.cpp
complex_coefficient.cpp
complex_fem.cpp
convergence.cpp
datacollection.cpp
@@ -162,6 +163,7 @@ set(SRCS
transfer.cpp
hyperbolic.cpp
integrator.cpp
bounds.cpp
)
set(HDRS
@@ -175,6 +177,7 @@ set(HDRS
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
coefficient.hpp
complex_coefficient.hpp
complex_fem.hpp
convergence.hpp
datacollection.hpp
@@ -246,6 +249,7 @@ set(HDRS
nonlinearform_ext.hpp
nonlininteg.hpp
qfunction.hpp
qinterp/det.hpp
qinterp/eval.hpp
qinterp/eval_hdiv.hpp
qinterp/grad.hpp
@@ -272,6 +276,7 @@ set(HDRS
transfer.hpp
hyperbolic.hpp
integrator.hpp
bounds.hpp
)
if (MFEM_USE_SIDRE)
+1
View File
@@ -515,6 +515,7 @@ struct InvTNewtonSolver<Geometry::SEGMENT, SDim, SType, max_team_x>
phys_tol += pptr[idx + d * npts] * pptr[idx + d * npts];
}
phys_tol = fmax(phys_rtol * phys_rtol, phys_tol * phys_rtol * phys_rtol);
hit_bdr[0] = prev_hit_bdr[0] = false;
}
// for each iteration
while (true)
+208 -170
View File
@@ -78,7 +78,7 @@ void MFBilinearFormExtension::AssembleDiagonal(Vector &y) const
dynamic_cast<const ElementRestriction*>(elem_restrict);
if (H1elem_restrict)
{
H1elem_restrict->MultTransposeUnsigned(localY, y);
H1elem_restrict->AbsMultTranspose(localY, y);
}
else
{
@@ -456,7 +456,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
dynamic_cast<const ElementRestriction*>(elem_restrict);
if (H1elem_restrict)
{
H1elem_restrict->MultTransposeUnsigned(localY, y);
H1elem_restrict->AbsMultTranspose(localY, y);
}
else
{
@@ -491,7 +491,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
assemble_diagonal_with_markers(*bdr_integs[i], bdr_markers[i],
bdr_attributes, bdr_face_Y);
}
bdr_face_restrict_lex->AddMultTransposeUnsigned(bdr_face_Y, y);
bdr_face_restrict_lex->AddAbsMultTranspose(bdr_face_Y, y);
}
}
@@ -526,7 +526,8 @@ void PABilinearFormExtension::FormLinearSystem(const Array<int> &ess_tdof_list,
A.Reset(oper); // A will own oper
}
void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
void PABilinearFormExtension::MultInternal(const Vector &x, Vector &y,
const bool useAbs) const
{
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
@@ -558,11 +559,13 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
{
if (integrators[i]->Patchwise())
{
MFEM_ASSERT(!useAbs, "AbsMult not implemented with NURBS!")
integrators[i]->AddMultNURBSPA(x, y);
}
else
{
integrators[i]->AddMultPA(x, y);
if (useAbs) { integrators[i]->AddAbsMultPA(x, y); }
else { integrators[i]->AddMultPA(x, y); }
}
}
}
@@ -571,14 +574,30 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
if (iSz)
{
Array<Array<int>*> &elem_markers = *a->GetDBFI_Marker();
elem_restrict->Mult(x, localX);
auto H1elem_restrict =
dynamic_cast<const ElementRestriction*>(elem_restrict);
if (H1elem_restrict && useAbs)
{
H1elem_restrict->AbsMult(x, localX);
}
else
{
elem_restrict->Mult(x, localX);
}
localY = 0.0;
for (int i = 0; i < iSz; ++i)
{
AddMultWithMarkers(*integrators[i], localX, elem_markers[i],
elem_attributes, false, localY);
elem_attributes, false, localY, useAbs);
}
if (H1elem_restrict && useAbs)
{
H1elem_restrict->AbsMultTranspose(localY, y);
}
else
{
elem_restrict->MultTranspose(localY, y);
}
elem_restrict->MultTranspose(localY, y);
}
else
{
@@ -590,6 +609,7 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
const int iFISz = intFaceIntegrators.Size();
if (int_face_restrict_lex && iFISz>0)
{
MFEM_ASSERT(!useAbs, "AbsMult not implemented for face integrators!")
// When assembling interior face integrators for DG spaces, we need to
// exchange the face-neighbor information. This happens inside member
// functions of the 'int_face_restrict_lex'. To avoid repeated calls to
@@ -651,6 +671,7 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
const bool has_bdr_integs = (n_bdr_face_integs > 0 || n_bdr_integs > 0);
if (bdr_face_restrict_lex && has_bdr_integs)
{
MFEM_ASSERT(!useAbs, "AbsMult not implemented for bdr integrators!")
Array<Array<int>*> &bdr_markers = *a->GetBBFI_Marker();
Array<Array<int>*> &bdr_face_markers = *a->GetBFBFI_Marker();
bdr_face_restrict_lex->Mult(x, bdr_face_X);
@@ -828,22 +849,39 @@ void PABilinearFormExtension::AddMultWithMarkers(
const Array<int> *markers,
const Array<int> &attributes,
const bool transpose,
Vector &y) const
Vector &y,
const bool useAbs) const
{
if (markers)
{
tmp_evec.SetSize(y.Size());
tmp_evec = 0.0;
if (transpose) { integ.AddMultTransposePA(x, tmp_evec); }
else { integ.AddMultPA(x, tmp_evec); }
if (useAbs)
{
if (transpose) { integ.AddAbsMultTransposePA(x, tmp_evec); }
else { integ.AddAbsMultPA(x, tmp_evec); }
}
else
{
if (transpose) { integ.AddMultTransposePA(x, tmp_evec); }
else { integ.AddMultPA(x, tmp_evec); }
}
const int ne = attributes.Size();
const int nd = x.Size() / ne;
AddWithMarkers_(ne, nd, tmp_evec, *markers, attributes, y);
}
else
{
if (transpose) { integ.AddMultTransposePA(x, y); }
else { integ.AddMultPA(x, y); }
if (useAbs)
{
if (transpose) { integ.AddAbsMultTransposePA(x, y); }
else { integ.AddAbsMultPA(x, y); }
}
else
{
if (transpose) { integ.AddMultTransposePA(x, y); }
else { integ.AddMultPA(x, y); }
}
}
}
@@ -1010,8 +1048,13 @@ void EABilinearFormExtension::Assemble()
}
}
void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
void EABilinearFormExtension::MultInternal(const Vector &x, Vector &y,
const bool useTranspose,
const bool useAbs) const
{
auto elemRest = dynamic_cast<const ElementRestriction*>(elem_restrict);
MFEM_ASSERT(useAbs?(elemRest!=nullptr):true,
"elem_restrict is not ElementRestriction*!")
// Apply the Element Restriction
const bool useRestrict = !DeviceCanUseCeed() && elem_restrict;
if (!useRestrict)
@@ -1019,6 +1062,11 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
}
else if (useAbs)
{
elemRest->AbsMult(x, localX);
localY = 0.0;
}
else
{
elem_restrict->Mult(x, localX);
@@ -1026,25 +1074,55 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
}
// Apply the Element Matrices
{
Vector abs_ea_data;
if (useAbs)
{
abs_ea_data = ea_data;
abs_ea_data.Abs();
}
const int NDOFS = elemDofs;
auto X = Reshape(useRestrict?localX.Read():x.Read(), NDOFS, ne);
auto Y = Reshape(useRestrict?localY.ReadWrite():y.ReadWrite(), NDOFS, ne);
auto A = Reshape(ea_data.Read(), NDOFS, NDOFS, ne);
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
auto A = Reshape(useAbs?abs_ea_data.Read():ea_data.Read(), NDOFS, NDOFS, ne);
if (!useTranspose)
{
const int e = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
res += A(i, j, e)*X(i, e);
}
Y(j, e) += res;
});
const int e = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A(i, j, e)*X(i, e);
}
Y(j, e) += res;
});
}
else
{
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int e = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A(j, i, e)*X(i, e);
}
Y(j, e) += res;
});
}
// Apply the Element Restriction transposed
if (useRestrict)
{
elem_restrict->MultTranspose(localY, y);
if (useAbs)
{
elemRest->AbsMultTranspose(localY, y);
}
else
{
elem_restrict->MultTranspose(localY, y);
}
}
}
@@ -1053,6 +1131,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
const int iFISz = intFaceIntegrators.Size();
if (int_face_restrict_lex && iFISz>0)
{
MFEM_VERIFY(!useAbs, "AbsMult not implemented with Face integrators!")
// Apply the Interior Face Restriction
int_face_restrict_lex->Mult(x, int_face_X);
if (int_face_X.Size()>0)
@@ -1064,7 +1143,65 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
auto Y = Reshape(int_face_Y.ReadWrite(), NDOFS, 2, nf_int);
if (!factorize_face_terms)
{
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
Vector abs_ea_data_int(ea_data_int.Size());
if (useAbs)
{
abs_ea_data_int = ea_data_int;
abs_ea_data_int.Abs();
}
auto A_int = Reshape(useAbs?abs_ea_data_int.Read():ea_data_int.Read(),
NDOFS, NDOFS, 2, nf_int);
if (!useTranspose)
{
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
else
{
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
}
Vector abs_ea_data_ext(ea_data_ext.Size());
if (useAbs)
{
abs_ea_data_ext = ea_data_ext;
abs_ea_data_ext.Abs();
}
auto A_ext = Reshape(useAbs?abs_ea_data_ext.Read():ea_data_ext.Read(),
NDOFS, NDOFS, 2, nf_int);
if (!useTranspose)
{
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
@@ -1072,35 +1209,37 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 0, f)*X(i, 0, f);
res += A_ext(i, j, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
Y(j, 1, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(i, j, 1, f)*X(i, 1, f);
res += A_ext(i, j, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
Y(j, 0, f) += res;
});
}
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
else
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
res += A_ext(i, j, 0, f)*X(i, 0, f);
}
Y(j, 1, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_ext(i, j, 1, f)*X(i, 1, f);
}
Y(j, 0, f) += res;
});
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_ext(j, i, 1, f)*X(i, 0, f);
}
Y(j, 1, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_ext(j, i, 0, f)*X(i, 1, f);
}
Y(j, 0, f) += res;
});
}
// Apply the Interior Face Restriction transposed
int_face_restrict_lex->AddMultTransposeInPlace(int_face_Y, y);
}
@@ -1109,7 +1248,9 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
// Treatment of boundary faces
if (!factorize_face_terms && bdr_face_restrict_lex && ea_data_bdr.Size() > 0)
{
MFEM_ASSERT(!useAbs, "AbsMult not implemented with Face integrators!")
// Apply the Boundary Face Restriction
// TODO: AbsMult if needed
bdr_face_restrict_lex->Mult(x, bdr_face_X);
bdr_face_Y = 0.0;
// Apply the boundary face matrices
@@ -1117,141 +1258,38 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
if (!useTranspose)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A(i, j, f)*X(i, f);
}
Y(j, f) += res;
});
// Apply the Boundary Face Restriction transposed
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
}
}
void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
{
// Apply the Element Restriction
const bool useRestrict = !DeviceCanUseCeed() && elem_restrict;
if (!useRestrict)
{
y.UseDevice(true); // typically this is a large vector, so store on device
y = 0.0;
}
else
{
elem_restrict->Mult(x, localX);
localY = 0.0;
}
// Apply the Element Matrices transposed
{
const int NDOFS = elemDofs;
auto X = Reshape(useRestrict?localX.Read():x.Read(), NDOFS, ne);
auto Y = Reshape(useRestrict?localY.ReadWrite():y.ReadWrite(), NDOFS, ne);
auto A = Reshape(ea_data.Read(), NDOFS, NDOFS, ne);
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int e = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A(j, i, e)*X(i, e);
}
Y(j, e) += res;
});
// Apply the Element Restriction transposed
if (useRestrict)
{
elem_restrict->MultTranspose(localY, y);
}
}
// Treatment of interior faces
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
const int iFISz = intFaceIntegrators.Size();
if (int_face_restrict_lex && iFISz>0)
{
// Apply the Interior Face Restriction
int_face_restrict_lex->Mult(x, int_face_X);
if (int_face_X.Size()>0)
{
int_face_Y = 0.0;
// Apply the interior face matrices transposed
const int NDOFS = faceDofs;
auto X = Reshape(int_face_X.Read(), NDOFS, 2, nf_int);
auto Y = Reshape(int_face_Y.ReadWrite(), NDOFS, 2, nf_int);
if (!factorize_face_terms)
{
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 0, f)*X(i, 0, f);
}
Y(j, 0, f) += res;
res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_int(j, i, 1, f)*X(i, 1, f);
}
Y(j, 1, f) += res;
});
}
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
// TODO: useAbs
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_ext(j, i, 1, f)*X(i, 0, f);
res += A(i, j, f)*X(i, f);
}
Y(j, 1, f) += res;
res = 0.0;
Y(j, f) += res;
});
}
else
{
// TODO: useAbs
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A_ext(j, i, 0, f)*X(i, 1, f);
res += A(j, i, f)*X(i, f);
}
Y(j, 0, f) += res;
Y(j, f) += res;
});
// Apply the Interior Face Restriction transposed
int_face_restrict_lex->AddMultTransposeInPlace(int_face_Y, y);
}
}
// Treatment of boundary faces
if (!factorize_face_terms && bdr_face_restrict_lex && ea_data_bdr.Size() > 0)
{
// Apply the Boundary Face Restriction
bdr_face_restrict_lex->Mult(x, bdr_face_X);
bdr_face_Y = 0.0;
// Apply the boundary face matrices transposed
const int NDOFS = faceDofs;
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
{
const int f = glob_j/NDOFS;
const int j = glob_j%NDOFS;
real_t res = 0.0;
for (int i = 0; i < NDOFS; i++)
{
res += A(j, i, f)*X(i, f);
}
Y(j, f) += res;
});
// Apply the Boundary Face Restriction transposed
// TODO: AbsMultTranspose if needed
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
}
}
@@ -1911,7 +1949,7 @@ void PAMixedBilinearFormExtension::AssembleDiagonal_ADAt(const Vector &D,
dynamic_cast<const ElementRestriction*>(elem_restrict_trial);
if (H1elem_restrict_trial)
{
H1elem_restrict_trial->MultUnsigned(D, localTrial);
H1elem_restrict_trial->AbsMult(D, localTrial);
}
else
{
@@ -1937,7 +1975,7 @@ void PAMixedBilinearFormExtension::AssembleDiagonal_ADAt(const Vector &D,
dynamic_cast<const ElementRestriction*>(elem_restrict_test);
if (H1elem_restrict_test)
{
H1elem_restrict_test->MultTransposeUnsigned(localTest, diag);
H1elem_restrict_test->AbsMultTranspose(localTest, diag);
}
else
{
@@ -1993,7 +2031,7 @@ void PADiscreteLinearOperatorExtension::Assemble()
dynamic_cast<const ElementRestriction*>(elem_restrict_test);
if (elem_restrict)
{
elem_restrict->MultTransposeUnsigned(ones, test_multiplicity);
elem_restrict->AbsMultTranspose(ones, test_multiplicity);
}
else
{
+22 -4
View File
@@ -91,12 +91,17 @@ public:
Vector &x, Vector &b,
OperatorHandle &A, Vector &X, Vector &B,
int copy_interior = 0) override;
void Mult(const Vector &x, Vector &y) const override;
void Mult(const Vector &x, Vector &y) const override
{ MultInternal(x,y); }
void AbsMult(const Vector &x, Vector &y) const override
{ MultInternal(x,y, true); }
void MultTranspose(const Vector &x, Vector &y) const override;
void Update() override;
protected:
void SetupRestrictionOperators(const L2FaceValues m);
void MultInternal(const Vector &x, Vector &y,
const bool useAbs = false) const;
/// @brief Accumulate the action (or transpose) of the integrator on @a x
/// into @a y, taking into account the (possibly null) @a markers array.
@@ -110,12 +115,14 @@ protected:
/// @param attributes Array of element or boundary element attributes.
/// @param transpose Compute the action or transpose of the integrator .
/// @param y Output E-vector
/// @param useAbs Apply absolute-value operator
void AddMultWithMarkers(const BilinearFormIntegrator &integ,
const Vector &x,
const Array<int> *markers,
const Array<int> &attributes,
const bool transpose,
Vector &y) const;
Vector &y,
const bool useAbs = false) const;
/// @brief Performs the same function as AddMultWithMarkers, but takes as
/// input and output face normal derivatives.
@@ -152,8 +159,15 @@ public:
EABilinearFormExtension(BilinearForm *form);
void Assemble() override;
void Mult(const Vector &x, Vector &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override;
void Mult(const Vector &x, Vector &y) const override
{ MultInternal(x, y, false); }
void AbsMult(const Vector &x, Vector &y) const override
{ MultInternal(x, y, false, true); }
void MultTranspose(const Vector &x, Vector &y) const override
{ MultInternal(x, y, true); }
void AbsMultTranspose(const Vector &x, Vector &y) const override
{ MultInternal(x, y, true, true); }
/// @brief Populates @a element_matrices with the element matrices.
///
@@ -165,6 +179,10 @@ public:
void GetElementMatrices(DenseTensor &element_matrices,
ElementDofOrdering ordering,
bool add_bdr);
// This method needs to be public due to 'nvcc' restriction.
void MultInternal(const Vector &x, Vector &y, const bool useTranspose,
const bool useAbs = false) const;
};
/// Data and methods for fully-assembled bilinear forms
+29
View File
@@ -121,6 +121,12 @@ void BilinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
" is not implemented for this class.");
}
void BilinearFormIntegrator::AddAbsMultPA(const Vector &, Vector &) const
{
MFEM_ABORT("BilinearFormIntegrator:AddAbsMultPA:(...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AddMultNURBSPA(const Vector &, Vector &) const
{
MFEM_ABORT("BilinearFormIntegrator::AddMultNURBSPA(...)\n"
@@ -133,6 +139,13 @@ void BilinearFormIntegrator::AddMultTransposePA(const Vector &, Vector &) const
" is not implemented for this class.");
}
void BilinearFormIntegrator::AddAbsMultTransposePA(const Vector &,
Vector &) const
{
MFEM_ABORT("BilinearFormIntegrator::AddAbsMultTransposePA(...)\n"
" is not implemented for this class.");
}
void BilinearFormIntegrator::AssembleMF(const FiniteElementSpace &fes)
{
MFEM_ABORT("BilinearFormIntegrator::AssembleMF(...)\n"
@@ -418,6 +431,14 @@ void SumIntegrator::AddMultPA(const Vector& x, Vector& y) const
}
}
void SumIntegrator::AddAbsMultPA(const Vector& x, Vector& y) const
{
for (int i = 0; i < integrators.Size(); i++)
{
integrators[i]->AddAbsMultPA(x, y);
}
}
void SumIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
{
for (int i = 0; i < integrators.Size(); i++)
@@ -426,6 +447,14 @@ void SumIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
}
}
void SumIntegrator::AddAbsMultTransposePA(const Vector &x, Vector &y) const
{
for (int i = 0; i < integrators.Size(); i++)
{
integrators[i]->AddAbsMultTransposePA(x, y);
}
}
void SumIntegrator::AssembleMF(const FiniteElementSpace &fes)
{
for (int i = 0; i < integrators.Size(); i++)
+18
View File
@@ -78,6 +78,8 @@ public:
called. */
void AddMultPA(const Vector &x, Vector &y) const override;
virtual void AddAbsMultPA(const Vector &x, Vector &y) const;
/// Method for partially assembled action on NURBS patches.
virtual void AddMultNURBSPA(const Vector&x, Vector&y) const;
@@ -90,6 +92,8 @@ public:
called. */
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
virtual void AddAbsMultTransposePA(const Vector &x, Vector &y) const;
/// Method defining element assembly.
/** The result of the element assembly is added to the @a emat Vector if
@a add is true. Otherwise, if @a add is false, we set @a emat. */
@@ -496,8 +500,12 @@ public:
void AddMultTransposePA(const Vector &x, Vector &y) const override;
void AddAbsMultTransposePA(const Vector &x, Vector &y) const override;
void AddMultPA(const Vector& x, Vector& y) const override;
void AddAbsMultPA(const Vector& x, Vector& y) const override;
void AssembleMF(const FiniteElementSpace &fes) override;
void AddMultMF(const Vector &x, Vector &y) const override;
@@ -2320,8 +2328,12 @@ public:
void AddMultPA(const Vector&, Vector&) const override;
void AddAbsMultPA(const Vector&, Vector&) const override;
void AddMultTransposePA(const Vector&, Vector&) const override;
void AddAbsMultTransposePA(const Vector&, Vector&) const override;
void AddMultNURBSPA(const Vector&, Vector&) const override;
void AddMultPatchPA(const int patch, const Vector &x, Vector &y) const;
@@ -2419,8 +2431,12 @@ public:
void AddMultPA(const Vector&, Vector&) const override;
void AddAbsMultPA(const Vector&, Vector&) const override;
void AddMultTransposePA(const Vector&, Vector&) const override;
void AddAbsMultTransposePA(const Vector&, Vector&) const override;
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
const ElementTransformation &Trans);
@@ -2816,6 +2832,7 @@ public:
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AddMultPA(const Vector &x, Vector &y) const override;
void AddAbsMultPA(const Vector &x, Vector &y) const override;
void AssembleDiagonalPA(Vector& diag) override;
const Coefficient *GetCoefficient() const { return Q; }
@@ -2933,6 +2950,7 @@ public:
void AssemblePA(const FiniteElementSpace &trial_fes,
const FiniteElementSpace &test_fes) override;
void AddMultPA(const Vector &x, Vector &y) const override;
void AddAbsMultPA(const Vector &x, Vector &y) const override;
void AddMultTransposePA(const Vector &x, Vector &y) const override;
void AssembleDiagonalPA(Vector& diag) override;
void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
+715
View File
@@ -0,0 +1,715 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
// Implementation of bounds
#include "bounds.hpp"
#include <limits>
#include <cstring>
#include <string>
#include <cmath>
#include <iostream>
#include <algorithm>
namespace mfem
{
using namespace std;
void PLBound::Setup(const int nb_i, const int ncp_i,
const int b_type_i, const int cp_type_i,
const real_t tol_i)
{
MFEM_VERIFY(b_type_i >= 0 && b_type_i <= 2, "Bases not supported. "
"Please read class description to see supported types.");
MFEM_VERIFY(cp_type_i == 0 || cp_type_i == 1,
"Control point type not supported. Please read class "
"description to see supported types.");
nb = nb_i;
ncp = ncp_i;
b_type = b_type_i;
cp_type = cp_type_i;
tol = tol_i;
lbound.SetSize(nb, ncp);
ubound.SetSize(nb, ncp);
nodes.SetSize(nb);
weights.SetSize(nb);
control_points.SetSize(ncp);
auto scalenodes = [](const Vector &in, const real_t a, const real_t b) -> Vector
{
Vector outVec(in.Size());
real_t maxv = in.Max();
real_t minv = in.Min();
for (int i = 0; i < in.Size(); i++)
{
outVec(i) = a + (b-a)*(in(i)-minv)/(maxv-minv);
}
return outVec;
};
MFEM_VERIFY(ncp >= 2,"At least 2 control points are required.");
if (cp_type == 0) // GL + End Point
{
control_points(0) = 0.0;
control_points(ncp-1) = 1.0;
if (ncp > 2)
{
const real_t *x = poly1d.GetPoints(ncp-3, 0);
MFEM_VERIFY(x, "Error in getting points.");
for (int i = 0; i < ncp-2; i++)
{
control_points(i+1) = x[i];
}
}
}
else if (cp_type == 1) // Chebyshev
{
auto GetChebyshevNodes = [](int n) -> Vector
{
Vector cheb(n);
for (int i = 0; i < n; ++i)
{
cheb(i) = -cos(M_PI * (static_cast<real_t>(i) / (n - 1)));
}
return cheb;
};
control_points = GetChebyshevNodes(ncp);
}
else
{
MFEM_ABORT("Unsupported interval points. Use [0,1].\n");
}
control_points = scalenodes(control_points, 0.0, 1.0); // rescale to [0,1]
Poly_1D::Basis &basis1d(poly1d.GetBasis(nb-1, b_type));
// Initialize bounds
lbound = 0.0;
ubound = 0.0;
Vector bmv(nb), bpv(nb), bv(nb); // basis values
Vector bdmv(nb), bdpv(nb), bdv(nb); // basis derivative values
Vector vals(3);
// See Section 3.1.1 of https://arxiv.org/pdf/2501.12349 for explanation of
// procedure below.
for (int j = 0; j < ncp; j++)
{
real_t x = control_points(j);
real_t xm = x;
if (j != 0)
{
xm = 0.5*(control_points(j-1)+control_points(j));
}
real_t xp = x;
if (j != ncp-1)
{
xp = 0.5*(control_points(j)+control_points(j+1));
}
basis1d.Eval(xm, bmv, bdmv);
basis1d.Eval(xp, bpv, bdpv);
basis1d.Eval(x, bv);
real_t dm = x-xm;
real_t dp = x-xp;
for (int i = 0; i < nb; i++)
{
if (j == 0)
{
lbound(i, j) = bv(i);
ubound(i, j) = bv(i);
}
else if (j == ncp-1)
{
lbound(i, j) = bv(i);
ubound(i, j) = bv(i);
}
else
{
vals(0) = bv(i);
vals(1) = bmv(i) + dm*bdmv(i);
vals(2) = bpv(i) + dp*bdpv(i);
lbound(i, j) = vals.Min()-tol; // tolerance for good measure
ubound(i, j) = vals.Max()+tol; // tolerance for good measure
}
}
}
IntegrationRule irule(nb);
if (b_type == 0)
{
QuadratureFunctions1D::GaussLegendre(nb, &irule);
for (int i = 0; i < nb; i++)
{
weights(i) = irule.IntPoint(i).weight;
nodes(i) = irule.IntPoint(i).x;
}
}
else if (b_type == 1)
{
QuadratureFunctions1D::GaussLobatto(nb, &irule);
for (int i = 0; i < nb; i++)
{
weights(i) = irule.IntPoint(i).weight;
nodes(i) = irule.IntPoint(i).x;
}
}
else if (b_type == 2)
{
QuadratureFunctions1D::ClosedUniform(nb, &irule);
for (int i = 0; i < nb; i++)
{
weights(i) = irule.IntPoint(i).weight;
nodes(i) = irule.IntPoint(i).x;
}
}
if (b_type == 2)
{
nodes_int.SetSize(nb);
weights_int.SetSize(nb);
IntegrationRule irule_int(nb);
{
QuadratureFunctions1D::GaussLobatto(nb, &irule_int);
for (int i = 0; i < nb; i++)
{
weights_int(i) = irule_int.IntPoint(i).weight;
nodes_int(i) = irule_int.IntPoint(i).x;
}
}
SetupBernsteinBasisMat(basisMatNodes, nodes);
// Setup memory for lu factors
basisMatLU = basisMatNodes;
lu_ip.SetSize(nb);
// Compute lu factors
LUFactors lu(basisMatLU.GetData(), lu_ip.GetData());
bool factor = lu.Factor(nb);
MFEM_VERIFY(factor,"Failure in LU factorization in PLBound.");
// Setup the Bernstein basis matrix for the GLL integration points. This
// is used to compute linear fit.
SetupBernsteinBasisMat(basisMatInt, nodes_int);
}
else
{
nodes_int.SetDataAndSize(nodes.GetData(), nb);
weights_int.SetDataAndSize(weights.GetData(), nb);
}
}
PLBound::PLBound(FiniteElementSpace *fes, int ncp_i, int cp_type_i)
{
MFEM_VERIFY(!fes->IsVariableOrder(),
"Variable order meshes not yet supported.");
const char *name = fes->FEColl()->Name();
string cname = name;
cp_type = cp_type_i;
b_type = BasisType::Invalid;
nb = fes->GetMaxElementOrder()+1;
tol = 0.0;
int minncp = 2;
if (nb > 12)
{
minncp = 2*nb;
}
else if (!strncmp(name, "H1_", 3) && strncmp(name, "H1_Trace_", 9))
{
// H1 GLL
b_type = BasisType::GaussLobatto;
minncp = min_ncp_gll_x[cp_type][nb-2];
}
else if (!strncmp(name, "H1Pos_", 6) && strncmp(name, "H1Pos_Trace_", 12))
{
// H1 Positive
b_type = BasisType::Positive;
minncp = min_ncp_pos_x[cp_type][nb-2];
}
else if (!strncmp(name, "L2_", 3) && strncmp(name, "L2_T", 4))
{
// L2 Gauss-Legendre
b_type = BasisType::GaussLegendre;
minncp = min_ncp_gl_x[cp_type][nb-2];
}
else if (!strncmp(name, "L2_T1", 5))
{
// L2 GLL
b_type = BasisType::GaussLobatto;
minncp = min_ncp_gll_x[cp_type][nb-2];
}
else if (!strncmp(name, "L2_T2", 5))
{
// L2 Positive
b_type = BasisType::Positive;
minncp = min_ncp_pos_x[cp_type][nb-2];
}
else
{
MFEM_ABORT("Only H1 GLL/Positive & L2 GL/GLL/Positive bases supported.");
}
ncp = std::max(minncp, ncp_i);
Setup(nb, ncp, b_type, cp_type, tol);
}
void PLBound::Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
{
real_t x,w;
intmin.SetSize(ncp);
intmax.SetSize(ncp);
intmin = 0.0;
intmax = 0.0;
Vector coeffm(nb);
coeffm = 0.0;
real_t a0 = 0.0;
real_t a1 = 0.0;
Vector nodal_vals, nodal_integ_vals;
if (b_type == 2) // compute values at equispaced nodes and GLL nodes
{
nodal_vals.SetSize(nb);
nodal_integ_vals.SetSize(nb);
Vector shape(nb);
for (int i = 0; i < nb; i++)
{
basisMatNodes.GetRow(i, shape);
nodal_vals(i) = shape*coeff;
basisMatInt.GetRow(i, shape);
nodal_integ_vals(i) = shape*coeff;
}
}
else
{
nodal_vals.SetDataAndSize(coeff.GetData(), nb);
nodal_integ_vals.SetDataAndSize(coeff.GetData(), nb);
}
// compute L2 projection for linear bases: a0 + a1*x
if (proj)
{
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes_int(i)-1;
w = 2.0*weights_int(i);
a0 += 0.5*nodal_integ_vals(i)*w;
a1 += 1.5*nodal_integ_vals(i)*w*x;
}
// offset the linear fit from nodal values
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes(i)-1;
coeffm(i) = nodal_vals(i) - a0 - a1*x;
}
// compute coefficients for Bernstein
if (b_type == 2)
{
LUFactors lu(basisMatLU.GetData(), lu_ip.GetData());
lu.Solve(nb, 1, coeffm.GetData());
}
// initialize the bounds to be the linear fit
for (int j = 0; j < ncp; j++)
{
x = 2.0*control_points(j)-1;
intmin(j) = a0 + a1*x;
intmax(j) = intmin(j);
}
}
else
{
coeffm.SetDataAndSize(coeff.GetData(), nb);
}
for (int i = 0; i < nb; i++)
{
real_t c = coeffm(i);
for (int j = 0; j < ncp; j++)
{
intmin(j) += min(lbound(i,j)*c, ubound(i,j)*c);
intmax(j) += max(lbound(i,j)*c, ubound(i,j)*c);
}
}
}
void PLBound::Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
{
intmin.SetSize(ncp*ncp);
intmax.SetSize(ncp*ncp);
intmin = 0.0;
intmax = 0.0;
Vector intminT(ncp*nb);
Vector intmaxT(ncp*nb);
// Get bounds for each row of the solution
for (int i = 0; i < nb; i++)
{
Vector solcoeff(coeff.GetData()+i*nb, nb);
Vector intminrow(intminT.GetData()+i*ncp, ncp);
Vector intmaxrow(intmaxT.GetData()+i*ncp, ncp);
Get1DBounds(solcoeff, intminrow, intmaxrow);
}
Vector intminT2 = intminT;
// Compute a0 and a1 for each column of nodes
Vector a0V(ncp), a1V(ncp);
a0V = 0.0;
a1V = 0.0;
real_t x,w,t;
if (proj)
{
if (b_type == 2)
{
// Note: DenseMatrix uses column-major ordering so we will need to
// transpose the matrix.
DenseMatrix intminTM(intminT.GetData(), ncp, nb),
intmaxTM(intmaxT.GetData(), ncp, nb),
intmeanTM(ncp, nb);
DenseMatrix minvalsM(nb, ncp), maxvalsM(nb, ncp), meanintvalsM(nb, ncp);
MultABt(basisMatNodes, intminTM, minvalsM);
MultABt(basisMatNodes, intmaxTM, maxvalsM);
intmeanTM = intminTM;
intmeanTM += intmaxTM;
intmeanTM *= 0.5;
MultABt(basisMatInt, intmeanTM, meanintvalsM);
// Compute the linear fit along each column and then offset it from
// the bounds on the coefficient.
// Note: Since Bernstein bases are positive, we can use the lower
// bounds to compute the lower bounding polynomial and subtract the
// linear fit before finding the Bernstein coefficients corresponding
// to the perturbation. Same for upper bounds. If the bases were not
// always positive, it is not yet clear if the perturbation
// coefficients will be this straightforward to compute.
for (int j = 0; j < ncp; j++) // row of interval points
{
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes_int(i)-1; // x-coordinate
w = 2.0*weights_int(i); // weight
t = meanintvalsM(i,j);
a0V(j) += 0.5*t*w;
a1V(j) += 1.5*t*w*x;
}
// Offset linear fit
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes(i)-1; // x-coordinate
minvalsM(i,j) -= a0V(j) + a1V(j)*x;
maxvalsM(i,j) -= a0V(j) + a1V(j)*x;
}
// Compute Bernstein coefficients
LUFactors lu(basisMatLU.GetData(), lu_ip.GetData());
lu.Solve(nb, 1, minvalsM.GetColumn(j));
lu.Solve(nb, 1, maxvalsM.GetColumn(j));
for (int i = 0; i < nb; i++)
{
intminT(i*ncp+j) = minvalsM(i,j);
intmaxT(i*ncp+j) = maxvalsM(i,j);
}
}
}
else
{
for (int j = 0; j < nb; j++) // row of nodes
{
x = 2.0*nodes(j)-1; // x-coordinate
w = 2.0*weights(j); // weight
for (int i = 0; i < ncp; i++) // column of interval points
{
t = 0.5*(intminT(j*ncp+i)+intmaxT(j*ncp+i));
a0V(i) += 0.5*t*w;
a1V(i) += 1.5*t*w*x;
}
}
// offset the linear fit from nodal values
for (int j = 0; j < nb; j++) // row of nodes
{
x = 2.0*nodes(j)-1; // x-coordinate
for (int i = 0; i < ncp; i++) // column of interval points
{
t = a0V(i) + a1V(i)*x;
intminT(j*ncp+i) -= t;
intmaxT(j*ncp+i) -= t;
}
}
}
// Initialize bounds using a0 and a1 values
for (int j = 0; j < ncp; j++) // row j
{
x = 2.0*control_points(j)-1;
for (int i = 0; i < ncp; i++) // column i
{
intmin(j*ncp+i) = a0V(i) + a1V(i)*x;
intmax(j*ncp+i) = intmin(j*ncp+i);
}
}
}
// Compute bounds
int id1 = 0, id2 = 0;
Vector vals(4);
for (int j = 0; j < nb; j++)
{
for (int i = 0; i < ncp; i++) // ith column
{
real_t w0 = intminT(id1++);
real_t w1 = intmaxT(id2++);
for (int k = 0; k < ncp; k++) // kth row
{
vals(0) = w0*lbound(j,k);
vals(1) = w0*ubound(j,k);
vals(2) = w1*lbound(j,k);
vals(3) = w1*ubound(j,k);
intmin(k*ncp+i) += vals.Min();
intmax(k*ncp+i) += vals.Max();
}
}
}
}
void PLBound::Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
{
int nb2 = nb*nb,
ncp2 = ncp*ncp,
ncp3 = ncp*ncp*ncp;
intmin.SetSize(ncp3);
intmax.SetSize(ncp3);
intmin = 0.0;
intmax = 0.0;
Vector intminT(ncp2*nb);
Vector intmaxT(ncp2*nb);
// Get bounds for each slice of the solution
for (int i = 0; i < nb; i++)
{
Vector solcoeff(coeff.GetData()+i*nb2, nb2);
Vector intminrow(intminT.GetData()+i*ncp2, ncp2);
Vector intmaxrow(intmaxT.GetData()+i*ncp2, ncp2);
Get2DBounds(solcoeff, intminrow, intmaxrow);
}
DenseMatrix intminTM(intminT.GetData(), ncp2, nb),
intmaxTM(intmaxT.GetData(), ncp2, nb);
// Compute a0 and a1 for each tower of nodes
Vector a0V(ncp2), a1V(ncp2);
a0V = 0.0;
a1V = 0.0;
real_t x,w,t;
if (proj)
{
if (b_type == 2) // Bernstein bases
{
// Compute the mean coefficients along each tower.
for (int j = 0; j < ncp2; j++) // slice of interval points
{
Vector meanBounds(nb), minBounds(nb), maxBounds(nb);
intminTM.GetRow(j, minBounds);
intmaxTM.GetRow(j, maxBounds);
for (int i = 0; i < nb; i++) // column of nodes
{
meanBounds(i) = 0.5*(minBounds(i)+maxBounds(i));
}
Vector meanNodalIntVals(nb);
Vector minNodalVals(nb);
Vector maxNodalVals(nb);
Vector row(nb);
for (int i = 0; i < nb; i++)
{
basisMatNodes.GetRow(i, row);
minNodalVals(i) = row*minBounds;
maxNodalVals(i) = row*maxBounds;
basisMatInt.GetRow(i, row);
meanNodalIntVals(i) = row*meanBounds;
}
// linear fit along each tower
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes_int(i)-1; // x-coordinate
w = 2.0*weights_int(i); // weight
a0V(j) += 0.5*meanNodalIntVals(i)*w;
a1V(j) += 1.5*meanNodalIntVals(i)*w*x;
}
// offset the linear fit from bounding coefficients
for (int i = 0; i < nb; i++)
{
x = 2.0*nodes(i)-1; // x-coordinate
minBounds(i) -= a0V(j) + a1V(j)*x;
maxBounds(i) -= a0V(j) + a1V(j)*x;
}
// Compute Bernstein coefficients
LUFactors lu(basisMatLU.GetData(), lu_ip.GetData());
lu.Solve(nb, 1, minBounds.GetData());
lu.Solve(nb, 1, maxBounds.GetData());
for (int i = 0; i < nb; i++)
{
intminT(i*ncp2+j) = minBounds(i);
intmaxT(i*ncp2+j) = maxBounds(i);
}
}
}
else
{
// nodal bases
for (int j = 0; j < nb; j++) // tower of nodes
{
x = 2.0*nodes(j)-1; // x-coordinate
w = 2.0*weights(j); // weight
for (int i = 0; i < ncp2; i++) // slice of interval points
{
t = 0.5*(intminT(j*ncp2+i)+intmaxT(j*ncp2+i));
a0V(i) += 0.5*t*w;
a1V(i) += 1.5*t*w*x;
}
}
// offset the linear fit from nodal values
for (int j = 0; j < nb; j++) // row of nodes
{
x = 2.0*nodes(j)-1; // x-coordinate
for (int i = 0; i < ncp2; i++) // column of interval points
{
t = a0V(i) + a1V(i)*x;
intminT(j*ncp2+i) -= t;
intmaxT(j*ncp2+i) -= t;
}
}
}
// Initialize bounds using a0 and a1 values
for (int j = 0; j < ncp; j++) // slice j
{
x = 2.0*control_points(j)-1;
for (int i = 0; i < ncp2; i++) // tower i
{
intmin(j*ncp2+i) = a0V(i) + a1V(i)*x;
intmax(j*ncp2+i) = a0V(i) + a1V(i)*x;
}
}
}
// Compute bounds
int id1 = 0, id2 = 0;
Vector vals(4);
for (int j = 0; j < nb; j++)
{
for (int i = 0; i < ncp2; i++) // ith tower
{
real_t w0 = intminT(id1++);
real_t w1 = intmaxT(id2++);
for (int k = 0; k < ncp; k++) // kth slice
{
vals(0) = w0*lbound(j,k);
vals(1) = w0*ubound(j,k);
vals(2) = w1*lbound(j,k);
vals(3) = w1*ubound(j,k);
intmin(k*ncp2+i) += vals.Min();
intmax(k*ncp2+i) += vals.Max();
}
}
}
}
void PLBound::GetNDBounds(int rdim, Vector &coeff,
Vector &intmin, Vector &intmax) const
{
if (rdim == 1)
{
Get1DBounds(coeff, intmin, intmax);
}
else if (rdim == 2)
{
Get2DBounds(coeff, intmin, intmax);
}
else if (rdim == 3)
{
Get3DBounds(coeff, intmin, intmax);
}
else
{
MFEM_ABORT("Currently not supported.");
}
}
void PLBound::SetupBernsteinBasisMat(DenseMatrix &basisMat,
Vector &nodesBern) const
{
const int nbern = nodesBern.Size();
L2_SegmentElement el(nbern-1, 2); // we use L2 to leverage lexicographic order
Array<int> ordering = el.GetLexicographicOrdering();
basisMat.SetSize(nbern, nbern);
Vector shape(nbern);
IntegrationPoint ip;
for (int i = 0; i < nbern; i++)
{
ip.x = nodesBern(i);
el.CalcShape(ip, shape);
basisMat.SetRow(i, shape);
}
}
constexpr int PLBound::min_ncp_gl_x[2][11];
constexpr int PLBound::min_ncp_gll_x[2][11];
constexpr int PLBound::min_ncp_pos_x[2][11];
int PLBound::GetMinimumPointsForGivenBases(int nb_i, int b_type_i,
int cp_type_i) const
{
MFEM_VERIFY(b_type_i >= 0 && b_type_i <= 2, "Invalid node type. Specify 0 "
"for GL, 1 for GLL, and 2 for positive " "bases.");
MFEM_VERIFY(cp_type_i == 0 || cp_type_i == 1, "Invalid control point type. "
"Specify 0 for GL+end points, 1 for Chebyshev.");
if (nb_i > 12)
{
MFEM_ABORT("GetMinimumPointsForGivenBases can only be used for maximum "
"order = 11, i.e. nb=12. 2*nb points should be sufficient to "
"bound the bases up to nb = 30.");
}
else if (b_type_i == 0)
{
return min_ncp_gl_x[cp_type_i][nb_i-2];
}
else if (b_type_i == 1)
{
return min_ncp_gll_x[cp_type_i][nb_i-2];
}
else if (b_type_i == 2)
{
return min_ncp_pos_x[cp_type_i][nb_i-2];
}
return 0;
}
void PLBound::Print(std::ostream &outp) const
{
outp << "PLBound nb: " << nb << std::endl;
outp << "PLBound ncp: " << ncp << std::endl;
outp << "PLBound b_type: " << b_type << std::endl;
outp << "PLBound cp_type: " << cp_type << std::endl;
outp << "Print nodes: " << std::endl;
nodes.Print(outp);
outp << "Print weights: " << std::endl;
weights.Print(outp);
outp << "Print control_points: " << std::endl;
control_points.Print(outp);
outp << "Print lower bounds: " << std::endl;
lbound.Print(outp);
outp << "Print upper bounds: " << std::endl;
ubound.Print(outp);
}
}
+136
View File
@@ -0,0 +1,136 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_BOUND
#define MFEM_BOUND
#include "../config/config.hpp"
#include "fespace.hpp"
namespace mfem
{
/** @name Piecewise linear bounds of bases
\brief Piecewise linear bounds of bases can be used to compute bounds on the grid function in each element. The bounds for the bases are constructed based on the following parameters:
(i) @b nb: number of bases/nodes in 1D (i.e. polynomial order+1),
(ii) @b b_type: bases type, 0 - Lagrange interpolants on Gauss-Legendre nodes, 1 - Lagrange interpolants on Gauss-Lobatto-Legendre nodes, and
2 - Positive/Bernstein bases on uniformly distributed nodes,
(iii) @b ncp: number of control points used to construct the piecewise linear bounds
(iv) @b cp_type: control point distribution. 0 - GL + end-points,
1 - Chebyshev.
Note: @b nb and @b b_type are inferred directly from the grid-function.
If the user does not specify @b ncp and @b cp_type, the minimum value of
@b ncp is used that would bound the bases for the @b cp_type. We default
to @b cp_type = 0 as it requires fewer number of points to bound the bases. Typically, @b ncp = 2 @b nb is sufficient to get fairly compact bounds, and increasing @b ncp results in tighter bounds.
Finally, only tensor-product elements are currently supported.
For more technical details see:
Mittal et al., "General Field Evaluation in High-Order Meshes on GPUs" &
Dzanic et al., "A method for bounding high-order finite element
functions: Applications to mesh validity and bounds-preserving limiters".
*/
class PLBound
{
private:
int nb; // #mesh nodes in 1D
int ncp; // #control points in 1D
int b_type; // bases type: 0 - GL, 1 - GLL, 2 - Bernstein
int cp_type; // control points type: 0 - GL+Ends, 1 - Chebyshev
bool proj = true; // Use linear projection to compute bounds.
real_t tol = 0.0; // offset bounds to avoid round-off errors
Vector nodes, weights, control_points;
DenseMatrix lbound, ubound; // nb x ncp matrices with bounds of all bases
// Some auxillary storage for computing the bounds with Bernstein
DenseMatrix basisMatNodes; // Bernstein bases at equispaced nodes
DenseMatrix basisMatInt; // Bernstein bases at GLL nodes
Vector nodes_int, weights_int; // Integration nodes and weights
DenseMatrix basisMatLU; // Used to compute LU factors for Bernstein
mutable Array<int> lu_ip;
// stores min_ncp for nb = 2..12 for Lagrange interpolants on GL nodes
// with GL+end points and Chebyshev points as control points
static constexpr int min_ncp_gl_x[2][11]= {{3,5,6,8,9,10,11,11,12,13,14},
{3,5,8,9,11,12,14,15,17,18,20}
};
// stores min_ncp for nb = 2..12 for Lagrange interpolants on GLL nodes
// with GL+end points and Chebyshev points as control points
static constexpr int min_ncp_gll_x[2][11]= {{3,5,7,8,9,10,12,13,14,15,16},
{3,5,8,10,12,13,15,17,19,21,22}
};
// stores min_ncp for nb = 2..12 for Bernstein bases with GL+end points
// and Chebyshev points as control points
static constexpr int min_ncp_pos_x[2][11]= {{3,5,7,8,8,9,10,10,11,12,13},
{3,5,8,9,11,12,13,13,14,15,16}
};
public:
// Constructor
PLBound(const int nb_i, const int ncp_i, const int b_type_i,
const int cp_type_i, const real_t tol_i)
{
Setup(nb_i, ncp_i, b_type_i, cp_type_i, tol_i);
}
// Constructor
PLBound(FiniteElementSpace *fes, int ncp_i = -1, int cp_type_i = 0);
// Get minimum number of control points needed to bound the given bases
int GetMinimumPointsForGivenBases(int nb_i, int b_type_i,
int cp_type_i) const;
// Print information about the bounds
void Print(std::ostream &outp = mfem::out) const;
// Enable (default) or disable linear projection before bounding.
// This projection increases the computational cost but results in tighter
// bounds.
void SetProjectionFlagForBounding(bool proj_) { proj = proj_; }
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D/2D/3D.
void GetNDBounds(int rdim, Vector &coeff,
Vector &intmin, Vector &intmax) const;
/// Get number of control points used to compute the bounds.
int GetNControlPoints() const { return ncp; }
private:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D.
void Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 2D.
void Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 3D.
void Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Setup matrix used to compute values at given 1D locations in [0,1]
/// for Bernstein bases.
void SetupBernsteinBasisMat(DenseMatrix &basisMat, Vector &nodesBern) const;
void Setup(const int nb_i, const int ncp_i, const int b_type_i,
const int cp_type_i, const real_t tol_i);
};
} // namespace mfem
#endif // MFEM_BOUND
+217
View File
@@ -0,0 +1,217 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "complex_fem.hpp"
#include "../general/forall.hpp"
using namespace std;
namespace mfem
{
real_t
RealPartCoefficient::Eval(ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_t val = complex_coef_.Eval(T, ip);
return val.real();
}
real_t
ImagPartCoefficient::Eval(ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_t val = complex_coef_.Eval(T, ip);
return val.imag();
}
RealPartVectorCoefficient::RealPartVectorCoefficient(ComplexVectorCoefficient &
complex_vcoef)
: VectorCoefficient(complex_vcoef.GetVDim()),
complex_vcoef_(complex_vcoef),
val_(vdim)
{}
void
RealPartVectorCoefficient::Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_vcoef_.Eval(val_, T, ip);
V = val_.real();
}
ImagPartVectorCoefficient::ImagPartVectorCoefficient(ComplexVectorCoefficient &
complex_vcoef)
: VectorCoefficient(complex_vcoef.GetVDim()),
complex_vcoef_(complex_vcoef),
val_(vdim)
{}
void
ImagPartVectorCoefficient::Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_vcoef_.Eval(val_, T, ip);
V = val_.imag();
}
RealPartMatrixCoefficient::RealPartMatrixCoefficient(ComplexMatrixCoefficient &
complex_mcoef)
: MatrixCoefficient(complex_mcoef.GetHeight(), complex_mcoef.GetWidth()),
complex_mcoef_(complex_mcoef),
val_(height, width)
{}
void
RealPartMatrixCoefficient::Eval(DenseMatrix &M, ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_mcoef_.Eval(val_, T, ip);
M = val_.real();
}
ImagPartMatrixCoefficient::ImagPartMatrixCoefficient(ComplexMatrixCoefficient &
complex_mcoef)
: MatrixCoefficient(complex_mcoef.GetHeight(), complex_mcoef.GetWidth()),
complex_mcoef_(complex_mcoef),
val_(height, width)
{}
void
ImagPartMatrixCoefficient::Eval(DenseMatrix &M, ElementTransformation &T,
const IntegrationPoint &ip)
{
complex_mcoef_.Eval(val_, T, ip);
M = val_.imag();
}
ComplexCoefficient::ComplexCoefficient()
: time(0.),
re_part_coef_(*this), im_part_coef_(*this),
real_coef_(re_part_coef_), imag_coef_(im_part_coef_)
{ }
ComplexCoefficient::ComplexCoefficient(Coefficient &c_r,
Coefficient &c_i)
: time(c_r.GetTime()),
re_part_coef_(*this), im_part_coef_(*this),
real_coef_(c_r), imag_coef_(c_i)
{
c_i.SetTime(time);
}
complex_t
ComplexCoefficient::Eval(ElementTransformation &T,
const IntegrationPoint &ip)
{
// Avoid circular dependency
MFEM_VERIFY(std::addressof(real_coef_) != std::addressof(re_part_coef_) &&
std::addressof(imag_coef_) != std::addressof(im_part_coef_),
"Classes dervied from ComplexCoefficient must either "
"implement an Eval method or supply Coefficients "
"for both the real and imaginary parts of the field.");
return complex_t(real_coef_.Eval(T, ip), imag_coef_.Eval(T, ip));
}
ComplexVectorCoefficient::ComplexVectorCoefficient(VectorCoefficient &v_r,
VectorCoefficient &v_i)
: vdim(v_r.GetVDim()), time(v_r.GetTime()),
re_part_vcoef_(*this), im_part_vcoef_(*this),
real_vcoef_(v_r), imag_vcoef_(v_i)
{
MFEM_ASSERT(v_r.GetVDim() == v_i.GetVDim(), "ComplexVectorCoefficient"
" - incompatible vector dimensions of real and imaginary parts.");
v_i.SetTime(time);
}
void ComplexVectorCoefficient::Eval(ComplexVector &V, ElementTransformation &T,
const IntegrationPoint &ip)
{
// Avoid circular dependency
MFEM_VERIFY(std::addressof(real_vcoef_) != std::addressof(re_part_vcoef_) &&
std::addressof(imag_vcoef_) != std::addressof(im_part_vcoef_),
"Classes dervied from ComplexVectorCoefficient must either "
"implement an Eval method or supply VectorCoefficients "
"for both the real and imaginary parts of the field.");
V_r_.SetSize(vdim);
V_i_.SetSize(vdim);
real_vcoef_.Eval(V_r_, T, ip);
imag_vcoef_.Eval(V_i_, T, ip);
V.Set(V_r_, V_i_);
}
ComplexConstantCoefficient::ComplexConstantCoefficient(
const complex_t z)
: val(z), real_coef(z.real()), imag_coef(z.imag())
{
real_coef_ = real_coef;
imag_coef_ = imag_coef;
}
ComplexConstantCoefficient::ComplexConstantCoefficient(
real_t z_r, real_t z_i)
: real_coef(z_r), imag_coef(z_i)
{
val = complex_t(z_r, z_i);
real_coef_ = real_coef;
imag_coef_ = imag_coef;
}
complex_t ComplexFunctionCoefficient::Eval(ElementTransformation & T,
const IntegrationPoint & ip)
{
real_t x[3];
Vector transip(x, 3);
T.Transform(ip, transip);
if (Function)
{
return Function(transip);
}
else
{
return TDFunction(transip, GetTime());
}
}
void ComplexVectorFunctionCoefficient::Eval(ComplexVector &V,
ElementTransformation &T,
const IntegrationPoint &ip)
{
real_t x[3];
Vector transip(x, 3);
T.Transform(ip, transip);
V.SetSize(vdim);
if (Function)
{
Function(transip, V);
}
else
{
TDFunction(transip, GetTime(), V);
}
if (Q)
{
V *= Q->Eval(T, ip, GetTime());
}
}
} // end namespace mfem
+523
View File
@@ -0,0 +1,523 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_COMPLEX_COEFFICIENT
#define MFEM_COMPLEX_COEFFICIENT
#include "../config/config.hpp"
#include "../linalg/linalg.hpp"
#include "coefficient.hpp"
#include "intrules.hpp"
#include "eltrans.hpp"
namespace mfem
{
class ComplexCoefficient;
class ComplexVectorCoefficient;
class ComplexMatrixCoefficient;
/// Standard Coefficient which returns the real part of a ComplexCoefficient
class RealPartCoefficient : public Coefficient
{
private:
ComplexCoefficient &complex_coef_;
public:
RealPartCoefficient(ComplexCoefficient & complex_coef)
: complex_coef_(complex_coef) {}
real_t Eval(ElementTransformation &T,
const IntegrationPoint &ip);
};
/// Standard Coefficient which returns the imaginary part of a
/// ComplexCoefficient
class ImagPartCoefficient : public Coefficient
{
private:
ComplexCoefficient &complex_coef_;
public:
ImagPartCoefficient(ComplexCoefficient & complex_coef)
: complex_coef_(complex_coef) {}
real_t Eval(ElementTransformation &T,
const IntegrationPoint &ip);
};
typedef ImagPartCoefficient ImaginaryPartCoefficient;
class RealPartVectorCoefficient : public VectorCoefficient
{
private:
ComplexVectorCoefficient &complex_vcoef_;
mutable ComplexVector val_;
public:
RealPartVectorCoefficient(ComplexVectorCoefficient & complex_vcoef);
void Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip);
};
class ImagPartVectorCoefficient : public VectorCoefficient
{
private:
ComplexVectorCoefficient &complex_vcoef_;
mutable ComplexVector val_;
public:
ImagPartVectorCoefficient(ComplexVectorCoefficient & complex_vcoef);
void Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip);
};
typedef ImagPartVectorCoefficient ImaginaryPartVectorCoefficient;
class RealPartMatrixCoefficient : public MatrixCoefficient
{
private:
ComplexMatrixCoefficient &complex_mcoef_;
mutable ComplexTypeDenseMatrix val_;
public:
RealPartMatrixCoefficient(ComplexMatrixCoefficient & complex_mcoef);
void Eval(DenseMatrix &M, ElementTransformation &T,
const IntegrationPoint &ip);
};
class ImagPartMatrixCoefficient : public MatrixCoefficient
{
private:
ComplexMatrixCoefficient &complex_mcoef_;
mutable ComplexTypeDenseMatrix val_;
public:
ImagPartMatrixCoefficient(ComplexMatrixCoefficient & complex_mcoef);
void Eval(DenseMatrix &V, ElementTransformation &T,
const IntegrationPoint &ip);
};
typedef ImagPartMatrixCoefficient ImaginaryPartMatrixCoefficient;
/** @brief Base class ComplexCoefficients that optionally depend on space and
time. These are used by the SesquilinearForm, ComplexLinearForm, and
ComplexGridFunction classes to represent the physical coefficients in
the PDEs that are being discretized. This class can also be used in a more
general way to represent functions that don't necessarily belong to a FE
space, e.g., to project onto ComplexGridFunctions to use as initial
conditions, exact solutions, etc. See, e.g., ex22 for these uses. */
class ComplexCoefficient
{
protected:
real_t time;
private:
RealPartCoefficient re_part_coef_;
ImagPartCoefficient im_part_coef_;
protected:
Coefficient &real_coef_;
Coefficient &imag_coef_;
public:
ComplexCoefficient();
ComplexCoefficient(Coefficient &c_r, Coefficient &c_i);
/// Set the time for time dependent coefficients
virtual void SetTime(real_t t)
{ time = t; real_coef_.SetTime(t); imag_coef_.SetTime(t); }
/// Get the time for time dependent coefficients
real_t GetTime() { return time; }
/** @brief Evaluate the coefficient in the element described by @a T at the
point @a ip. */
/** @note When this method is called, the caller must make sure that the
IntegrationPoint associated with @a T is the same as @a ip. This can be
achieved by calling T.SetIntPoint(&ip). */
virtual complex_t Eval(ElementTransformation &T,
const IntegrationPoint &ip);
/** @brief Evaluate the coefficient in the element described by @a T at the
point @a ip at time @a t. */
/** @note When this method is called, the caller must make sure that the
IntegrationPoint associated with @a T is the same as @a ip. This can be
achieved by calling T.SetIntPoint(&ip). */
complex_t Eval(ElementTransformation &T,
const IntegrationPoint &ip, real_t t)
{
SetTime(t);
return Eval(T, ip);
}
/** @brief Access a standard Coefficient object reproducing the real part of
the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its real part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual Coefficient & real() { return real_coef_; }
/** @brief Access a standard Coefficient object reproducing the imaginary
part of the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its imaginary part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual Coefficient & imag() { return imag_coef_; }
virtual ~ComplexCoefficient() { }
};
/** @brief Base class ComplexVectorCoefficients that optionally depend
on space and time. These are used by the SesquilinearForm,
ComplexLinearForm, and ComplexGridFunction classes to represent
the physical vector-valued coefficients in the PDEs that are being
discretized. This class can also be used in a more general way to
represent functions that don't necessarily belong to a FE space,
e.g., to project onto ComplexGridFunctions to use as initial
conditions, exact solutions, etc. See, e.g., ex22 for these
uses. */
class ComplexVectorCoefficient
{
protected:
int vdim;
real_t time;
private:
RealPartVectorCoefficient re_part_vcoef_;
ImagPartVectorCoefficient im_part_vcoef_;
protected:
VectorCoefficient &real_vcoef_;
VectorCoefficient &imag_vcoef_;
mutable Vector V_r_;
mutable Vector V_i_;
public:
ComplexVectorCoefficient(int vd)
: vdim(vd), time(0.),
re_part_vcoef_(*this), im_part_vcoef_(*this),
real_vcoef_(re_part_vcoef_), imag_vcoef_(im_part_vcoef_)
{ }
ComplexVectorCoefficient(VectorCoefficient &v_r, VectorCoefficient &v_i);
/// Set the time for time dependent coefficients
virtual void SetTime(real_t t)
{ time = t; real_vcoef_.SetTime(t); imag_vcoef_.SetTime(t); }
/// Get the time for time dependent coefficients
real_t GetTime() { return time; }
/// Returns dimension of the vector.
int GetVDim() { return vdim; }
/** @brief Evaluate the vector coefficient in the element described by @a T
at the point @a ip, storing the result in @a V. */
/** @note When this method is called, the caller must make sure that the
IntegrationPoint associated with @a T is the same as @a ip. This can be
achieved by calling T.SetIntPoint(&ip). */
virtual void Eval(ComplexVector &V, ElementTransformation &T,
const IntegrationPoint &ip);
/** @brief Evaluate the vector coefficient in the element described by @a T
at the point @a ip at time @a t, storing the result in @a V. */
/** @note When this method is called, the caller must make sure that the
IntegrationPoint associated with @a T is the same as @a ip. This can be
achieved by calling T.SetIntPoint(&ip). */
void Eval(ComplexVector &V, ElementTransformation &T,
const IntegrationPoint &ip, real_t t)
{
SetTime(t);
Eval(V, T, ip);
}
/** @brief Access a standard Coefficient object reproducing the real part of
the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its real part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual VectorCoefficient & real() { return real_vcoef_; }
/** @brief Access a standard Coefficient object reproducing the imaginary
part of the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its imaginary part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual VectorCoefficient & imag() { return imag_vcoef_; }
virtual ~ComplexVectorCoefficient() { }
};
/** @brief Base class ComplexMatrixCoefficients that optionally depend
on space and time. These are used by the SesquilinearForm,
ComplexLinearForm, and ComplexGridFunction classes to represent
the physical matrix-valued coefficients in the PDEs that are being
discretized. This class can also be used in a more general way to
represent functions that don't necessarily belong to a FE space.
See, e.g., ex22 for these uses. */
class ComplexMatrixCoefficient
{
protected:
int height, width;
real_t time;
private:
RealPartMatrixCoefficient re_part_mcoef_;
ImagPartMatrixCoefficient im_part_mcoef_;
protected:
MatrixCoefficient &real_mcoef_;
MatrixCoefficient &imag_mcoef_;
mutable DenseMatrix M_r_;
mutable DenseMatrix M_i_;
public:
/// Construct a dim x dim matrix coefficient.
explicit ComplexMatrixCoefficient(int dim)
: height(dim), width(dim), time(0.),
re_part_mcoef_(*this), im_part_mcoef_(*this),
real_mcoef_(re_part_mcoef_), imag_mcoef_(im_part_mcoef_)
{ }
/// Construct a h x w matrix coefficient.
ComplexMatrixCoefficient(int h, int w) :
height(h), width(w), time(0.),
re_part_mcoef_(*this), im_part_mcoef_(*this),
real_mcoef_(re_part_mcoef_), imag_mcoef_(im_part_mcoef_)
{ }
/// Set the time for time dependent coefficients
virtual void SetTime(real_t t) { time = t; }
/// Get the time for time dependent coefficients
real_t GetTime() { return time; }
/// Get the height of the matrix.
int GetHeight() const { return height; }
/// Get the width of the matrix.
int GetWidth() const { return width; }
/// For backward compatibility get the width of the matrix.
int GetVDim() const { return width; }
/** @brief Evaluate the matrix coefficient in the element described by @a T
at the point @a ip, storing the result in @a K. */
/** @note When this method is called, the caller must make sure that the
IntegrationPoint associated with @a T is the same as @a ip. This can be
achieved by calling T.SetIntPoint(&ip). */
virtual void Eval(ComplexTypeDenseMatrix &K, ElementTransformation &T,
const IntegrationPoint &ip) = 0;
/** @brief Access a standard Coefficient object reproducing the real part of
the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its real part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual MatrixCoefficient & real() { return real_mcoef_; }
/** @brief Access a standard Coefficient object reproducing the imaginary
part of the complex-valued field */
/** @note By default this method returns an internal object which
computes the complex value using the above Eval method and
returns its imaginary part. Custom implementations may choose to
override this method with a more efficient real-valued
coefficient. */
virtual MatrixCoefficient & imag() { return imag_mcoef_; }
virtual ~ComplexMatrixCoefficient() { }
};
/// A complex-valued coefficient that is constant across space and time
class ComplexConstantCoefficient : public ComplexCoefficient
{
private:
complex_t val;
ConstantCoefficient real_coef;
ConstantCoefficient imag_coef;
public:
ComplexConstantCoefficient(const complex_t z);
ComplexConstantCoefficient(real_t z_r, real_t z_i = 0.);
complex_t Eval(ElementTransformation &T,
const IntegrationPoint &ip) { return val; }
};
/// Complex-valued vector coefficient that is constant in space and time.
class ComplexVectorConstantCoefficient : public ComplexVectorCoefficient
{
private:
ComplexVector vec;
public:
/// Construct the coefficient with constant vector @a v.
ComplexVectorConstantCoefficient(const ComplexVector &v)
: ComplexVectorCoefficient(v.Size()), vec(v) { }
/// Construct the coefficient with constant vector @a v.
ComplexVectorConstantCoefficient(const Vector &v)
: ComplexVectorCoefficient(v.Size()), vec(v) { }
using ComplexVectorCoefficient::Eval;
/// Evaluate the vector coefficient at @a ip.
void Eval(ComplexVector &V, ElementTransformation &T,
const IntegrationPoint &ip) override { V = vec; }
/// Return a reference to the constant vector in this class.
const ComplexVector& GetVec() const { return vec; }
};
/// Complex-valued vector coefficient that is constant in space and time.
class ComplexMatrixConstantCoefficient : public ComplexMatrixCoefficient
{
private:
ComplexTypeDenseMatrix mat;
public:
/// Construct the coefficient with constant vector @a v.
ComplexMatrixConstantCoefficient(const ComplexTypeDenseMatrix &m)
: ComplexMatrixCoefficient(m.Height(), m.Width()), mat(m) { }
/// Construct the coefficient with constant vector @a v.
ComplexMatrixConstantCoefficient(const DenseMatrix &m)
: ComplexMatrixCoefficient(m.Height(), m.Width()), mat(m) { }
using ComplexMatrixCoefficient::Eval;
/// Evaluate the matrix coefficient at @a ip.
void Eval(ComplexTypeDenseMatrix &M, ElementTransformation &T,
const IntegrationPoint &ip) override { M = mat; }
/// Return a reference to the constant matrix in this class.
const ComplexTypeDenseMatrix& GetMat() const { return mat; }
};
/// A general complex-valued function coefficient
class ComplexFunctionCoefficient : public ComplexCoefficient
{
protected:
std::function<complex_t(const Vector &)> Function;
std::function<complex_t(const Vector &, real_t)> TDFunction;
public:
/// Define a time-independent coefficient from a std function
/** \param F time-independent std::function */
ComplexFunctionCoefficient(std::function<complex_t
(const Vector &)> F)
: Function(std::move(F))
{ }
/// Define a time-dependent coefficient from a std function
/** \param TDF time-dependent function */
ComplexFunctionCoefficient(std::function<complex_t
(const Vector &, real_t)> TDF)
: TDFunction(std::move(TDF))
{ }
/// (DEPRECATED) Define a time-independent coefficient from a C-function
/** @deprecated Use the method where the C-function, @a f, uses a const
Vector argument instead of Vector. */
MFEM_DEPRECATED ComplexFunctionCoefficient(complex_t
(*f)(Vector &))
{
// Cast first to (void*) to suppress a warning from newer version of
// Clang when using -Wextra.
Function = reinterpret_cast<complex_t(*)
(const Vector&)>((void*)f);
TDFunction = NULL;
}
/// (DEPRECATED) Define a time-dependent coefficient from a C-function
/** @deprecated Use the method where the C-function, @a tdf, uses a const
Vector argument instead of Vector. */
MFEM_DEPRECATED ComplexFunctionCoefficient(complex_t
(*tdf)(Vector &, real_t))
{
Function = NULL;
// Cast first to (void*) to suppress a warning from newer version of
// Clang when using -Wextra.
TDFunction =
reinterpret_cast<complex_t(*)(const Vector&,
real_t)>((void*)tdf);
}
/// Evaluate the coefficient at @a ip.
complex_t Eval(ElementTransformation &T,
const IntegrationPoint &ip) override;
};
/// A general vector function coefficient
class ComplexVectorFunctionCoefficient : public ComplexVectorCoefficient
{
private:
std::function<void(const Vector &, ComplexVector &)> Function;
std::function<void(const Vector &, real_t, ComplexVector &)> TDFunction;
ComplexCoefficient *Q;
public:
/// Define a time-independent complex-valued vector coefficient
/// from a std function
/** \param dim - the size of the vector
\param F - time-independent function
\param q - optional scalar Coefficient to scale the vector coefficient */
ComplexVectorFunctionCoefficient(int dim,
std::function<void(const Vector &,
ComplexVector &)> F,
ComplexCoefficient *q = nullptr)
: ComplexVectorCoefficient(dim), Function(std::move(F)), Q(q)
{ }
/// Define a time-dependent complex-valued vector coefficient from
/// a std function
/** \param dim - the size of the vector
\param TDF - time-dependent function
\param q - optional scalar ComplexCoefficient to scale the vector coefficient */
ComplexVectorFunctionCoefficient(int dim,
std::function<void(const Vector &, real_t,
ComplexVector &)> TDF,
ComplexCoefficient *q = nullptr)
: ComplexVectorCoefficient(dim), TDFunction(std::move(TDF)), Q(q)
{ }
using ComplexVectorCoefficient::Eval;
/// Evaluate the vector coefficient at @a ip.
void Eval(ComplexVector &V, ElementTransformation &T,
const IntegrationPoint &ip) override;
virtual ~ComplexVectorFunctionCoefficient() { }
};
} // end namespace mfem
#endif
+240
View File
@@ -96,6 +96,23 @@ ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff)
{
gfr->SyncMemory(*this);
gfi->SyncMemory(*this);
gfr->ProjectCoefficient(real_coeff);
*gfi = 0.0;
gfr->SyncAliasMemory(*this);
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectCoefficient(ComplexCoefficient &coeff)
{
this->ProjectCoefficient(coeff.real(), coeff.imag());
}
void
ComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff)
@@ -108,6 +125,23 @@ ComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff)
{
gfr->SyncMemory(*this);
gfi->SyncMemory(*this);
gfr->ProjectCoefficient(real_vcoeff);
*gfi = 0.0;
gfr->SyncAliasMemory(*this);
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectCoefficient(ComplexVectorCoefficient &vcoeff)
{
this->ProjectCoefficient(vcoeff.real(), vcoeff.imag());
}
void
ComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
@@ -121,6 +155,26 @@ ComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Array<int> &attr)
{
ConstantCoefficient zero_coeff(0.0);
gfr->SyncMemory(*this);
gfi->SyncMemory(*this);
gfr->ProjectBdrCoefficient(real_coeff, attr);
gfi->ProjectBdrCoefficient(zero_coeff, attr);
gfr->SyncAliasMemory(*this);
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficient(ComplexCoefficient &coeff,
Array<int> &attr)
{
this->ProjectBdrCoefficient(coeff.real(), coeff.imag(), attr);
}
void
ComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff,
@@ -134,6 +188,28 @@ ComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient &real_vcoeff,
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient &real_vcoeff,
Array<int> &attr)
{
Vector zero_vec(real_vcoeff.GetVDim()); zero_vec = 0.;
VectorConstantCoefficient zero_vcoeff(zero_vec);
gfr->SyncMemory(*this);
gfi->SyncMemory(*this);
gfr->ProjectBdrCoefficientNormal(real_vcoeff, attr);
gfi->ProjectBdrCoefficientNormal(zero_vcoeff, attr);
gfr->SyncAliasMemory(*this);
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficientNormal(
ComplexVectorCoefficient &vcoeff,
Array<int> &attr)
{
this->ProjectBdrCoefficientNormal(vcoeff.real(), vcoeff.imag(), attr);
}
void
ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
@@ -149,6 +225,80 @@ ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
Array<int> &attr)
{
Vector zero_vec(real_vcoeff.GetVDim()); zero_vec = 0.;
VectorConstantCoefficient zero_vcoeff(zero_vec);
gfr->SyncMemory(*this);
gfi->SyncMemory(*this);
gfr->ProjectBdrCoefficientTangent(real_vcoeff, attr);
gfi->ProjectBdrCoefficientTangent(zero_vcoeff, attr);
gfr->SyncAliasMemory(*this);
gfi->SyncAliasMemory(*this);
}
void
ComplexGridFunction::ProjectBdrCoefficientTangent(
ComplexVectorCoefficient &vcoeff,
Array<int> &attr)
{
this->ProjectBdrCoefficientTangent(vcoeff.real(), vcoeff.imag(), attr);
}
real_t
ComplexGridFunction::ComputeL2Error(Coefficient &re_exsol,
Coefficient &im_exsol,
const IntegrationRule *irs[],
const Array<int> *elems) const
{
real_t err_r = gfr->ComputeL2Error(re_exsol, irs, elems);
real_t err_i = gfi->ComputeL2Error(im_exsol, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
real_t
ComplexGridFunction::ComputeL2Error(Coefficient &re_exsol,
const IntegrationRule *irs[],
const Array<int> *elems) const
{
ConstantCoefficient zero_coef(0.0);
real_t err_r = gfr->ComputeL2Error(re_exsol, irs, elems);
real_t err_i = gfi->ComputeL2Error(zero_coef, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
real_t
ComplexGridFunction::ComputeL2Error(VectorCoefficient &re_exsol,
VectorCoefficient &im_exsol,
const IntegrationRule *irs[],
const Array<int> *elems) const
{
real_t err_r = gfr->ComputeL2Error(re_exsol, irs, elems);
real_t err_i = gfi->ComputeL2Error(im_exsol, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
real_t
ComplexGridFunction::ComputeL2Error(VectorCoefficient &re_exsol,
const IntegrationRule *irs[],
const Array<int> *elems) const
{
Vector zero_vec(re_exsol.GetVDim()); zero_vec = 0.0;
VectorConstantCoefficient zero_coef(zero_vec);
real_t err_r = gfr->ComputeL2Error(re_exsol, irs, elems);
real_t err_i = gfi->ComputeL2Error(zero_coef, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
ComplexLinearForm::ComplexLinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention convention)
@@ -731,6 +881,17 @@ ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff)
{
pgfr->SyncMemory(*this);
pgfi->SyncMemory(*this);
pgfr->ProjectCoefficient(real_coeff);
*pgfi = 0.0;
pgfr->SyncAliasMemory(*this);
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
VectorCoefficient &imag_vcoeff)
@@ -743,6 +904,17 @@ ParComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff,
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectCoefficient(VectorCoefficient &real_vcoeff)
{
pgfr->SyncMemory(*this);
pgfi->SyncMemory(*this);
pgfr->ProjectCoefficient(real_vcoeff);
*pgfi = 0.0;
pgfr->SyncAliasMemory(*this);
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff,
@@ -756,6 +928,19 @@ ParComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficient(Coefficient &real_coeff,
Array<int> &attr)
{
ConstantCoefficient zero_coeff(0.0);
pgfr->SyncMemory(*this);
pgfi->SyncMemory(*this);
pgfr->ProjectBdrCoefficient(real_coeff, attr);
pgfi->ProjectBdrCoefficient(zero_coeff, attr);
pgfr->SyncAliasMemory(*this);
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient
&real_vcoeff,
@@ -771,6 +956,21 @@ ParComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficientNormal(VectorCoefficient
&real_vcoeff,
Array<int> &attr)
{
Vector zero_vec(real_vcoeff.GetVDim()); zero_vec = 0.;
VectorConstantCoefficient zero_vcoeff(zero_vec);
pgfr->SyncMemory(*this);
pgfi->SyncMemory(*this);
pgfr->ProjectBdrCoefficientNormal(real_vcoeff, attr);
pgfi->ProjectBdrCoefficientNormal(zero_vcoeff, attr);
pgfr->SyncAliasMemory(*this);
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
@@ -786,6 +986,21 @@ ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
&real_vcoeff,
Array<int> &attr)
{
Vector zero_vec(real_vcoeff.GetVDim()); zero_vec = 0.;
VectorConstantCoefficient zero_vcoeff(zero_vec);
pgfr->SyncMemory(*this);
pgfi->SyncMemory(*this);
pgfr->ProjectBdrCoefficientTangent(real_vcoeff, attr);
pgfi->ProjectBdrCoefficientTangent(zero_vcoeff, attr);
pgfr->SyncAliasMemory(*this);
pgfi->SyncAliasMemory(*this);
}
void
ParComplexGridFunction::Distribute(const Vector *tv)
{
@@ -825,6 +1040,31 @@ ParComplexGridFunction::ParallelProject(Vector &tv) const
tvi.SyncAliasMemory(tv);
}
real_t
ParComplexGridFunction::ComputeL2Error(Coefficient &exsolr,
const IntegrationRule *irs[],
Array<int> *elems) const
{
ConstantCoefficient zeroCoef(0.0);
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(zeroCoef, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
real_t
ParComplexGridFunction::ComputeL2Error(VectorCoefficient &exsolr,
const IntegrationRule *irs[],
Array<int> *elems) const
{
Vector zeroVec(exsolr.GetVDim()); zeroVec = 0.0;
VectorConstantCoefficient zeroCoef(zeroVec);
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(zeroCoef, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ComplexOperator::Convention
+1307 -21
View File
File diff suppressed because it is too large Load Diff
+12 -10
View File
@@ -764,9 +764,9 @@ ParaViewDataCollectionBase::ParaViewDataCollectionBase(
{
cycle = 0;
#ifdef MFEM_USE_ZLIB
compression = true; // if we have zlib, enable compression
#else
compression = false; // otherwise, disable compression
// If we have zlib, enable compression. Otherwise, compression is disabled in
// the DataCollection base class constructor.
compression = true;
#endif
}
@@ -784,13 +784,8 @@ void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
{
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
"Compression level must be between -1 and 9 (inclusive).");
if (compression_level_ != 0) { SetCompression(true);}
compression_level = compression_level_;
compression = compression_level_ != 0;
}
void ParaViewDataCollectionBase::SetCompression(bool compression_)
{
compression = compression_;
}
int ParaViewDataCollectionBase::GetCompressionLevel() const
@@ -1174,7 +1169,14 @@ const char *ParaViewDataCollection::GetDataTypeString() const
ParaViewHDFDataCollection::ParaViewHDFDataCollection(
const std::string &collection_name, Mesh *mesh)
: ParaViewDataCollectionBase(collection_name, mesh)
{ }
{
compression = true;
}
void ParaViewHDFDataCollection::SetCompression(bool compression_)
{
compression = compression_;
}
void ParaViewHDFDataCollection::EnsureVTKHDF()
{
+6 -7
View File
@@ -537,13 +537,6 @@ public:
/// Any nonzero compression level will enable compression.
void SetCompressionLevel(int compression_level_);
/// @brief Enable or disable zlib compression.
///
/// If the input is true, use the default zlib compression level (unless the
/// compression level has previously been set by calling
/// SetCompressionLevel()).
void SetCompression(bool compression_) override;
/// @brief Sets whether or not to output the data as high-order elements
/// (false by default).
///
@@ -633,6 +626,12 @@ public:
ParaViewHDFDataCollection(const std::string& collection_name,
Mesh *mesh_ = nullptr);
/// @brief Enable or disable compression.
///
/// The compression level can be set with SetCompressionLevel()). VTKHDF
/// compression does not require MFEM to be compiled with zlib support.
void SetCompression(bool compression_) override;
/// Save the collection.
void Save() override;
+1
View File
@@ -241,6 +241,7 @@ public:
{
MFEM_ASSERT(!action_callbacks.empty(), "no integrators have been set");
prolongation(solutions, solutions_t, solutions_l);
residual_l = 0.0;
for (auto &action : action_callbacks)
{
action(solutions_l, parameters_l, residual_l);
+2 -1
View File
@@ -101,10 +101,11 @@ public:
// Setup DofToQuad information
dtq.nqpt = (int)floor(std::pow(ir.GetNPoints(), 1.0 / mesh.Dimension()) + 0.5);
dtq.ndof = dtq.nqpt;
dtq.mode = used_in_tensor_product ? DofToQuad::TENSOR : DofToQuad::FULL;
// Calculate sizes
const int num_qp = used_in_tensor_product ?
std::pow(dtq.nqpt, mesh.Dimension()) :
static_cast<int>(std::pow(dtq.nqpt, mesh.Dimension())) :
ir.GetNPoints();
tsize = vdim * num_qp * mesh.GetNE();
+9 -8
View File
@@ -987,7 +987,7 @@ get_restriction_transpose(
{
auto RT = [=](const Vector &v_e, Vector &v_l)
{
v_l = v_e;
v_l += v_e;
};
return std::make_tuple(RT, 1);
}
@@ -996,7 +996,7 @@ get_restriction_transpose(
const Operator *R = get_restriction<entity_t>(f, o);
std::function<void(const Vector&, Vector&)> RT = [=](const Vector &x, Vector &y)
{
R->MultTranspose(x, y);
R->AddMultTranspose(x, y);
};
return std::make_tuple(RT, R->Height());
}
@@ -1702,12 +1702,13 @@ std::array<DofToQuadMap, N> load_dtq_mem(
std::array<DofToQuadMap, N> f;
for (std::size_t i = 0; i < N; i++)
{
const auto [nqp_b, dim_b, ndof_b] = dtq[i].B.GetShape();
const auto B = Reshape(&dtq[i].B[0], nqp_b, dim_b, ndof_b);
auto mem_Bi = Reshape(reinterpret_cast<real_t *>(mem) + offset, nqp_b, dim_b,
ndof_b);
if (dtq[i].which_input != -1)
{
const auto [nqp_b, dim_b, ndof_b] = dtq[i].B.GetShape();
const auto B = Reshape(&dtq[i].B[0], nqp_b, dim_b, ndof_b);
auto mem_Bi = Reshape(reinterpret_cast<real_t *>(mem) + offset, nqp_b, dim_b,
ndof_b);
MFEM_FOREACH_THREAD(q, x, nqp_b)
{
MFEM_FOREACH_THREAD(d, y, ndof_b)
@@ -2158,7 +2159,7 @@ template <
std::size_t... Is>
std::array<DofToQuadMap, N> create_dtq_maps_impl(
field_operator_ts &fops,
std::vector<const DofToQuad*> dtqs,
std::vector<const DofToQuad*> &dtqs,
const std::array<int, N> &field_map,
std::index_sequence<Is...>)
{
@@ -2243,7 +2244,7 @@ template <
std::size_t num_fields>
std::array<DofToQuadMap, num_fields> create_dtq_maps(
field_operator_ts &fops,
std::vector<const DofToQuad*> dtqmaps,
std::vector<const DofToQuad*> &dtqmaps,
const std::array<int, num_fields> &to_field_map)
{
return create_dtq_maps_impl<entity_t>(
+10
View File
@@ -20,6 +20,16 @@ namespace mfem
using namespace std;
DofToQuad DofToQuad::Abs() const
{
DofToQuad d2q(*this);
d2q.B.Abs();
d2q.Bt.Abs();
d2q.G.Abs();
d2q.Gt.Abs();
return d2q;
}
FiniteElement::FiniteElement(int D, Geometry::Type G,
int Do, int O, int F)
: Nodes(Do)
+3
View File
@@ -219,6 +219,9 @@ public:
- #ndof x #nqpt, for H(div) vector elements, or
- #ndof x #nqpt x cdim, for H(curl) vector elements. */
Array<real_t> Gt;
/// Returns absolute value of the maps
DofToQuad Abs() const;
};
/// Describes the function space on each element
+1
View File
@@ -49,6 +49,7 @@
#include "lor/lor.hpp"
#include "dgmassinv.hpp"
#include "hyperbolic.hpp"
#include "bounds.hpp"
#include "dfem/doperator.hpp"
-3
View File
@@ -4278,9 +4278,6 @@ void FiniteElementSpace::Update(bool want_transform)
void FiniteElementSpace::PRefineAndUpdate(const Array<pRefinement> & refs,
bool want_transfer)
{
MFEM_VERIFY(PRefinementSupported(),
"p-refinement is not supported in this space");
if (want_transfer)
{
fesPrev.reset(new FiniteElementSpace(mesh, fec, vdim, ordering));
+132 -1
View File
@@ -4334,7 +4334,7 @@ real_t LSZZErrorEstimator(BilinearFormIntegrator &blfi, // input
u.GetSubVector(udofs, ul);
utrans.InvTransformPrimal(ul);
Transf = ufes->GetElementTransformation(ielem);
FiniteElement *dummy = nullptr;
const auto *dummy = ufes->GetFE(ielem);
blfi.ComputeElementFlux(*ufes->GetFE(ielem), *Transf, ul,
*dummy, fl, with_coeff, ir);
@@ -4563,4 +4563,135 @@ GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
return sol2d;
}
void GridFunction::GetElementBoundsAtControlPoints(const int elem,
const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
{
const FiniteElement *fe = fes->GetFE(elem);
int fes_dim = fes->GetVDim();
int rdim = fe->GetDim();
const TensorBasisElement *tbe =
dynamic_cast<const TensorBasisElement *>(fe);
MFEM_VERIFY(tbe != NULL, "TensorBasis FiniteElement expected.");
const Array<int> &dof_map = tbe->GetDofMap();
Vector loc_data;
Array<int> dof_idx;
fes->GetElementDofs(elem, dof_idx);
int ndofs = dof_idx.Size();
int n_c_pts = std::pow(plb.GetNControlPoints(), rdim);
lower.SetSize(n_c_pts*(vdim > 0 ? 1 : fes_dim));
upper.SetSize(n_c_pts*(vdim > 0 ? 1 : fes_dim));
for (int d = 0; d < fes_dim; d++)
{
if (vdim > 0 && d != vdim-1) { continue; }
const int d_off = vdim > 0 ? 0 : d;
Array<int> dof_idx_c = dof_idx;
Vector lowerT(lower, d_off*n_c_pts, n_c_pts);
Vector upperT(upper, d_off*n_c_pts, n_c_pts);
fes->DofsToVDofs(vdim > 0 ? vdim-1 : d, dof_idx_c);
GetSubVector(dof_idx_c, loc_data);
Vector nodal_data;
if (dof_map.Size() == 0)
{
nodal_data.SetDataAndSize(loc_data.GetData(), ndofs);
}
else
{
nodal_data.SetSize(ndofs);
for (int j = 0; j < ndofs; j++)
{
nodal_data(j) = loc_data(dof_map[j]);
}
}
plb.GetNDBounds(rdim, nodal_data, lowerT, upperT);
}
}
void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
{
Vector lowerC, upperC;
GetElementBoundsAtControlPoints(elem, plb, lowerC, upperC, vdim);
const FiniteElement *fe = fes->GetFE(elem);
int rdim = fe->GetDim();
int n_c_pts = std::pow(plb.GetNControlPoints(), rdim);
int fes_dim = fes->GetVDim();
lower.SetSize((vdim > 0 ? 1 :fes_dim));
upper.SetSize((vdim > 0 ? 1 :fes_dim));
for (int d = 0; d < fes_dim; d++)
{
if (vdim > 0 && d != vdim-1) { continue; }
const int d_off = vdim > 0 ? 0 : d;
Vector lowerT(lowerC, d_off*n_c_pts, n_c_pts);
Vector upperT(upperC, d_off*n_c_pts, n_c_pts);
lower(d_off) = lowerT.Min();
upper(d_off) = upperT.Max();
}
}
void GridFunction::GetElementBounds(const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
{
int nel = fes->GetNE();
int fes_dim = fes->GetVDim();
lower.SetSize(nel*(vdim > 0 ? 1 :fes_dim));
upper.SetSize(nel*(vdim > 0 ? 1 :fes_dim));
for (int e = 0; e < nel; e++)
{
Vector lt, ut;
GetElementBounds(e, plb, lt, ut, vdim);
for (int d = 0; d < fes_dim ; d++)
{
if (vdim > 0 && d != vdim-1) { continue; }
const int d_off = vdim > 0 ? 0 : d;
lower(e + d_off*nel) = lt(d_off);
upper(e + d_off*nel) = ut(d_off);
}
}
}
PLBound GridFunction::GetElementBounds(Vector &lower,
Vector &upper,
const int ref_factor,
const int vdim)
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
GetElementBounds(plb, lower, upper, vdim);
return plb;
}
PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
Vector lel, uel;
GetElementBounds(plb, lel, uel, vdim);
int nel = fes->GetNE();
int fes_dim = fes->GetVDim();
lower.SetSize(vdim > 0 ? 1 : fes_dim);
upper.SetSize(vdim > 0 ? 1 : fes_dim);
for (int d = 0; d < fes_dim; d++)
{
if (vdim > 0 && d != vdim-1) { continue; }
const int d_off = vdim > 0 ? 0 : d;
Vector lelt(lel, d_off*nel, nel);
Vector uelt(uel, d_off*nel, nel);
lower(d_off) = lelt.Min();
upper(d_off) = uelt.Max();
}
return plb;
}
}
+47 -1
View File
@@ -16,6 +16,7 @@
#include "fespace.hpp"
#include "coefficient.hpp"
#include "bilininteg.hpp"
#include "bounds.hpp"
#ifdef MFEM_USE_ADIOS2
#include "../general/adios2stream.hpp"
#endif
@@ -1561,11 +1562,56 @@ public:
must be 2 and that quad elements will be broken into two triangles.*/
void SaveSTL(std::ostream &out, int TimesToRefine = 1);
/** @name Methods to compute bounds on the grid function
\brief See bounds.hpp for \ref PLBound that constructs piecewise linear
bounds for a given set of bases. These piecewise bounds can be used to compute bounds on a grid function. Currently tensor-product elements are
supported with Lagrange interpolants on Gauss Legendre nodes and Gauss Lobatto Legendre nodes, and Bernstein bases.
*/
///@{
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the overall bounds for each
/// vdim (across all elements) in @b lower and @b upper. We also return the
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
/// Note: For most cases, this method/interface will be sufficient.
virtual PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each element
/// ordered byVDim:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}. We also return the
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
PLBound GetElementBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
/// Compute piecewise linear bounds on the given element at the grid of
/// [plb.ncp x plb.ncp x plb.ncp] control points for each of the vdim
/// components of the gridfunction.
void GetElementBoundsAtControlPoints(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
/// Compute bounds on the grid function for the given element.
/// The bounds are stored in @b lower and @b upper.
void GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
/// Compute bounds on the grid function for all the elements. The bounds
/// are returned in @b lower and @b upper, ordered byVDim:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
const int vdim=-1);
///@}
/// Destroys grid function.
virtual ~GridFunction() { Destroy(); }
};
/** Overload operator<< for std::ostream and GridFunction; valid also for the
derived class ParGridFunction */
std::ostream &operator<<(std::ostream &out, const GridFunction &sol);
+3 -1
View File
@@ -30,7 +30,9 @@ namespace mfem
{
/** \brief FindPointsGSLIB can robustly evaluate a GridFunction on an arbitrary
* collection of points.
* collection of points. See Mittal et al., "General Field Evaluation in
* High-Order Meshes on GPUs". (2025). Computers & Fluids. for technical
* details.
*
* There are three key functions in FindPointsGSLIB:
*
+64
View File
@@ -202,4 +202,68 @@ void CurlCurlIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
}
void CurlCurlIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
{
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
auto absO = mapsO->Abs();
auto absC = mapsC->Abs();
if (dim == 3)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23:
return internal::SmemPACurlCurlApply3D<2,3>(
dofs1D, quad1D,
symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
case 0x34:
return internal::SmemPACurlCurlApply3D<3,4>(
dofs1D, quad1D,
symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
case 0x45:
return internal::SmemPACurlCurlApply3D<4,5>(
dofs1D, quad1D,
symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
case 0x56:
return internal::SmemPACurlCurlApply3D<5,6>(
dofs1D, quad1D,
symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
default:
return internal::SmemPACurlCurlApply3D<0,0>(
dofs1D, quad1D, symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
}
}
else
{
internal::PACurlCurlApply3D<0,0>(
dofs1D, quad1D, symmetric, ne,
absO.B, absC.B, absO.Bt, absC.Bt, absC.G, absC.Gt,
abs_pa_data, x, y, true);
}
}
else if (dim == 2)
{
internal::PACurlCurlApply2D(dofs1D, quad1D, ne, absO.B, absO.Bt,
absC.G, absC.Gt, abs_pa_data, x, y, true);
}
else
{
MFEM_ABORT("Unsupported dimension!");
}
}
} // namespace mfem
+26 -38
View File
@@ -483,19 +483,6 @@ inline void SmemPADiffusionDiagonal3D(const int NE,
});
}
void PADiffusionApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const bool symm,
const Array<real_t> &B,
const Array<real_t> &G,
const Array<real_t> &Bt,
const Array<real_t> &Gt,
const Vector &D,
const Vector &X,
Vector &Y);
#ifdef MFEM_USE_OCCA
// OCCA PA Diffusion Apply 2D kernel
void OccaPADiffusionApply2D(const int D1D,
@@ -1022,6 +1009,7 @@ inline void SmemPADiffusionApply3D(const int NE,
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_VERIFY(D1D <= Q1D, "THREAD_DIRECT requires D1D <= Q1D");
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -1051,11 +1039,11 @@ inline void SmemPADiffusionApply3D(const int NE,
real_t (*QDD0)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+0);
real_t (*QDD1)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+1);
real_t (*QDD2)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
@@ -1063,9 +1051,9 @@ inline void SmemPADiffusionApply3D(const int NE,
}
if (MFEM_THREAD_ID(z) == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
{
B[qx][dy] = b(qx,dy);
G[qx][dy] = g(qx,dy);
@@ -1073,11 +1061,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0;
MFEM_UNROLL(MD1)
@@ -1093,11 +1081,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MD1)
@@ -1114,11 +1102,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MD1)
@@ -1149,9 +1137,9 @@ inline void SmemPADiffusionApply3D(const int NE,
MFEM_SYNC_THREAD;
if (MFEM_THREAD_ID(z) == 0)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
{
Bt[dy][qx] = b(qx,dy);
Gt[dy][qx] = g(qx,dy);
@@ -1159,11 +1147,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MQ1)
@@ -1180,11 +1168,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(qz,z,Q1D)
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(Q1D)
@@ -1201,11 +1189,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(dz,z,D1D)
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
{
MFEM_FOREACH_THREAD(dy,y,D1D)
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MQ1)
+30
View File
@@ -164,6 +164,36 @@ void DiffusionIntegrator::AssemblePatchPA(const int patch,
SetupPatchPA(patch, mesh); // For full quadrature, unitWeights = false
}
void DiffusionIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
{
if (DeviceCanUseCeed())
{
MFEM_ABORT("Ceed AbsMult not implemented yet");
}
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
auto abs_maps = maps->Abs();
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
abs_pa_data, x, y, dofs1D, quad1D);
}
void DiffusionIntegrator::AddAbsMultTransposePA(const Vector &x,
Vector &y) const
{
if (symmetric)
{
AddAbsMultPA(x, y);
}
else
{
MFEM_ABORT("DiffusionIntegrator::AddAbsMultTransposePA only implemented "
"in the symmetric case.")
}
}
// This version uses full 1D quadrature rules, taking into account the
// minimum interaction between basis functions and integration points.
void DiffusionIntegrator::AddMultPatchPA(const int patch, const Vector &x,
+6 -5
View File
@@ -212,7 +212,7 @@ void ElasticityAddMultPA_(const int nDofs, const FiniteElementSpace &fespace,
const int iIndex = isComponent ? 0 : i;
div += gradx(iIndex,i);
}
const real_t w = ipWeights[p] /det(invJ);
const real_t w = ipWeights[p]/det(invJ);
for (int m = 0; m < d; m++)
{
for (int q = qLower; q < qUpper; q++)
@@ -226,8 +226,8 @@ void ElasticityAddMultPA_(const int nDofs, const FiniteElementSpace &fespace,
{
for (int a = 0; a < d; a++)
{
contraction += 2*((a == q)*invJ(m,j_block) + (j_block==q)*invJ(m,a))*(gradx(0,
a));
contraction += 2*((a == q)*invJ(m,j_block)
+ (j_block==q)*invJ(m,a))*(gradx(0, a));
}
}
else
@@ -236,7 +236,7 @@ void ElasticityAddMultPA_(const int nDofs, const FiniteElementSpace &fespace,
{
for (int b = 0; b < d; b++)
{
contraction += ((a == q)*invJ(m,b) + (b==q)*invJ(m,a))
contraction += ((a == q)*invJ(m,b) + (b == q)*invJ(m,a))
*(gradx(a,b) + gradx(b, a));
}
}
@@ -244,7 +244,8 @@ void ElasticityAddMultPA_(const int nDofs, const FiniteElementSpace &fespace,
// lambda*div(u)*div(v) + 2*mu*sym(grad(u))*sym(grad(v))
// contraction = 4*sym(grad(u))sym(grad(v))
const int qIndex = isComponent ? 0 : q;
Q(p,m,qIndex,e) = w*(lamDev(p, e)*invJ(m,q)*div + 0.5*muDev(p, e)*contraction);
Q(p,m,qIndex,e) = w*(lamDev(p, e)*invJ(m,q)*div
+ 0.5*muDev(p, e)*contraction);
}
}
}
+6 -3
View File
@@ -662,7 +662,8 @@ void PACurlCurlApply2D(const int D1D,
const Array<real_t> &gct,
const Vector &pa_data,
const Vector &x,
Vector &y)
Vector &y,
const bool useAbs)
{
auto Bo = Reshape(bo.Read(), Q1D, D1D-1);
@@ -717,7 +718,8 @@ void PACurlCurlApply2D(const int D1D,
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t wy = (c == 0) ? -Gc(qy,dy) : Bo(qy,dy);
const int sign = useAbs ? 1 : -1;
const real_t wy = (c == 0) ? (sign*Gc(qy,dy)) : Bo(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
curl[qy][qx] += gradX[qx] * wy;
@@ -760,7 +762,8 @@ void PACurlCurlApply2D(const int D1D,
}
for (int dy = 0; dy < D1Dy; ++dy)
{
const real_t wy = (c == 0) ? -Gct(dy,qy) : Bot(dy,qy);
const int sign = useAbs ? 1 : -1;
const real_t wy = (c == 0) ? (sign*Gct(dy,qy)) : Bot(dy,qy);
for (int dx = 0; dx < D1Dx; ++dx)
{
+132 -27
View File
@@ -828,7 +828,7 @@ inline void SmemPACurlCurlAssembleDiagonal3D(const int d1d,
}); // end of element loop
}
// PA H(curl) curl-curl Apply 2D kernel
// PA H(curl) curl-curl Apply/AbsApply 2D kernel
void PACurlCurlApply2D(const int D1D,
const int Q1D,
const int NE,
@@ -838,9 +838,10 @@ void PACurlCurlApply2D(const int D1D,
const Array<real_t> &gct,
const Vector &pa_data,
const Vector &x,
Vector &y);
Vector &y,
const bool useAbs = false);
// PA H(curl) curl-curl Apply 3D kernel
// PA H(curl) curl-curl Apply/AbsApply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void PACurlCurlApply3D(const int d1d,
const int q1d,
@@ -854,7 +855,8 @@ inline void PACurlCurlApply3D(const int d1d,
const Array<real_t> &gct,
const Vector &pa_data,
const Vector &x,
Vector &y)
Vector &y,
const bool useAbs = false)
{
MFEM_VERIFY(T_D1D || d1d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Error: d1d > HCURL_MAX_D1D");
@@ -970,7 +972,16 @@ inline void PACurlCurlApply3D(const int d1d,
{
// \hat{\nabla}\times\hat{u} is [0, (u_0)_{x_2}, -(u_0)_{x_1}]
curl[qz][qy][qx][1] += gradXY[qy][qx][1] * wDz; // (u_0)_{x_2}
curl[qz][qy][qx][2] -= gradXY[qy][qx][0] * wz; // -(u_0)_{x_1}
if (useAbs)
{
// +(u_0)_{x_1}
curl[qz][qy][qx][2] += gradXY[qy][qx][0] * wz;
}
else
{
// -(u_0)_{x_1}
curl[qz][qy][qx][2] -= gradXY[qy][qx][0] * wz;
}
}
}
}
@@ -1038,7 +1049,16 @@ inline void PACurlCurlApply3D(const int d1d,
for (int qx = 0; qx < Q1D; ++qx)
{
// \hat{\nabla}\times\hat{u} is [-(u_1)_{x_2}, 0, (u_1)_{x_0}]
curl[qz][qy][qx][0] -= gradXY[qy][qx][1] * wDz; // -(u_1)_{x_2}
if (useAbs)
{
// +(u_1)_{x_2}
curl[qz][qy][qx][0] += gradXY[qy][qx][1] * wDz;
}
else
{
// -(u_1)_{x_2}
curl[qz][qy][qx][0] -= gradXY[qy][qx][1] * wDz;
}
curl[qz][qy][qx][2] += gradXY[qy][qx][0] * wz; // (u_1)_{x_0}
}
}
@@ -1109,7 +1129,16 @@ inline void PACurlCurlApply3D(const int d1d,
{
// \hat{\nabla}\times\hat{u} is [(u_2)_{x_1}, -(u_2)_{x_0}, 0]
curl[qz][qy][qx][0] += gradYZ[qz][qy][1] * wx; // (u_2)_{x_1}
curl[qz][qy][qx][1] -= gradYZ[qz][qy][0] * wDx; // -(u_2)_{x_0}
if (useAbs)
{
// +(u_2)_{x_0}
curl[qz][qy][qx][1] += gradYZ[qz][qy][0] * wDx;
}
else
{
// -(u_2)_{x_0}
curl[qz][qy][qx][1] -= gradYZ[qz][qy][0] * wDx;
}
}
}
}
@@ -1209,9 +1238,21 @@ inline void PACurlCurlApply3D(const int d1d,
for (int dx = 0; dx < D1Dx; ++dx)
{
// \hat{\nabla}\times\hat{u} is [0, (u_0)_{x_2}, -(u_0)_{x_1}]
// (u_0)_{x_2} * (op * curl)_1 - (u_0)_{x_1} * (op * curl)_2
Y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc,
e) += (gradXY21[dy][dx] * wDz) - (gradXY12[dy][dx] * wz);
const int idx = dx + ((dy + (dz * D1Dy)) * D1Dx) + osc;
if (useAbs)
{
// (u_0)_{x_2} * (op * curl)_1 +
// (u_0)_{x_1} * (op * curl)_2
Y(idx, e) += (gradXY21[dy][dx] * wDz) +
(gradXY12[dy][dx] * wz);
}
else
{
// (u_0)_{x_2} * (op * curl)_1 -
// (u_0)_{x_1} * (op * curl)_2
Y(idx, e) += (gradXY21[dy][dx] * wDz) -
(gradXY12[dy][dx] * wz);
}
}
}
}
@@ -1278,10 +1319,22 @@ inline void PACurlCurlApply3D(const int d1d,
{
for (int dx = 0; dx < D1Dx; ++dx)
{
const int idx = dx + ((dy + (dz * D1Dy)) * D1Dx) + osc;
// \hat{\nabla}\times\hat{u} is [-(u_1)_{x_2}, 0, (u_1)_{x_0}]
// -(u_1)_{x_2} * (op * curl)_0 + (u_1)_{x_0} * (op * curl)_2
Y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc,
e) += (-gradXY20[dy][dx] * wDz) + (gradXY02[dy][dx] * wz);
if (useAbs)
{
// +(u_1)_{x_2} * (op * curl)_0 +
// (u_1)_{x_0} * (op * curl)_2
Y(idx, e) += (gradXY20[dy][dx] * wDz) +
(gradXY02[dy][dx] * wz);
}
else
{
// -(u_1)_{x_2} * (op * curl)_0 +
// (u_1)_{x_0} * (op * curl)_2
Y(idx, e) += (-gradXY20[dy][dx] * wDz) +
(gradXY02[dy][dx] * wz);
}
}
}
}
@@ -1351,10 +1404,22 @@ inline void PACurlCurlApply3D(const int d1d,
{
for (int dz = 0; dz < D1Dz; ++dz)
{
const int idx = dx + ((dy + (dz * D1Dy)) * D1Dx) + osc;
// \hat{\nabla}\times\hat{u} is [(u_2)_{x_1}, -(u_2)_{x_0}, 0]
// (u_2)_{x_1} * (op * curl)_0 - (u_2)_{x_0} * (op * curl)_1
Y(dx + ((dy + (dz * D1Dy)) * D1Dx) + osc,
e) += (gradYZ10[dz][dy] * wx) - (gradYZ01[dz][dy] * wDx);
if (useAbs)
{
// (u_2)_{x_1} * (op * curl)_0 +
// (u_2)_{x_0} * (op * curl)_1
Y(idx, e) += (gradYZ10[dz][dy] * wx) +
(gradYZ01[dz][dy] * wDx);
}
else
{
// (u_2)_{x_1} * (op * curl)_0 -
// (u_2)_{x_0} * (op * curl)_1
Y(idx, e) += (gradYZ10[dz][dy] * wx) -
(gradYZ01[dz][dy] * wDx);
}
}
}
}
@@ -1363,7 +1428,7 @@ inline void PACurlCurlApply3D(const int d1d,
}); // end of element loop
}
// Shared memory PA H(curl) curl-curl Apply 3D kernel
// Shared memory PA H(curl) curl-curl Apply/AbsApply 3D kernel
template<int T_D1D = 0, int T_Q1D = 0>
inline void SmemPACurlCurlApply3D(const int d1d,
const int q1d,
@@ -1377,7 +1442,8 @@ inline void SmemPACurlCurlApply3D(const int d1d,
const Array<real_t> &gct,
const Vector &pa_data,
const Vector &x,
Vector &y)
Vector &y,
const bool useAbs = false)
{
MFEM_VERIFY(T_D1D || d1d <= DeviceDofQuadLimits::Get().HCURL_MAX_D1D,
"Error: d1d > HCURL_MAX_D1D");
@@ -1531,7 +1597,8 @@ inline void SmemPACurlCurlApply3D(const int d1d,
}
curl[qy][qx][1] += v; // (u_0)_{x_2}
curl[qy][qx][2] -= u; // -(u_0)_{x_1}
if (useAbs) { curl[qy][qx][2] += u; } // +(u_0)_{x_1}
else { curl[qy][qx][2] -= u; } // -(u_0)_{x_1}
}
else if (c == 1) // y component
{
@@ -1558,7 +1625,8 @@ inline void SmemPACurlCurlApply3D(const int d1d,
}
}
curl[qy][qx][0] -= v; // -(u_1)_{x_2}
if (useAbs) { curl[qy][qx][0] += v; } // +(u_1)_{x_2}
else { curl[qy][qx][0] -= v; } // -(u_1)_{x_2}
curl[qy][qx][2] += u; // (u_1)_{x_0}
}
else // z component
@@ -1587,7 +1655,8 @@ inline void SmemPACurlCurlApply3D(const int d1d,
}
curl[qy][qx][0] += v; // (u_2)_{x_1}
curl[qy][qx][1] -= u; // -(u_2)_{x_0}
if (useAbs) { curl[qy][qx][1] += u; }// +(u_2)_{x_0}
else { curl[qy][qx][1] -= u; } // -(u_2)_{x_0}
}
} // qx
} // qy
@@ -1642,18 +1711,54 @@ inline void SmemPACurlCurlApply3D(const int d1d,
if (dx < D1D-1)
{
// \hat{\nabla}\times\hat{u} is [0, (u_0)_{x_2}, -(u_0)_{x_1}]
// (u_0)_{x_2} * (op * curl)_1 - (u_0)_{x_1} * (op * curl)_2
const real_t wx = sBo[dx][qx];
dxyz1 += (wx * c2 * wcy * wcDz) - (wx * c3 * wcDy * wcz);
if (useAbs)
{
// (u_0)_{x_2} * (op * curl)_1 +
// (u_0)_{x_1} * (op * curl)_2
dxyz1 += (wx * c2 * wcy * wcDz) +
(wx * c3 * wcDy * wcz);
}
else
{
// (u_0)_{x_2} * (op * curl)_1 -
// (u_0)_{x_1} * (op * curl)_2
dxyz1 += (wx * c2 * wcy * wcDz) -
(wx * c3 * wcDy * wcz);
}
}
// \hat{\nabla}\times\hat{u} is [-(u_1)_{x_2}, 0, (u_1)_{x_0}]
// -(u_1)_{x_2} * (op * curl)_0 + (u_1)_{x_0} * (op * curl)_2
dxyz2 += (-wy * c1 * wcx * wcDz) + (wy * c3 * wDx * wcz);
if (useAbs)
{
// +(u_1)_{x_2} * (op * curl)_0 +
// (u_1)_{x_0} * (op * curl)_2
dxyz2 += (wy * c1 * wcx * wcDz) +
(wy * c3 * wDx * wcz);
}
else
{
// -(u_1)_{x_2} * (op * curl)_0 +
// (u_1)_{x_0} * (op * curl)_2
dxyz2 += (-wy * c1 * wcx * wcDz) +
(wy * c3 * wDx * wcz);
}
// \hat{\nabla}\times\hat{u} is [(u_2)_{x_1}, -(u_2)_{x_0}, 0]
// (u_2)_{x_1} * (op * curl)_0 - (u_2)_{x_0} * (op * curl)_1
dxyz3 += (wcDy * wz * c1 * wcx) - (wcy * wz * c2 * wDx);
if (useAbs)
{
// (u_2)_{x_1} * (op * curl)_0 +
// (u_2)_{x_0} * (op * curl)_1
dxyz3 += (wcDy * wz * c1 * wcx) +
(wcy * wz * c2 * wDx);
}
else
{
// (u_2)_{x_1} * (op * curl)_0 -
// (u_2)_{x_0} * (op * curl)_1
dxyz3 += (wcDy * wz * c1 * wcx) -
(wcy * wz * c2 * wDx);
}
} // qx
} // qy
} // dx
+28 -1
View File
@@ -62,7 +62,7 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
const int NE = ne;
const int Q1D = quad1D;
const int NQ = pow(Q1D, dim);
const int NQ = static_cast<int>(std::pow(Q1D, dim));
const bool const_c = coeff.Size() == 1;
const bool by_val = map_type == FiniteElement::VALUE;
const auto W = Reshape(ir->GetWeights().Read(), NQ);
@@ -199,10 +199,37 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
}
void MassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
{
if (DeviceCanUseCeed())
{
MFEM_ABORT("AddAbsMultPA not implemented with CEED!");
ceedOp->AddMult(x, y);
}
else
{
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absB(maps->B);
Array<real_t> absBt(maps->Bt);
absB.Abs();
absBt.Abs();
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, absB, absBt, abs_pa_data,
x, y, dofs1D, quad1D);
}
}
void MassIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
{
// Mass integrator is symmetric
AddMultPA(x, y);
}
void MassIntegrator::AddAbsMultTransposePA(const Vector &x, Vector &y) const
{
// Mass integrator is symmetric
AddAbsMultPA(x, y);
}
} // namespace mfem
+123
View File
@@ -313,6 +313,129 @@ void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
}
void VectorFEMassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
{
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
Vector abs_pa_data(pa_data);
abs_pa_data.Abs();
Array<real_t> absBo(mapsO->B);
Array<real_t> absBc(mapsC->B);
Array<real_t> absBto(mapsO->Bt);
Array<real_t> absBtc(mapsC->Bt);
Array<real_t> absBto_t(mapsOtest->Bt);
Array<real_t> absBtc_t(mapsCtest->Bt);
absBo.Abs();
absBc.Abs();
absBto.Abs();
absBtc.Abs();
absBto_t.Abs();
absBtc_t.Abs();
if (dim == 3)
{
if (trial_curl && test_curl)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23:
return internal::SmemPAHcurlMassApply3D<2,3>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x34:
return internal::SmemPAHcurlMassApply3D<3,4>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x45:
return internal::SmemPAHcurlMassApply3D<4,5>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
case 0x56:
return internal::SmemPAHcurlMassApply3D<5,6>(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
default:
return internal::SmemPAHcurlMassApply3D(
dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
}
else
{
internal::PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(3, dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if (trial_curl && test_div)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, true, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else if (trial_div && test_curl)
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, false, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
else // 2D
{
if (trial_curl && test_curl)
{
internal::PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if (trial_div && test_div)
{
internal::PAHdivMassApply(2, dofs1D, quad1D, ne, symmetric,
absBo, absBc, absBto, absBtc,
abs_pa_data, x, y);
}
else if ((trial_curl && test_div) || (trial_div && test_curl))
{
const bool scalarCoeff = !(DQ || MQ);
internal::PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne,
scalarCoeff, trial_curl, false,
absBo, absBc, absBto_t, absBtc_t,
abs_pa_data, x, y);
}
else
{
MFEM_ABORT("Unknown kernel.");
}
}
}
void VectorFEMassIntegrator::AddMultTransposePA(const Vector &x,
Vector &y) const
{
+1 -1
View File
@@ -673,7 +673,7 @@ public:
int myid;
MPI_Comm_rank(comm, &myid);
int seed = (seed_ > 0) ? seed_ + myid : (int)time(0) + myid;
int seed = (seed_ > 0) ? seed_ + myid : time(nullptr) + myid;
SetSeed(seed);
}
#else
+2 -2
View File
@@ -5259,7 +5259,7 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
gc.GetNeighborLTDofTable(nbr_ltdof);
const int nb_connections = nbr_ltdof.Size_of_connections();
shr_ltdof.SetSize(nb_connections);
shr_ltdof.CopyFrom(nbr_ltdof.GetJ());
if (nb_connections > 0) { shr_ltdof.CopyFrom(nbr_ltdof.GetJ()); }
shr_buf.SetSize(nb_connections);
shr_buf.UseDevice(true);
shr_buf_offsets = nbr_ltdof.GetIMemory();
@@ -5288,7 +5288,7 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
gc.GetNeighborLDofTable(nbr_ldof);
const int nb_connections = nbr_ldof.Size_of_connections();
ext_ldof.SetSize(nb_connections);
ext_ldof.CopyFrom(nbr_ldof.GetJ());
if (nb_connections > 0) { ext_ldof.CopyFrom(nbr_ldof.GetJ()); }
ext_ldof.GetMemory().UseDevice(true);
ext_buf.SetSize(nb_connections);
ext_buf.UseDevice(true);
+12
View File
@@ -577,7 +577,13 @@ public:
void Mult(const Vector &x, Vector &y) const override;
void AbsMult(const Vector &x, Vector &y) const override
{ Mult(x,y); }
void MultTranspose(const Vector &x, Vector &y) const override;
void AbsMultTranspose(const Vector &x, Vector &y) const override
{ MultTranspose(x,y); }
};
/// Auxiliary device class used by ParFiniteElementSpace.
@@ -628,7 +634,13 @@ public:
void Mult(const Vector &x, Vector &y) const override;
void AbsMult(const Vector &x, Vector &y) const override
{ Mult(x,y); }
void MultTranspose(const Vector &x, Vector &y) const override;
void AbsMultTranspose(const Vector &x, Vector &y) const override
{ MultTranspose(x,y); }
};
}
+12
View File
@@ -1406,6 +1406,18 @@ real_t L2ZZErrorEstimator(BilinearFormIntegrator &flux_integrator,
return pow(glob_error, 1.0/norm_p);
}
PLBound ParGridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
{
PLBound plb = GridFunction::GetBounds(lower, upper, ref_factor, vdim);
int siz = vdim > 0 ? 1 : fes->GetVDim();
MPI_Allreduce(MPI_IN_PLACE, lower.HostReadWrite(), siz,
MFEM_MPI_REAL_T, MPI_MIN, pfes->GetComm());
MPI_Allreduce(MPI_IN_PLACE, upper.HostReadWrite(), siz,
MFEM_MPI_REAL_T, MPI_MAX, pfes->GetComm());
return plb;
}
} // namespace mfem
#endif // MFEM_USE_MPI
+8
View File
@@ -581,6 +581,14 @@ public:
GridFunction &flux,
bool wcoef = true, int subdomain = -1) override;
/// Computes the PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each
/// vdim across all elements in @b lower and @b upper. We also return the
/// PLBound object used to compute the bounds. Note: if vdim < 1, we compute
/// the bounds for each vector dimension.
PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) override;
/** Save the local portion of the ParGridFunction. This differs from the
serial GridFunction::Save in that it takes into account the signs of
the local dofs. */
+3 -280
View File
@@ -9,278 +9,16 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../quadinterpolator.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../fem/kernels.hpp"
#include "../../linalg/kernels.hpp"
using namespace mfem;
#include "det.hpp"
namespace mfem
{
namespace internal
{
namespace quadrature_interpolator
{
static void Det1D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d,
const int q1d,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(d_buff);
const auto G = Reshape(g, q1d, d1d);
const auto X = Reshape(x, d1d, NE);
auto Y = Reshape(y, q1d, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < q1d; q++)
{
real_t u = 0.0;
for (int d = 0; d < d1d; d++)
{
u += G(q, d) * X(d, e);
}
Y(q, e) = u;
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void Det2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 2;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XY[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
MFEM_SHARED real_t QQ[2*SDIM][NBZ][MQ1*MQ1];
kernels::internal::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::internal::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[4];
kernels::internal::PullGrad<MQ1,NBZ>(Q1D,qx,qy,QQ,J);
Y(qx,qy,e) = kernels::Det<2>(J);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void Det2DSurface(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 3;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XYZ[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
// Load XYZ components
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int d = 0; d < SDIM; ++d)
{
XYZ[d][tidz][dx + dy*D1D] = X(dx,dy,d,e);
}
}
}
MFEM_SYNC_THREAD;
ConstDeviceMatrix B_mat(BG[0], D1D, Q1D);
ConstDeviceMatrix G_mat(BG[1], D1D, Q1D);
// x contraction
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
for (int d = 0; d < SDIM; ++d)
{
real_t u = 0.0;
real_t v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const real_t xval = XYZ[d][tidz][dx + dy*D1D];
u += xval * G_mat(dx,qx);
v += xval * B_mat(dx,qx);
}
DQ[d][tidz][dy + qx*D1D] = u;
DQ[3 + d][tidz][dy + qx*D1D] = v;
}
}
}
MFEM_SYNC_THREAD;
// y contraction and determinant computation
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J_[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
for (int d = 0; d < SDIM; ++d)
{
for (int dy = 0; dy < D1D; ++dy)
{
J_[d] += DQ[d][tidz][dy + qx*D1D] * B_mat(dy,qy);
J_[3 + d] += DQ[3 + d][tidz][dy + qx*D1D] * G_mat(dy,qy);
}
}
DeviceTensor<2> J(J_, 3, 2);
const real_t E = J(0,0)*J(0,0) + J(1,0)*J(1,0) + J(2,0)*J(2,0);
const real_t F = J(0,0)*J(0,1) + J(1,0)*J(1,1) + J(2,0)*J(2,1);
const real_t G = J(0,1)*J(0,1) + J(1,1)*J(1,1) + J(2,1)*J(2,1);
Y(qx,qy,e) = std::sqrt(E*G - F*F);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, bool SMEM = true>
static void Det3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr) // used only with SMEM = false
{
constexpr int DIM = 3;
static constexpr int GRID = SMEM ? 0 : 128;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
real_t *GM = nullptr;
if (!SMEM)
{
const DeviceDofQuadLimits &limits = DeviceDofQuadLimits::Get();
const int max_q1d = T_Q1D ? T_Q1D : limits.MAX_Q1D;
const int max_d1d = T_D1D ? T_D1D : limits.MAX_D1D;
const int max_qd = std::max(max_q1d, max_d1d);
const int mem_size = max_qd * max_qd * max_qd * 9;
d_buff->SetSize(2*mem_size*GRID);
GM = d_buff->Write();
}
mfem::forall_3D_grid(NE, Q1D, Q1D, Q1D, GRID, [=] MFEM_HOST_DEVICE (int e)
{
static constexpr int MQ1 = T_Q1D ? T_Q1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_Q1D);
static constexpr int MD1 = T_D1D ? T_D1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_D1D);
static constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
static constexpr int MSZ = MDQ * MDQ * MDQ * 9;
const int bid = MFEM_BLOCK_ID(x);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t SM0[SMEM?MSZ:1];
MFEM_SHARED real_t SM1[SMEM?MSZ:1];
real_t *lm0 = SMEM ? SM0 : GM + MSZ*bid;
real_t *lm1 = SMEM ? SM1 : GM + MSZ*(GRID+bid);
real_t (*DDD)[MD1*MD1*MD1] = (real_t (*)[MD1*MD1*MD1]) (lm0);
real_t (*DDQ)[MD1*MD1*MQ1] = (real_t (*)[MD1*MD1*MQ1]) (lm1);
real_t (*DQQ)[MD1*MQ1*MQ1] = (real_t (*)[MD1*MQ1*MQ1]) (lm0);
real_t (*QQQ)[MQ1*MQ1*MQ1] = (real_t (*)[MQ1*MQ1*MQ1]) (lm1);
kernels::internal::LoadX<MD1>(e,D1D,X,DDD);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
kernels::internal::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
kernels::internal::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[9];
kernels::internal::PullGrad<MQ1>(Q1D, qx,qy,qz, QQQ, J);
Y(qx,qy,qz,e) = kernels::Det<3>(J);
}
}
}
});
}
void InitDetKernels()
{
using k = QuadratureInterpolator::DetKernels;
@@ -302,27 +40,12 @@ void InitDetKernels()
}
} // namespace quadrature_interpolator
} // namespace internal
/// @cond Suppress_Doxygen_warnings
namespace
{
using DetKernel = QuadratureInterpolator::DetKernelType;
}
template<int DIM, int SDIM, int D1D, int Q1D>
DetKernel QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
}
DetKernel QuadratureInterpolator::DetKernels::Fallback(
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Fallback(
int DIM, int SDIM, int D1D, int Q1D)
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
+304
View File
@@ -0,0 +1,304 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_QUADINTERP_DET_HPP
#define MFEM_QUADINTERP_DET_HPP
#include "../quadinterpolator.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../fem/kernels.hpp"
#include "../../linalg/kernels.hpp"
namespace mfem
{
namespace internal
{
namespace quadrature_interpolator
{
inline void Det1D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d,
const int q1d,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(d_buff);
const auto G = Reshape(g, q1d, d1d);
const auto X = Reshape(x, d1d, NE);
auto Y = Reshape(y, q1d, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < q1d; q++)
{
real_t u = 0.0;
for (int d = 0; d < d1d; d++)
{
u += G(q, d) * X(d, e);
}
Y(q, e) = u;
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void Det2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 2;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XY[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
MFEM_SHARED real_t QQ[2*SDIM][NBZ][MQ1*MQ1];
kernels::internal::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::internal::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[4];
kernels::internal::PullGrad<MQ1,NBZ>(Q1D,qx,qy,QQ,J);
Y(qx,qy,e) = kernels::Det<2>(J);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void Det2DSurface(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 3;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XYZ[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
// Load XYZ components
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int d = 0; d < SDIM; ++d)
{
XYZ[d][tidz][dx + dy*D1D] = X(dx,dy,d,e);
}
}
}
MFEM_SYNC_THREAD;
ConstDeviceMatrix B_mat(BG[0], D1D, Q1D);
ConstDeviceMatrix G_mat(BG[1], D1D, Q1D);
// x contraction
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
for (int d = 0; d < SDIM; ++d)
{
real_t u = 0.0;
real_t v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const real_t xval = XYZ[d][tidz][dx + dy*D1D];
u += xval * G_mat(dx,qx);
v += xval * B_mat(dx,qx);
}
DQ[d][tidz][dy + qx*D1D] = u;
DQ[3 + d][tidz][dy + qx*D1D] = v;
}
}
}
MFEM_SYNC_THREAD;
// y contraction and determinant computation
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J_[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
for (int d = 0; d < SDIM; ++d)
{
for (int dy = 0; dy < D1D; ++dy)
{
J_[d] += DQ[d][tidz][dy + qx*D1D] * B_mat(dy,qy);
J_[3 + d] += DQ[3 + d][tidz][dy + qx*D1D] * G_mat(dy,qy);
}
}
DeviceTensor<2> J(J_, 3, 2);
const real_t E = J(0,0)*J(0,0) + J(1,0)*J(1,0) + J(2,0)*J(2,0);
const real_t F = J(0,0)*J(0,1) + J(1,0)*J(1,1) + J(2,0)*J(2,1);
const real_t G = J(0,1)*J(0,1) + J(1,1)*J(1,1) + J(2,1)*J(2,1);
Y(qx,qy,e) = std::sqrt(E*G - F*F);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, bool SMEM = true>
inline void Det3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr) // used only with SMEM = false
{
constexpr int DIM = 3;
static constexpr int GRID = SMEM ? 0 : 128;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
real_t *GM = nullptr;
if (!SMEM)
{
const DeviceDofQuadLimits &limits = DeviceDofQuadLimits::Get();
const int max_q1d = T_Q1D ? T_Q1D : limits.MAX_Q1D;
const int max_d1d = T_D1D ? T_D1D : limits.MAX_D1D;
const int max_qd = std::max(max_q1d, max_d1d);
const int mem_size = max_qd * max_qd * max_qd * 9;
d_buff->SetSize(2*mem_size*GRID);
GM = d_buff->Write();
}
mfem::forall_3D_grid(NE, Q1D, Q1D, Q1D, GRID, [=] MFEM_HOST_DEVICE (int e)
{
static constexpr int MQ1 = T_Q1D ? T_Q1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_Q1D);
static constexpr int MD1 = T_D1D ? T_D1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_D1D);
static constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
static constexpr int MSZ = MDQ * MDQ * MDQ * 9;
const int bid = MFEM_BLOCK_ID(x);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t SM0[SMEM?MSZ:1];
MFEM_SHARED real_t SM1[SMEM?MSZ:1];
real_t *lm0 = SMEM ? SM0 : GM + MSZ*bid;
real_t *lm1 = SMEM ? SM1 : GM + MSZ*(GRID+bid);
real_t (*DDD)[MD1*MD1*MD1] = (real_t (*)[MD1*MD1*MD1]) (lm0);
real_t (*DDQ)[MD1*MD1*MQ1] = (real_t (*)[MD1*MD1*MQ1]) (lm1);
real_t (*DQQ)[MD1*MQ1*MQ1] = (real_t (*)[MD1*MQ1*MQ1]) (lm0);
real_t (*QQQ)[MQ1*MQ1*MQ1] = (real_t (*)[MQ1*MQ1*MQ1]) (lm1);
kernels::internal::LoadX<MD1>(e,D1D,X,DDD);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
kernels::internal::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
kernels::internal::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[9];
kernels::internal::PullGrad<MQ1>(Q1D, qx,qy,qz, QQQ, J);
Y(qx,qy,qz,e) = kernels::Det<3>(J);
}
}
}
});
}
} // namespace quadrature_interpolator
} // namespace internal
/// @cond Suppress_Doxygen_warnings
template<int DIM, int SDIM, int D1D, int Q1D>
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
}
/// @endcond
} // namespace mfem
#endif // MFEM_QUADINTERP_DET_HPP
+2 -2
View File
@@ -566,8 +566,8 @@ void QuadratureInterpolator::Mult(const Vector &e_vec,
}
else // use_tensor_eval == false
{
EvalKernels::Run(dim, vdim, maps.ndof, maps.nqpt, ne,vdim,q_layout,
geom, maps,e_vec, q_val,q_der,q_det,eval_flags);
EvalKernels::Run(dim, vdim, maps.ndof, maps.nqpt, ne,vdim, q_layout,
geom, maps, e_vec, q_val, q_der, q_det, eval_flags);
}
}
+6 -5
View File
@@ -128,7 +128,7 @@ void ElementRestriction::Mult(const Vector& x, Vector& y) const
});
}
void ElementRestriction::MultUnsigned(const Vector& x, Vector& y) const
void ElementRestriction::AbsMult(const Vector& x, Vector& y) const
{
// Assumes all elements have the same number of dofs
const int nd = dof;
@@ -193,7 +193,7 @@ void ElementRestriction::AddMultTranspose(const Vector& x, Vector& y,
TAddMultTranspose<ADD>(x, y);
}
void ElementRestriction::MultTransposeUnsigned(const Vector& x, Vector& y) const
void ElementRestriction::AbsMultTranspose(const Vector& x, Vector& y) const
{
// Assumes all elements have the same number of dofs
const int nd = dof;
@@ -653,7 +653,8 @@ ConformingFaceRestriction::ConformingFaceRestriction(
: ConformingFaceRestriction(fes, f_ordering, type, true)
{ }
void ConformingFaceRestriction::Mult(const Vector& x, Vector& y) const
void ConformingFaceRestriction::MultInternal(const Vector& x, Vector& y,
const bool useAbs) const
{
if (nf==0) { return; }
// Assumes all elements have the same number of dofs
@@ -666,7 +667,7 @@ void ConformingFaceRestriction::Mult(const Vector& x, Vector& y) const
mfem::forall(nfdofs, [=] MFEM_HOST_DEVICE (int i)
{
const int s_idx = d_indices[i];
const int sgn = (s_idx >= 0) ? 1 : -1;
const int sgn = (useAbs || s_idx >= 0) ? 1 : -1;
const int idx = (s_idx >= 0) ? s_idx : -1 - s_idx;
const int dof = i % nface_dofs;
const int face = i / nface_dofs;
@@ -724,7 +725,7 @@ void ConformingFaceRestriction::AddMultTranspose(
true, a);
}
void ConformingFaceRestriction::AddMultTransposeUnsigned(
void ConformingFaceRestriction::AddAbsMultTranspose(
const Vector& x, Vector& y, const real_t a) const
{
ConformingFaceRestriction_AddMultTranspose(
+55 -7
View File
@@ -59,9 +59,18 @@ public:
const real_t a = 1.0) const override;
/// Compute Mult without applying signs based on DOF orientations.
void MultUnsigned(const Vector &x, Vector &y) const;
void AbsMult(const Vector &x, Vector &y) const override;
/// Compute MultTranspose without applying signs based on DOF orientations.
void MultTransposeUnsigned(const Vector &x, Vector &y) const;
void AbsMultTranspose(const Vector &x, Vector &y) const override;
/// @deprecated Use AbsMult() instead.
MFEM_DEPRECATED void MultUnsigned(const Vector &x, Vector &y) const
{ AbsMult(x, y); }
/// @deprecated Use AbsMultTranspose() instead.
MFEM_DEPRECATED void MultTransposeUnsigned(const Vector &x, Vector &y) const
{ AbsMultTranspose(x, y); }
/// Compute MultTranspose by setting (rather than adding) element
/// contributions; this is a left inverse of the Mult() operation
@@ -184,12 +193,19 @@ public:
/** @brief Add the face degrees of freedom @a x to the element degrees of
freedom @a y ignoring the signs from DOF orientation. */
virtual void AddMultTransposeUnsigned(const Vector &x, Vector &y,
const real_t a = 1.0) const
virtual void AddAbsMultTranspose(const Vector &x, Vector &y,
const real_t a = 1.0) const
{
AddMultTranspose(x, y, a);
}
/// @deprecated Use AddAbsMultTranspose() instead.
MFEM_DEPRECATED void AddMultTransposeUnsigned(const Vector &x, Vector &y,
const real_t a = 1.0) const
{
AddAbsMultTranspose(x, y, a);
}
/** @brief Add the face degrees of freedom @a x to the element degrees of
freedom @a y. Perform the same computation as AddMultTranspose, but
@a x is invalid after calling this method.
@@ -219,6 +235,12 @@ public:
AddMultTranspose(x, y);
}
void AbsMultTranspose(const Vector &x, Vector &y) const override
{
y = 0.0;
AddAbsMultTranspose(x, y);
}
/** @brief For each face, sets @a y to the partial derivative of @a x with
respect to the reference coordinate whose direction is
perpendicular to the face on the reference element.
@@ -319,7 +341,16 @@ public:
requested by @a type in the constructor.
The face_dofs are ordered according to the given
ElementDofOrdering. */
void Mult(const Vector &x, Vector &y) const override;
void Mult(const Vector &x, Vector &y) const override
{ MultInternal(x, y); }
/// Compute Mult without applying signs based on DOF orientations.
void AbsMult(const Vector &x, Vector &y) const override
{ MultInternal(x, y, true); }
/// @deprecated Use AbsMult() instead.
MFEM_DEPRECATED void MultUnsigned(const Vector &x, Vector &y) const
{ AbsMult(x, y); }
using FaceRestriction::AddMultTransposeInPlace;
@@ -341,8 +372,20 @@ public:
L-Vector @b not taking into account signs from DOF orientations.
@sa AddMultTranspose(). */
void AddMultTransposeUnsigned(const Vector &x, Vector &y,
const real_t a = 1.0) const override;
void AddAbsMultTranspose(const Vector &x, Vector &y,
const real_t a = 1.0) const override;
/// @deprecated Use AddAbsMultTranspose() instead.
MFEM_DEPRECATED void AddMultTransposeUnsigned(const Vector &x, Vector &y) const
{
AddAbsMultTranspose(x, y);
}
void AbsMultTranspose(const Vector &x, Vector &y) const override
{
y = 0.0;
AddAbsMultTranspose(x, y);
}
private:
/** @brief Compute the scatter indices: L-vector to E-vector, and the offsets
@@ -395,6 +438,11 @@ protected:
void SetFaceDofsGatherIndices(const Mesh::FaceInformation &face,
const int face_index,
const ElementDofOrdering f_ordering);
public:
// This method needs to be public due to 'nvcc' restriction.
void MultInternal(const Vector &x, Vector &y,
const bool useAbs = false) const;
};
/// @brief Alias for ConformingFaceRestriction, for backwards compatibility and
+9 -27
View File
@@ -5122,33 +5122,32 @@ real_t TMOP_Integrator::GetSurfaceFittingWeight()
void TMOP_Integrator::EnableNormalization(const GridFunction &x)
{
ComputeNormalizationEnergies(x, metric_normal, lim_normal, surf_fit_normal);
ComputeNormalizationEnergies(x, metric_normal, lim_normal);
metric_normal = 1.0 / metric_normal;
lim_normal = 1.0 / lim_normal;
//if (surf_fit_gf) { surf_fit_normal = 1.0 / surf_fit_normal; }
if (surf_fit_gf || surf_fit_pos) { surf_fit_normal = lim_normal; }
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::ParEnableNormalization(const ParGridFunction &x)
{
real_t loc[3];
ComputeNormalizationEnergies(x, loc[0], loc[1], loc[2]);
real_t rdc[3];
MPI_Allreduce(loc, rdc, 3, MPITypeMap<real_t>::mpi_type, MPI_SUM,
real_t loc[2];
ComputeNormalizationEnergies(x, loc[0], loc[1]);
real_t rdc[2];
MPI_Allreduce(loc, rdc, 2, MPITypeMap<real_t>::mpi_type, MPI_SUM,
x.ParFESpace()->GetComm());
metric_normal = 1.0 / rdc[0];
lim_normal = 1.0 / rdc[1];
// if (surf_fit_gf) { surf_fit_normal = 1.0 / rdc[2]; }
if (surf_fit_gf || surf_fit_pos) { surf_fit_normal = lim_normal; }
}
#endif
void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy,
real_t &lim_energy,
real_t &surf_fit_gf_energy)
real_t &lim_energy)
{
metric_energy = 0.0;
lim_energy = 0.0;
if (PA.enabled)
{
MFEM_VERIFY(PA.E.Size() > 0, "Must be called after AssemblePA!");
@@ -5191,9 +5190,6 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
Jpr.SetSize(dim);
Jpt.SetSize(dim);
metric_energy = 0.0;
lim_energy = 0.0;
surf_fit_gf_energy = 0.0;
for (int i = 0; i < fes->GetNE(); i++)
{
const FiniteElement *fe = fes->GetFE(i);
@@ -5225,21 +5221,7 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
lim_energy += weight;
}
// Normalization of the surface fitting term.
if (surf_fit_gf)
{
Array<int> dofs;
Vector sigma_e;
surf_fit_gf->FESpace()->GetElementDofs(i, dofs);
surf_fit_gf->GetSubVector(dofs, sigma_e);
for (int s = 0; s < dofs.Size(); s++)
{
if ((*surf_fit_marker)[dofs[s]] == true)
{
surf_fit_gf_energy += sigma_e(s) * sigma_e(s);
}
}
}
// TODO: Normalization of the surface fitting term.
}
// Cases when integration is not over the target element, or when the
+1 -2
View File
@@ -2038,8 +2038,7 @@ protected:
} PA;
void ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy, real_t &lim_energy,
real_t &surf_fit_gf_energy);
real_t &metric_energy, real_t &lim_energy);
void AssembleElementVectorExact(const FiniteElement &el,
ElementTransformation &T,
+1
View File
@@ -39,6 +39,7 @@ list(APPEND HDRS
arrays_by_name.hpp
backends.hpp
binaryio.hpp
complex_type.hpp
cuda.hpp
device.hpp
error.hpp
+14
View File
@@ -15,6 +15,7 @@
#include "array.hpp"
#include "../general/forall.hpp"
#include <fstream>
#include <type_traits>
namespace mfem
{
@@ -110,6 +111,19 @@ void Array<T>::PartialSum()
}
}
template <class T>
void Array<T>::Abs()
{
static_assert(std::is_arithmetic<T>::value, "Use with arithmetic types!");
const bool useDevice = UseDevice();
const int N = size;
auto y = ReadWrite(useDevice);
mfem::forall_switch(useDevice, N, [=] MFEM_HOST_DEVICE (int i)
{
y[i] = std::abs(y[i]);
});
}
// Sum
template <class T>
T Array<T>::Sum() const
+8 -1
View File
@@ -305,6 +305,9 @@ public:
/// Fill the entries of the array with the cumulative sum of the entries.
void PartialSum();
/// Replace each entry of the array with its absolute value.
void Abs();
/// Return the sum of all the array entries using the '+'' operator for class 'T'.
T Sum() const;
@@ -323,7 +326,11 @@ public:
the Size to match this Capacity after this.*/
template <typename U>
inline void CopyFrom(const U *src)
{ std::memcpy(begin(), src, MemoryUsage()); }
{
if (!begin() || size == 0) { return; }
MFEM_ASSERT(begin() && src, "Error in Array::CopyFrom");
std::memcpy(begin(), src, MemoryUsage());
}
/// STL-like begin. Returns pointer to the first element of the array.
inline T* begin() { return data; }
+1
View File
@@ -62,6 +62,7 @@
#define MFEM_THREAD_ID(k) 0
#define MFEM_THREAD_SIZE(k) 1
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) MFEM_FOREACH_THREAD(i,k,N)
#endif
// 'double' and 'float' atomicAdd implementation for previous versions of CUDA
+125
View File
@@ -0,0 +1,125 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_COMPLEX_TYPE
#define MFEM_COMPLEX_TYPE
#include "../config/config.hpp"
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
#include <complex>
#include <utility>
#endif
#if defined(MFEM_USE_CUDA)
#include <cuComplex.h>
#endif
#if defined(MFEM_USE_HIP)
#include <hip/hip_complex.h>
#endif
namespace mfem
{
/// @brief Complex number type for device.
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
#define zAbs std::abs
#define zExp std::exp
#define zNorm std::norm
using complex_t = std::complex<real_t>;
#else // CUDA or HIP
#if defined(MFEM_USE_CUDA)
using DoubleComplex_t = cuDoubleComplex;
#endif
#if defined(MFEM_USE_HIP)
using DoubleComplex_t = hipDoubleComplex;
#endif
struct Complex : public DoubleComplex_t
{
MFEM_HOST_DEVICE Complex() = default;
MFEM_HOST_DEVICE Complex(real_t r) { x = r, y = 0.0; }
MFEM_HOST_DEVICE Complex(real_t r, real_t i) { x = r, y = i; }
MFEM_HOST_DEVICE real_t real() const { return x; }
MFEM_HOST_DEVICE void real(real_t r) { x = r; }
MFEM_HOST_DEVICE real_t imag() const { return y; }
MFEM_HOST_DEVICE void imag(real_t i) { y = i; }
template <typename U>
MFEM_HOST_DEVICE inline Complex &operator*=(const U &z)
{
return *this = *this * z, *this;
}
template <typename U>
MFEM_HOST_DEVICE inline Complex &operator/=(const U &z)
{
return *this = *this / z, *this;
}
};
MFEM_HOST_DEVICE inline Complex operator*(const Complex &x, const real_t &y)
{
return Complex(x.real() * y, x.imag() * y);
}
MFEM_HOST_DEVICE inline Complex operator+(const Complex &a, const Complex &b)
{
return Complex(a.real() + b.real(), a.imag() + b.imag());
}
MFEM_HOST_DEVICE inline Complex operator*(const real_t d, const Complex &z)
{
return Complex(z.real() * d, z.imag() * d);
}
MFEM_HOST_DEVICE inline Complex operator*(const Complex &a, const Complex &b)
{
return Complex(a.real() * b.real() - a.imag() * b.imag(),
a.real() * b.imag() + a.imag() * b.real());
}
MFEM_HOST_DEVICE inline Complex operator/(const Complex &z, const real_t &d)
{
return Complex(z.real() / d, z.imag() / d);
}
MFEM_HOST_DEVICE inline real_t zAbs(const Complex &z)
{
return std::hypot(z.real(), z.imag());
}
MFEM_HOST_DEVICE inline Complex zExp(const Complex &q)
{
Complex z;
real_t s, c, e = std::exp(q.real());
sincos(q.imag(), &s, &c);
z.real(c * e), z.imag(s * e);
return z;
}
MFEM_HOST_DEVICE inline real_t zNorm(const Complex &z)
{
return z.real() * z.real() + z.imag() * z.imag();
}
using complex_t = Complex;
#endif // MFEM_USE_CUDA || MFEM_USE_HIP
} // namespace mfem
#endif // MFEM_COMPLEX_TYPE
+1
View File
@@ -47,6 +47,7 @@
#define MFEM_THREAD_ID(k) threadIdx.k
#define MFEM_THREAD_SIZE(k) blockDim.k
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=threadIdx.k; i<N; i+=blockDim.k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) if(const int i=threadIdx.k; i<N)
#endif
namespace mfem
+36
View File
@@ -16,6 +16,7 @@
#include "../fem/ceed/interface/util.hpp"
#endif
#ifdef MFEM_USE_MPI
#include "communication.hpp"
#include "../linalg/hypre.hpp"
#endif
@@ -145,6 +146,11 @@ Device::Device()
Configure(device);
device_env = true;
}
if (GetEnv("MFEM_GPU_AWARE_MPI"))
{
SetGPUAwareMPI(true);
}
}
Device::~Device()
@@ -196,6 +202,29 @@ void Device::Configure(const std::string &device, const int device_id)
{
bmap[internal::backend_name[i]] = internal::backend_list[i];
}
// auto-detect GPU configurations
// assumes only one of HIP or CUDA are available
#ifdef MFEM_USE_HIP
bmap["gpu"] = Backend::HIP;
#ifdef MFEM_USE_RAJA
bmap["raja-gpu"] = Backend::RAJA_HIP;
#endif
#ifdef MFEM_USE_CEED
bmap["ceed-gpu"] = Backend::CEED_HIP;
#endif
// no OCCA+HIP?
#elif defined(MFEM_USE_CUDA)
bmap["gpu"] = Backend::CUDA;
#ifdef MFEM_USE_RAJA
bmap["raja-gpu"] = Backend::RAJA_CUDA;
#endif
#ifdef MFEM_USE_CEED
bmap["ceed-gpu"] = Backend::CEED_CUDA;
#endif
#ifdef MFEM_USE_OCCA
bmap["occa-gpu"] = Backend::OCCA_CUDA;
#endif
#endif
std::string device_option;
std::string::size_type beg = 0, end;
while (1)
@@ -313,6 +342,13 @@ void Device::Print(std::ostream &os)
{
os << ',' << MemoryTypeName[static_cast<int>(device_mem_type)];
}
#ifdef MFEM_USE_MPI
if (Allows(Backend::DEVICE_MASK) &&
Mpi::IsInitialized() && !Mpi::IsFinalized())
{
os << "\nUse GPU-aware MPI: " << (GetGPUAwareMPI() ? "yes" : "no");
}
#endif
os << std::endl;
}
+4
View File
@@ -198,6 +198,10 @@ public:
'ceed-hip', 'hip', 'debug',
'occa-omp', 'raja-omp', 'omp',
'ceed-cpu', 'occa-cpu', 'raja-cpu', 'cpu'.
- The following backend aliases are also available: 'ceed-gpu',
'occa-gpu', 'raja-gpu', and 'gpu' where they alias their respective
'*-cuda' or '*-hip' backends depending on the MFEM build-time
configuration.
- Multiple backends can be configured at the same time.
- Only one 'occa-*' backend can be configured at a time.
- The backend 'occa-cuda' enables the 'cuda' backend unless 'raja-cuda'
-156
View File
@@ -23,9 +23,6 @@
#include <_hypre_utilities.h>
#endif
#include "array.hpp"
#include "reducers.hpp"
namespace mfem
{
@@ -853,159 +850,6 @@ inline MemoryClass GetHypreForallMemoryClass()
#endif // MFEM_USE_MPI
namespace internal
{
/**
@brief Device portion of a reduction over a 1D sequence [0, N)
@tparam B Reduction body. Must be callable with the signature void(int i, value_type&
v), where i is the index to evaluate and v is the value to update.
@tparam R Reducer capable of combining values of type value_type. See reducers.hpp for
pre-defined reducers.
*/
template<class B, class R> struct reduction_kernel
{
/// value type body and reducer operate on.
using value_type = typename R::value_type;
/// workspace for the intermediate reduction results
mutable value_type *work;
B body;
R reducer;
/// Length of sequence to reduce over.
int N;
/// How many items is each thread responsible for during the serial phase
int items_per_thread;
constexpr static MFEM_HOST_DEVICE int max_blocksize() { return 256; }
/// helper for computing the reduction block size
static int block_log2(unsigned N)
{
#if defined(__GNUC__) || defined(__clang__)
return N ? (sizeof(unsigned) * 8 - __builtin_clz(N)) : 0;
#elif defined(_MSC_VER)
return sizeof(unsigned) * 8 - __lzclz(N);
#else
int res = 0;
while (N)
{
N >>= 1;
++res;
}
return res;
#endif
}
MFEM_HOST_DEVICE void operator()(int work_idx) const
{
MFEM_SHARED value_type buffer[max_blocksize()];
reducer.SetInitialValue(buffer[MFEM_THREAD_ID(x)]);
// serial part
for (int idx = 0; idx < items_per_thread; ++idx)
{
int i = MFEM_THREAD_ID(x) +
(idx + work_idx * items_per_thread) * MFEM_THREAD_SIZE(x);
if (i < N)
{
body(i, buffer[MFEM_THREAD_ID(x)]);
}
else
{
break;
}
}
// binary tree reduction
for (int i = (MFEM_THREAD_SIZE(x) >> 1); i > 0; i >>= 1)
{
MFEM_SYNC_THREAD;
if (MFEM_THREAD_ID(x) < i)
{
reducer.Join(buffer[MFEM_THREAD_ID(x)], buffer[MFEM_THREAD_ID(x) + i]);
}
}
if (MFEM_THREAD_ID(x) == 0)
{
work[work_idx] = buffer[0];
}
}
};
}
/**
@brief Performs a 1D reduction on the range [0,N).
@a res initial value and where the result will be written.
@a body reduction function body.
@a reducer helper for joining two reduced values.
@a use_dev true to perform the reduction on the device, if possible.
@a workspace temporary workspace used for device reductions. May be resized to
a larger capacity as needed. Preferably should have MemoryType::MANAGED or
MemoryType::HOST_PINNED. TODO: replace with internal temporary workspace
vectors once that's added to the memory manager.
@tparam T value_type to operate on
*/
template <class T, class B, class R>
void reduce(int N, T &res, B &&body, const R &reducer, bool use_dev,
Array<T> &workspace)
{
if (N == 0)
{
return;
}
#if defined(MFEM_USE_HIP) || defined(MFEM_USE_CUDA)
if (use_dev &&
mfem::Device::Allows(Backend::CUDA | Backend::HIP | Backend::RAJA_CUDA |
Backend::RAJA_HIP))
{
using red_type = internal::reduction_kernel<typename std::decay<B>::type,
typename std::decay<R>::type>;
// max block size is 256, but can be smaller
int block_size = std::min<int>(red_type::max_blocksize(),
1ll << red_type::block_log2(N));
int num_mp = Device::NumMultiprocessors(Device::GetId());
#if defined(MFEM_USE_CUDA)
// good value of mp_sat found experimentally on Lassen
constexpr int mp_sat = 8;
#elif defined(MFEM_USE_HIP)
// good value of mp_sat found experimentally on Tuolumne
constexpr int mp_sat = 4;
#else
num_mp = 1;
constexpr int mp_sat = 1;
#endif
// determine how many items each thread should sum during the serial
// portion
int nblocks = std::min(mp_sat * num_mp, (N + block_size - 1) / block_size);
int items_per_thread =
(N + block_size * nblocks - 1) / (block_size * nblocks);
red_type red{nullptr, std::forward<B>(body), reducer, N, items_per_thread};
// allocate res to fit block_size entries
auto mt = workspace.GetMemory().GetMemoryType();
if (mt != MemoryType::HOST_PINNED && mt != MemoryType::MANAGED)
{
mt = MemoryType::HOST_PINNED;
}
workspace.SetSize(nblocks, mt);
auto work = workspace.HostWrite();
red.work = work;
forall_2D(nblocks, block_size, 1, std::move(red));
// wait for results
MFEM_DEVICE_SYNC;
for (int i = 0; i < nblocks; ++i)
{
reducer.Join(res, work[i]);
}
return;
}
#endif
for (int i = 0; i < N; ++i)
{
body(i, res);
}
}
} // namespace mfem
#endif // MFEM_FORALL_HPP
+3 -1
View File
@@ -47,7 +47,9 @@
#define MFEM_THREAD_ID(k) hipThreadIdx_ ##k
#define MFEM_THREAD_SIZE(k) hipBlockDim_ ##k
#define MFEM_FOREACH_THREAD(i,k,N) \
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) \
if(const int i=hipThreadIdx_ ##k; i<N)
#endif
namespace mfem
+1 -1
View File
@@ -641,7 +641,7 @@ public:
UmpireMemorySpace(name, "DEVICE") {}
void Alloc(Memory &base) override
{ base.d_ptr = allocator.allocate(base.bytes); }
void Dealloc(Memory &base) override { rm.deallocate(base.d_ptr); }
void Dealloc(Memory &base) override { allocator.deallocate(base.d_ptr); }
void *HtoD(void *dst, const void *src, size_t bytes) override
{
#ifdef MFEM_USE_CUDA
+156 -3
View File
@@ -12,11 +12,10 @@
#ifndef MFEM_REDUCERS_HPP
#define MFEM_REDUCERS_HPP
#include "array.hpp"
#include "forall.hpp"
#include <climits>
#include <cmath>
#include <cstdint>
#include <limits>
#include <type_traits>
@@ -439,6 +438,160 @@ template <class I> struct ArgMinMaxReducer<double, I>
}
};
namespace internal
{
/**
@brief Device portion of a reduction over a 1D sequence [0, N)
@tparam B Reduction body. Must be callable with the signature void(int i, value_type&
v), where i is the index to evaluate and v is the value to update.
@tparam R Reducer capable of combining values of type value_type. See reducers.hpp for
pre-defined reducers.
*/
template<class B, class R> struct reduction_kernel
{
/// value type body and reducer operate on.
using value_type = typename R::value_type;
/// workspace for the intermediate reduction results
mutable value_type *work;
B body;
R reducer;
/// Length of sequence to reduce over.
int N;
/// How many items is each thread responsible for during the serial phase
int items_per_thread;
constexpr static MFEM_HOST_DEVICE int max_blocksize() { return 256; }
/// helper for computing the reduction block size
static int block_log2(unsigned N)
{
#if defined(__GNUC__) or defined(__clang__)
return N ? (sizeof(unsigned) * 8 - __builtin_clz(N)) : 0;
#elif defined(_MSC_VER)
return sizeof(unsigned) * 8 - __lzclz(N);
#else
int res = 0;
while (N)
{
N >>= 1;
++res;
}
return res;
#endif
}
MFEM_HOST_DEVICE void operator()(int work_idx) const
{
MFEM_SHARED value_type buffer[max_blocksize()];
reducer.SetInitialValue(buffer[MFEM_THREAD_ID(x)]);
// serial part
for (int idx = 0; idx < items_per_thread; ++idx)
{
int i = MFEM_THREAD_ID(x) +
(idx + work_idx * items_per_thread) * MFEM_THREAD_SIZE(x);
if (i < N)
{
body(i, buffer[MFEM_THREAD_ID(x)]);
}
else
{
break;
}
}
// binary tree reduction
for (int i = (MFEM_THREAD_SIZE(x) >> 1); i > 0; i >>= 1)
{
MFEM_SYNC_THREAD;
if (MFEM_THREAD_ID(x) < i)
{
reducer.Join(buffer[MFEM_THREAD_ID(x)], buffer[MFEM_THREAD_ID(x) + i]);
}
}
if (MFEM_THREAD_ID(x) == 0)
{
work[work_idx] = buffer[0];
}
}
};
}
/**
@brief Performs a 1D reduction on the range [0,N).
@a res initial value and where the result will be written.
@a body reduction function body.
@a reducer helper for joining two reduced values.
@a use_dev true to perform the reduction on the device, if possible.
@a workspace temporary workspace used for device reductions. May be resized to
a larger capacity as needed. Preferably should have MemoryType::MANAGED or
MemoryType::HOST_PINNED. TODO: replace with internal temporary workspace
vectors once that's added to the memory manager.
@tparam T value_type to operate on
*/
template <class T, class B, class R>
void reduce(int N, T &res, B &&body, const R &reducer, bool use_dev,
Array<T> &workspace)
{
if (N == 0)
{
return;
}
#if defined(MFEM_USE_HIP) || defined(MFEM_USE_CUDA)
if (use_dev &&
mfem::Device::Allows(Backend::CUDA | Backend::HIP | Backend::RAJA_CUDA |
Backend::RAJA_HIP))
{
using red_type = internal::reduction_kernel<typename std::decay<B>::type,
typename std::decay<R>::type>;
// max block size is 256, but can be smaller
int block_size = std::min<int>(red_type::max_blocksize(),
1ll << red_type::block_log2(N));
int num_mp = Device::NumMultiprocessors(Device::GetId());
#if defined(MFEM_USE_CUDA)
// good value of mp_sat found experimentally on Lassen
constexpr int mp_sat = 8;
#elif defined(MFEM_USE_HIP)
// good value of mp_sat found experimentally on Tuolumne
constexpr int mp_sat = 4;
#else
num_mp = 1;
constexpr int mp_sat = 1;
#endif
// determine how many items each thread should sum during the serial
// portion
int nblocks = std::min(mp_sat * num_mp, (N + block_size - 1) / block_size);
int items_per_thread =
(N + block_size * nblocks - 1) / (block_size * nblocks);
red_type red{nullptr, std::forward<B>(body), reducer, N, items_per_thread};
// allocate res to fit block_size entries
auto mt = workspace.GetMemory().GetMemoryType();
if (mt != MemoryType::HOST_PINNED && mt != MemoryType::MANAGED)
{
mt = MemoryType::HOST_PINNED;
}
workspace.SetSize(nblocks, mt);
auto work = workspace.HostWrite();
red.work = work;
forall_2D(nblocks, block_size, 1, std::move(red));
// wait for results
MFEM_DEVICE_SYNC;
for (int i = 0; i < nblocks; ++i)
{
reducer.Join(res, work[i]);
}
return;
}
#endif
for (int i = 0; i < N; ++i)
{
body(i, res);
}
}
} // namespace mfem
#endif
#endif // MFEM_REDUCERS_HPP
+2
View File
@@ -21,6 +21,7 @@ list(APPEND SRCS
blockvector.cpp
complex_densemat.cpp
complex_operator.cpp
complex_vector.cpp
constraints.cpp
densemat.cpp
symmat.cpp
@@ -47,6 +48,7 @@ list(APPEND HDRS
blockvector.hpp
complex_densemat.hpp
complex_operator.hpp
complex_vector.hpp
constraints.hpp
densemat.hpp
dinvariants.hpp
+302
View File
@@ -9,6 +9,7 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../general/forall.hpp"
#include "complex_densemat.hpp"
#include "lapack.hpp"
#include <complex>
@@ -16,6 +17,8 @@
namespace mfem
{
using namespace std;
DenseMatrix & ComplexDenseMatrix::real()
{
MFEM_ASSERT(Op_Real_, "ComplexDenseMatrix has no real part!");
@@ -1017,4 +1020,303 @@ void ComplexCholeskyFactors::GetInverseMatrix(int m, real_t * X_r,
delete [] X;
}
ComplexTypeDenseMatrix::ComplexTypeDenseMatrix()
: height(0), width(0)
{}
ComplexTypeDenseMatrix::ComplexTypeDenseMatrix(const ComplexTypeDenseMatrix &m)
: height(m.Height()), width(m.Width())
{
const int hw = height * width;
if (hw > 0)
{
MFEM_ASSERT(m.data, "invalid source matrix");
data.New(hw);
std::memcpy(data, m.data, sizeof(complex_t)*hw);
}
}
ComplexTypeDenseMatrix::ComplexTypeDenseMatrix(const DenseMatrix &m)
: height(m.Height()), width(m.Width())
{
const int hw = height * width;
if (hw > 0)
{
MFEM_ASSERT(m.data, "invalid source matrix");
data.New(hw);
for (int i = 0; i < hw; i++)
{
data[i] = m.data[i];
}
}
}
ComplexTypeDenseMatrix::ComplexTypeDenseMatrix(int s)
: height(s), width(s)
{
MFEM_ASSERT(s >= 0, "invalid DenseMatrix size: " << s);
if (s > 0)
{
data.New(s*s);
*this = 0.0; // init with zeroes
}
}
ComplexTypeDenseMatrix::ComplexTypeDenseMatrix(int m, int n)
: height(m), width(n)
{
MFEM_ASSERT(m >= 0 && n >= 0,
"invalid DenseMatrix size: " << m << " x " << n);
const int capacity = m*n;
if (capacity > 0)
{
data.New(capacity);
*this = 0.0; // init with zeroes
}
}
void ComplexTypeDenseMatrix::SetSize(int h, int w)
{
MFEM_ASSERT(h >= 0 && w >= 0,
"invalid ComplexTypeDenseMatrix size: " << h << " x " << w);
if (Height() == h && Width() == w)
{
return;
}
height = h;
width = w;
const int hw = h*w;
if (hw > data.Capacity())
{
data.Delete();
data.New(hw);
*this = 0.0; // init with zeroes
}
}
/// Returns reference to a_{ij}.
complex_t &ComplexTypeDenseMatrix::Elem(int i, int j)
{
return (*this)(i,j);
}
/// Returns constant reference to a_{ij}.
const complex_t &ComplexTypeDenseMatrix::Elem(int i, int j) const
{
return (*this)(i,j);
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=(real_t c)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] = c;
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=(complex_t c)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] = c;
}
return *this;
}
/// Copy the matrix entries from the given array
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=(const real_t *d)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] = d[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=
(const complex_t *d)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] = d[i];
}
return *this;
}
/// Sets the matrix size and elements equal to those of m
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=(const DenseMatrix &m)
{
SetSize(m.height, m.width);
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] = m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator=
(const ComplexTypeDenseMatrix &m)
{
SetSize(m.height, m.width);
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] = m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator+=(const real_t *m)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] += m[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator+=
(const complex_t *m)
{
const int s = Height()*Width();
for (int i = 0; i < s; i++)
{
data[i] += m[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator+=(const DenseMatrix &m)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] += m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator+=
(const ComplexTypeDenseMatrix &m)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] += m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator-=(const DenseMatrix &m)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] -= m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator-=
(const ComplexTypeDenseMatrix &m)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] -= m.data[i];
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator*=(real_t c)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] *= c;
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::operator*=(complex_t c)
{
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] *= c;
}
return *this;
}
ComplexTypeDenseMatrix &ComplexTypeDenseMatrix::Set(const DenseMatrix &Mr,
const DenseMatrix &Mi)
{
MFEM_ASSERT(height == Mr.Height() && height == Mi.Height() &&
width == Mr.Width() && width == Mi.Width(),
"incompatible Matrices!");
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
data[i] = complex_t(Mr.data[i], Mi.data[i]);
}
return *this;
}
void ComplexTypeDenseMatrix::Swap(ComplexTypeDenseMatrix &other)
{
mfem::Swap(width, other.width);
mfem::Swap(height, other.height);
mfem::Swap(data, other.data);
}
ComplexTypeDenseMatrix::~ComplexTypeDenseMatrix()
{
data.Delete();
}
const DenseMatrix &ComplexTypeDenseMatrix::real() const
{
re_part.SetSize(height, width);
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
re_part.data[i] = data[i].real();
}
return re_part;
}
const DenseMatrix &ComplexTypeDenseMatrix::imag() const
{
im_part.SetSize(height, width);
const int hw = height * width;
for (int i = 0; i < hw; i++)
{
im_part.data[i] = data[i].imag();
}
return im_part;
}
} // mfem namespace
+215
View File
@@ -13,6 +13,7 @@
#define MFEM_COMPLEX_DENSEMAT
#include "complex_operator.hpp"
#include "../general/complex_type.hpp"
#include <complex>
namespace mfem
@@ -241,6 +242,220 @@ public:
};
class ComplexTypeDenseMatrix
{
protected:
int height; ///< Dimension of the output / number of rows in the matrix.
int width; ///< Dimension of the input / number of columns in the matrix.
private:
Memory<complex_t > data;
mutable DenseMatrix re_part;
mutable DenseMatrix im_part;
public:
/** Default constructor for DenseMatrix.
Sets data = NULL and height = width = 0. */
ComplexTypeDenseMatrix();
/// Copy constructor
ComplexTypeDenseMatrix(const ComplexTypeDenseMatrix &);
ComplexTypeDenseMatrix(const DenseMatrix &);
/// Creates square matrix of size s.
explicit ComplexTypeDenseMatrix(int s);
/// Creates rectangular matrix of size m x n.
ComplexTypeDenseMatrix(int m, int n);
/// Construct a ComplexTypeDenseMatrix using an existing data array.
/** The ComplexTypeDenseMatrix does not assume ownership of the data array,
i.e. it will not delete the array. */
ComplexTypeDenseMatrix(complex_t *d, int h, int w)
: height(h), width(w) { UseExternalData(d, h, w); }
/// Create a dense matrix using a braced initializer list
/// The inner lists correspond to rows of the matrix
template <int M, int N, typename T = real_t>
explicit ComplexTypeDenseMatrix(const T (&values)[M][N]) :
ComplexTypeDenseMatrix(
M, N)
{
// DenseMatrix is column-major so copies have to be element-wise
for (int i = 0; i < M; i++)
{
for (int j = 0; j < N; j++)
{
(*this)(i,j) = values[i][j];
}
}
}
/// Change the data array and the size of the DenseMatrix.
/** The DenseMatrix does not assume ownership of the data array, i.e. it will
not delete the data array @a d. This method should not be used with
DenseMatrix that owns its current data array. */
void UseExternalData(complex_t *d, int h, int w)
{
data.Wrap(d, h*w, false);
height = h; width = w;
}
/// Change the data array and the size of the DenseMatrix.
/** The DenseMatrix does not assume ownership of the data array, i.e. it will
not delete the new array @a d. This method will delete the current data
array, if owned. */
void Reset(complex_t *d, int h, int w)
{ if (OwnsData()) { data.Delete(); } UseExternalData(d, h, w); }
/** Clear the data array and the dimensions of the DenseMatrix. This method
should not be used with DenseMatrix that owns its current data array. */
void ClearExternalData() { data.Reset(); height = width = 0; }
/// Delete the matrix data array (if owned) and reset the matrix state.
void Clear()
{ if (OwnsData()) { data.Delete(); } ClearExternalData(); }
/// Get the height (size of output) of the Operator. Synonym with NumRows().
inline int Height() const { return height; }
/** @brief Get the number of rows (size of output) of the Operator. Synonym
with Height(). */
inline int NumRows() const { return height; }
/// Get the width (size of input) of the Operator. Synonym with NumCols().
inline int Width() const { return width; }
/** @brief Get the number of columns (size of input) of the Operator. Synonym
with Width(). */
inline int NumCols() const { return width; }
/// For backward compatibility define Size to be synonym of Width()
int Size() const { return Width(); }
// Total size = width*height
int TotalSize() const { return width*height; }
/// Change the size of the DenseMatrix to s x s.
void SetSize(int s) { SetSize(s, s); }
/// Change the size of the DenseMatrix to h x w.
void SetSize(int h, int w);
/// Returns the matrix data array.
inline complex_t *Data() const
{
return const_cast<complex_t*>
((const complex_t*)data);
}
/// Returns the matrix data array.
inline complex_t *GetData() const { return Data(); }
Memory<complex_t > &GetMemory() { return data; }
const Memory<complex_t > &GetMemory() const { return data; }
/// Return the DenseMatrix data (host pointer) ownership flag.
inline bool OwnsData() const { return data.OwnsHostPtr(); }
/// Returns reference to a_{ij}.
inline complex_t &operator()(int i, int j);
/// Returns constant reference to a_{ij}.
inline const complex_t &operator()(int i, int j) const;
/// Returns reference to a_{ij}.
complex_t &Elem(int i, int j);
/// Returns constant reference to a_{ij}.
const complex_t &Elem(int i, int j) const;
/// Sets the matrix elements equal to constant c
ComplexTypeDenseMatrix &operator=(real_t c);
ComplexTypeDenseMatrix &operator=(complex_t c);
/// Copy the matrix entries from the given array
ComplexTypeDenseMatrix &operator=(const real_t *d);
ComplexTypeDenseMatrix &operator=(const complex_t *d);
/// Sets the matrix size and elements equal to those of m
ComplexTypeDenseMatrix &operator=(const DenseMatrix &m);
ComplexTypeDenseMatrix &operator=(const ComplexTypeDenseMatrix &m);
ComplexTypeDenseMatrix &operator+=(const real_t *m);
ComplexTypeDenseMatrix &operator+=(const complex_t *m);
ComplexTypeDenseMatrix &operator+=(const DenseMatrix &m);
ComplexTypeDenseMatrix &operator+=(const ComplexTypeDenseMatrix &m);
ComplexTypeDenseMatrix &operator-=(const DenseMatrix &m);
ComplexTypeDenseMatrix &operator-=(const ComplexTypeDenseMatrix &m);
ComplexTypeDenseMatrix &operator*=(real_t c);
ComplexTypeDenseMatrix &operator*=(complex_t c);
/// (*this) = x + i * y
ComplexTypeDenseMatrix &Set(const DenseMatrix &x, const DenseMatrix &y);
std::size_t MemoryUsage() const
{ return data.Capacity() * sizeof(complex_t); }
/// Shortcut for mfem::Read( GetMemory(), TotalSize(), on_dev).
const complex_t *Read(bool on_dev = true) const
{ return mfem::Read(data, Height()*Width(), on_dev); }
/// Shortcut for mfem::Read(GetMemory(), TotalSize(), false).
const complex_t *HostRead() const
{ return mfem::Read(data, Height()*Width(), false); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), on_dev).
complex_t *Write(bool on_dev = true)
{ return mfem::Write(data, Height()*Width(), on_dev); }
/// Shortcut for mfem::Write(GetMemory(), TotalSize(), false).
complex_t *HostWrite()
{ return mfem::Write(data, Height()*Width(), false); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), on_dev).
complex_t *ReadWrite(bool on_dev = true)
{ return mfem::ReadWrite(data, Height()*Width(), on_dev); }
/// Shortcut for mfem::ReadWrite(GetMemory(), TotalSize(), false).
complex_t *HostReadWrite()
{ return mfem::ReadWrite(data, Height()*Width(), false); }
void Swap(ComplexTypeDenseMatrix &other);
/// Return a reference to the real part of this matrix
const DenseMatrix &real() const;
/// Return a reference to the imaginary part of this matrix
const DenseMatrix &imag() const;
/// Destroys dense matrix.
virtual ~ComplexTypeDenseMatrix();
};
/// Specialization of the template function Swap<> for class ComplexTypeDenseMatrix
template<> inline void Swap<ComplexTypeDenseMatrix>(ComplexTypeDenseMatrix &a,
ComplexTypeDenseMatrix &b)
{
a.Swap(b);
}
// Inline methods
inline complex_t &ComplexTypeDenseMatrix::operator()(int i, int j)
{
MFEM_ASSERT(data && i >= 0 && i < height && j >= 0 && j < width, "");
return data[i+j*height];
}
inline const complex_t &ComplexTypeDenseMatrix::operator()
(int i, int j) const
{
MFEM_ASSERT(data && i >= 0 && i < height && j >= 0 && j < width, "");
return data[i+j*height];
}
} // namespace mfem
#endif // MFEM_COMPLEX_DENSEMAT
+424
View File
@@ -0,0 +1,424 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../general/forall.hpp"
#include "../general/reducers.hpp"
#include "complex_vector.hpp"
using namespace std;
namespace mfem
{
ComplexVector::ComplexVector(const ComplexVector &v)
{
const int s = v.Size();
size = s;
if (s > 0)
{
MFEM_ASSERT(!v.data.Empty(), "invalid source vector");
data.New(s, v.data.GetMemoryType());
data.CopyFrom(v.data, s);
}
UseDevice(v.UseDevice());
}
ComplexVector::ComplexVector(const Vector &v)
{
const int s = v.Size();
size = s;
if (s > 0)
{
MFEM_ASSERT(!v.data.Empty(), "invalid source vector");
data.New(s, v.data.GetMemoryType());
MFEM_FORALL(i, size, data[i] = v.data[i]; );
}
UseDevice(v.UseDevice());
}
ComplexVector::ComplexVector(ComplexVector &&v)
{
*this = std::move(v);
}
complex_t &ComplexVector::Elem(int i)
{
return operator()(i);
}
const complex_t &ComplexVector::Elem(int i) const
{
return operator()(i);
}
complex_t ComplexVector::operator*(const complex_t *v) const
{
HostRead();
complex_t dot = 0.0;
#ifdef MFEM_USE_LEGACY_OPENMP
#pragma omp parallel for reduction(+:dot)
#endif
for (int i = 0; i < size; i++)
{
dot += data[i] * v[i];
}
return dot;
}
complex_t ComplexVector::operator*(const real_t *v) const
{
HostRead();
complex_t dot = 0.0;
#ifdef MFEM_USE_LEGACY_OPENMP
#pragma omp parallel for reduction(+:dot)
#endif
for (int i = 0; i < size; i++)
{
dot += data[i] * v[i];
}
return dot;
}
complex_t ComplexVector::operator*(const ComplexVector &v) const
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
if (size == 0) { return 0.0; }
const bool use_dev = UseDevice() || v.UseDevice();
const auto m_data = Read(use_dev), v_data = v.Read(use_dev);
// The standard way of computing the dot product is non-deterministic
complex_t prod = 0.0;
for (int i = 0; i < size; i++)
{
prod += m_data[i] * v_data[i];
}
return prod;
}
complex_t ComplexVector::operator*(const Vector &v) const
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
if (size == 0) { return 0.0; }
const bool use_dev = UseDevice() || v.UseDevice();
const auto m_data = Read(use_dev);
const auto v_data = v.Read(use_dev);
// The standard way of computing the dot product is non-deterministic
complex_t prod = 0.0;
for (int i = 0; i < size; i++)
{
prod += m_data[i] * v_data[i];
}
return prod;
}
ComplexVector &ComplexVector::operator=(const complex_t *v)
{
HostRead();
MFEM_FORALL(i, size, data[i] = v[i]; );
return *this;
}
ComplexVector &ComplexVector::operator=(const real_t *v)
{
HostRead();
MFEM_FORALL(i, size, data[i] = v[i]; );
return *this;
}
ComplexVector &ComplexVector::operator=(const ComplexVector &v)
{
#if 0
SetSize(v.Size(), v.data.GetMemoryType());
data.CopyFrom(v.data, v.Size());
UseDevice(v.UseDevice());
#else
SetSize(v.Size());
const bool vuse = v.UseDevice();
const bool use_dev = UseDevice() || vuse;
v.UseDevice(use_dev);
// keep 'data' where it is, unless 'use_dev' is true
if (use_dev) { Write(); }
data.CopyFrom(v.data, v.Size());
v.UseDevice(vuse);
#endif
return *this;
}
ComplexVector &ComplexVector::operator=(const Vector &v)
{
SetSize(v.Size());
const bool vuse = v.UseDevice();
const bool use_dev = UseDevice() || vuse;
v.UseDevice(use_dev);
// keep 'data' where it is, unless 'use_dev' is true
if (use_dev) { Write(); }
MFEM_FORALL(i, size, data[i] = v[i]; );
v.UseDevice(vuse);
return *this;
}
ComplexVector &ComplexVector::operator=(ComplexVector &&v)
{
v.Swap(*this);
if (this != &v) { v.Destroy(); }
return *this;
}
ComplexVector &ComplexVector::operator=(complex_t value)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] = value; });
return *this;
}
ComplexVector &ComplexVector::operator=(real_t value)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] = value; });
return *this;
}
ComplexVector &ComplexVector::operator*=(complex_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= c; });
return *this;
}
ComplexVector &ComplexVector::operator*=(real_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= c; });
return *this;
}
ComplexVector &ComplexVector::operator*=(const ComplexVector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator*=(const Vector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator/=(complex_t c)
{
const bool use_dev = UseDevice();
const int N = size;
const complex_t m = conj(c) / norm(c);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= m; });
return *this;
}
ComplexVector &ComplexVector::operator/=(real_t c)
{
const bool use_dev = UseDevice();
const int N = size;
const real_t m = 1.0/c;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] *= m; });
return *this;
}
ComplexVector &ComplexVector::operator/=(const ComplexVector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] /= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator/=(const Vector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] /= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator-=(complex_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] -= c; });
return *this;
}
ComplexVector &ComplexVector::operator-=(real_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] -= c; });
return *this;
}
ComplexVector &ComplexVector::operator-=(const ComplexVector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] -= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator-=(const Vector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] -= x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator+=(complex_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] += c; });
return *this;
}
ComplexVector &ComplexVector::operator+=(real_t c)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] += c; });
return *this;
}
ComplexVector &ComplexVector::operator+=(const ComplexVector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] += x[i]; });
return *this;
}
ComplexVector &ComplexVector::operator+=(const Vector &v)
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] += x[i]; });
return *this;
}
ComplexVector &ComplexVector::Set(const Vector &Vr, const Vector &Vi)
{
MFEM_ASSERT(size == Vr.size && size == Vi.size, "incompatible Vectors!");
const bool use_dev = UseDevice() || Vr.UseDevice() || Vi.UseDevice();
const int N = size;
const auto x = Vr.Read(use_dev);
const auto y = Vi.Read(use_dev);
auto z = Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ z[i] = complex_t(x[i], y[i]); });
return *this;
}
const Vector &ComplexVector::real() const
{
re_part.SetSize(size);
const bool use_dev = UseDevice();
const int N = size;
const auto z = Read(use_dev);
auto x = re_part.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ x[i] = z[i].real(); });
return re_part;
}
const Vector &ComplexVector::imag() const
{
im_part.SetSize(size);
const bool use_dev = UseDevice();
const int N = size;
const auto z = Read(use_dev);
auto y = im_part.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{ y[i] = z[i].imag(); });
return im_part;
}
}
+479
View File
@@ -0,0 +1,479 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_COMPLEX_VECTOR
#define MFEM_COMPLEX_VECTOR
#include "vector.hpp"
#include "../general/complex_type.hpp"
namespace mfem
{
class ComplexVector
{
private:
Memory<complex_t > data;
int size;
mutable Vector re_part;
mutable Vector im_part;
public:
/// Default constructor for ComplexVector. Sets size = 0
ComplexVector() : size(0) { }
/// Copy constructor. Allocates a new data array and copies the data.
ComplexVector(const ComplexVector &);
/// Copy constructor. Allocates a new data array and copies the
/// data into real part of this vector.
ComplexVector(const Vector &);
/// Move constructor. "Steals" data from its argument.
ComplexVector(ComplexVector&& v);
/// @brief Creates vector of size s.
/// @warning Entries are not initialized to zero!
explicit ComplexVector(int s);
/// Creates a vector referencing an array of complex<doubles>,
/// owned by someone else.
/// The pointer @a data_ can be NULL. The data array can be replaced later
/// with SetData().
ComplexVector(complex_t *data_, int size_)
{ data.Wrap(data_, size_, false); size = size_; }
/// @brief Create a ComplexVector referencing a sub-vector of the
// ComplexVector @a base starting at the given offset, @a
// base_offset, and size @a size_.
ComplexVector(ComplexVector &base, int base_offset, int size_)
: data(base.data, base_offset, size_), size(size_) { }
/// Create a ComplexVector of size @a size_ using MemoryType @a mt.
ComplexVector(int size_, MemoryType mt)
: data(size_, mt), size(size_) { }
/// @brief Create a ComplexVector of size @a size_ using host
/// MemoryType @a h_mt and device MemoryType @a d_mt.
ComplexVector(int size_, MemoryType h_mt, MemoryType d_mt)
: data(size_, h_mt, d_mt), size(size_) { }
/// Create a vector from a statically sized C-style array of convertible type
template <typename CT, int N>
explicit ComplexVector(const CT (&values)[N]) : ComplexVector(N)
{ std::copy(values, values + N, begin()); }
/// Create a vector using a braced initializer list
template <typename CT, typename std::enable_if<
std::is_convertible<CT,complex_t >::value,bool>::type = true>
explicit ComplexVector(std::initializer_list<CT> values) : ComplexVector(
values.size())
{ std::copy(values.begin(), values.end(), begin()); }
/// Enable execution of Vector operations using the mfem::Device.
/// The default is to use Backend::CPU (serial execution on each MPI rank),
/// regardless of the mfem::Device configuration.
///
/// When appropriate, MFEM functions and class methods will enable the use
/// of the mfem::Device for their Vector parameters.
///
/// Some derived classes, e.g. GridFunction, enable the use of the
/// mfem::Device by default.
virtual void UseDevice(bool use_dev) const { data.UseDevice(use_dev); }
/// Return the device flag of the Memory object used by the Vector
virtual bool UseDevice() const { return data.UseDevice(); }
/// @brief Resize the vector to size @a s.
/// If the new size is less than or equal to Capacity() then the internal
/// data array remains the same. Otherwise, the old array is deleted, if
/// owned, and a new array of size @a s is allocated without copying the
/// previous content of the ComplexVector.
/// @warning In the second case above (new size greater than current one),
/// the vector will allocate new data array, even if it did not own the
/// original data! Also, new entries are not initialized!
void SetSize(int s);
/// Resize the vector to size @a s using MemoryType @a mt.
void SetSize(int s, MemoryType mt);
/// Resize the vector to size @a s using the MemoryType of @a v.
void SetSize(int s, const ComplexVector &v)
{ SetSize(s, v.GetMemory().GetMemoryType()); }
/// Resize the vector to size @a s using the MemoryType of @a v.
void SetSize(int s, const Vector &v)
{ SetSize(s, v.GetMemory().GetMemoryType()); }
/// Set the Vector data.
/// @warning This method should be called only when OwnsData() is false.
void SetData(complex_t *d)
{ data.Wrap(d, data.Capacity(), false); }
/// Set the Vector data and size.
/// The Vector does not assume ownership of the new data. The new size is
/// also used as the new Capacity().
/// @warning This method should be called only when OwnsData() is false.
/// @sa NewDataAndSize().
void SetDataAndSize(complex_t *d, int s)
{ data.Wrap(d, s, false); size = s; }
/// Set the Vector data and size, deleting the old data, if owned.
/// The Vector does not assume ownership of the new data. The new size is
/// also used as the new Capacity().
/// @sa SetDataAndSize().
void NewDataAndSize(complex_t *d, int s)
{
data.Delete();
SetDataAndSize(d, s);
}
/// Reset the Vector to use the given external Memory @a mem and size @a s.
/// If @a own_mem is false, the Vector will not own any of the pointers of
/// @a mem.
///
/// Note that when @a own_mem is true, the @a mem object can be destroyed
/// immediately by the caller but `mem.Delete()` should NOT be called since
/// the Vector object takes ownership of all pointers owned by @a mem.
///
/// @sa NewDataAndSize().
inline void NewMemoryAndSize(const Memory<complex_t > &mem,
int s, bool own_mem);
/// Reset the Vector to be a reference to a sub-vector of @a base.
inline void MakeRef(ComplexVector &base, int offset, int size);
/// @brief Reset the Vector to be a reference to a sub-vector of @a base
/// without changing its current size.
inline void MakeRef(ComplexVector &base, int offset);
/// Set the Vector data (host pointer) ownership flag.
void MakeDataOwner() const { data.SetHostPtrOwner(true); }
/// Destroy a vector
void Destroy();
/// @brief Delete the device pointer, if owned. If @a copy_to_host is true
/// and the data is valid only on device, move it to host before deleting.
/// Invalidates the device memory.
void DeleteDevice(bool copy_to_host = true)
{ data.DeleteDevice(copy_to_host); }
/// Returns the size of the vector.
inline int Size() const { return size; }
/// Return the size of the currently allocated data array.
/// It is always true that Capacity() >= Size().
inline int Capacity() const { return data.Capacity(); }
/// Return a pointer to the beginning of the ComplexVector data.
/// @warning This method should be used with caution as it gives write access
/// to the data of const-qualified ComplexVector%s.
inline complex_t *GetData() const
{ return const_cast<complex_t*>((const complex_t*)data); }
/// STL-like begin.
inline complex_t *begin() { return data; }
/// STL-like end.
inline complex_t *end() { return data + size; }
/// STL-like begin (const version).
inline const complex_t *begin() const { return data; }
/// STL-like end (const version).
inline const complex_t *end() const { return data + size; }
/// Return a reference to the Memory object used by the Vector.
Memory<complex_t > &GetMemory() { return data; }
/// @brief Return a reference to the Memory object used by the
/// ComplexVector, const version.
const Memory<complex_t > &GetMemory() const { return data; }
/// Update the memory location of the vector to match @a v.
void SyncMemory(const ComplexVector &v) const
{ GetMemory().Sync(v.GetMemory()); }
/// Update the alias memory location of the vector to match @a v.
void SyncAliasMemory(const ComplexVector &v) const
{ GetMemory().SyncAlias(v.GetMemory(),Size()); }
/// Read the Vector data (host pointer) ownership flag.
inline bool OwnsData() const { return data.OwnsHostPtr(); }
/// Changes the ownership of the data; after the call the Vector is empty
inline void StealData(complex_t **p)
{ *p = data; data.Reset(); size = 0; }
/// Changes the ownership of the data; after the call the Vector is empty
inline complex_t *StealData()
{ complex_t *p; StealData(&p); return p; }
/// Access Vector entries. Index i = 0 .. size-1.
complex_t &Elem(int i);
/// Read only access to Vector entries. Index i = 0 .. size-1.
const complex_t &Elem(int i) const;
/// Access Vector entries using () for 0-based indexing.
/// @note If MFEM_DEBUG is enabled, bounds checking is performed.
inline complex_t &operator()(int i);
/// Read only access to Vector entries using () for 0-based indexing.
/// @note If MFEM_DEBUG is enabled, bounds checking is performed.
inline const complex_t &operator()(int i) const;
/// Access Vector entries using [] for 0-based indexing.
/// @note If MFEM_DEBUG is enabled, bounds checking is performed.
inline complex_t &operator[](int i) { return (*this)(i); }
/// Read only access to Vector entries using [] for 0-based indexing.
/// @note If MFEM_DEBUG is enabled, bounds checking is performed.
inline const complex_t &operator[](int i) const
{ return (*this)(i); }
/// Dot product with a `complex<double> *` array.
/// @note No complex conjugate is performed
complex_t operator*(const complex_t *v) const;
complex_t operator*(const real_t *v) const;
/// Return the inner-product.
/// @note No complex conjugate is performed
complex_t operator*(const ComplexVector &v) const;
complex_t operator*(const Vector &v) const;
/// Copy Size() entries from @a v.
ComplexVector &operator=(const complex_t *v);
ComplexVector &operator=(const real_t *v);
/// Copy assignment.
/// @note Defining this method overwrites the implicitly defined copy
/// assignment operator.
ComplexVector &operator=(const ComplexVector &v);
ComplexVector &operator=(const Vector &v);
/// Move assignment
ComplexVector &operator=(ComplexVector&& v);
/// Redefine '=' for vector = constant.
ComplexVector &operator=(complex_t value);
ComplexVector &operator=(real_t value);
/// Scale vector by a constant
ComplexVector &operator*=(complex_t c);
ComplexVector &operator*=(real_t c);
/// Component-wise scaling: (*this)(i) *= v(i)
ComplexVector &operator*=(const ComplexVector &v);
ComplexVector &operator*=(const Vector &v);
/// Divide vector by a consant
ComplexVector &operator/=(complex_t c);
ComplexVector &operator/=(real_t c);
/// Component-wise division: (*this)(i) /= v(i)
ComplexVector &operator/=(const ComplexVector &v);
ComplexVector &operator/=(const Vector &v);
/// Subtract a constant from this vector
ComplexVector &operator-=(complex_t c);
ComplexVector &operator-=(real_t c);
/// Subtract a vector from this vector
ComplexVector &operator-=(const ComplexVector &v);
ComplexVector &operator-=(const Vector &v);
/// Add a constant to this vector
ComplexVector &operator+=(complex_t c);
ComplexVector &operator+=(real_t c);
/// Add a vector to this vector
ComplexVector &operator+=(const ComplexVector &v);
ComplexVector &operator+=(const Vector &v);
/// (*this) = x + i * y
ComplexVector &Set(const Vector &x, const Vector &y);
/// Swap the contents of two Vectors
inline void Swap(ComplexVector &other);
/// Return a reference to the real part of this vector
const Vector &real() const;
/// Return a reference to the imaginary part of this vector
const Vector &imag() const;
/// Destroys vector.
virtual ~ComplexVector();
/// Shortcut for mfem::Read(vec.GetMemory(), vec.Size(), on_dev).
virtual const complex_t *Read(bool on_dev = true) const
{ return mfem::Read(data, size, on_dev); }
/// Shortcut for mfem::Read(vec.GetMemory(), vec.Size(), false).
virtual const complex_t *HostRead() const
{ return mfem::Read(data, size, false); }
/// Shortcut for mfem::Write(vec.GetMemory(), vec.Size(), on_dev).
virtual complex_t *Write(bool on_dev = true)
{ return mfem::Write(data, size, on_dev); }
/// Shortcut for mfem::Write(vec.GetMemory(), vec.Size(), false).
virtual complex_t *HostWrite()
{ return mfem::Write(data, size, false); }
/// Shortcut for mfem::ReadWrite(vec.GetMemory(), vec.Size(), on_dev).
virtual complex_t *ReadWrite(bool on_dev = true)
{ return mfem::ReadWrite(data, size, on_dev); }
/// Shortcut for mfem::ReadWrite(vec.GetMemory(), vec.Size(), false).
virtual complex_t *HostReadWrite()
{ return mfem::ReadWrite(data, size, false); }
};
inline ComplexVector::ComplexVector(int s)
{
MFEM_ASSERT(s>=0,"Unexpected negative size.");
size = s;
if (s > 0)
{
data.New(s);
}
}
inline void ComplexVector::SetSize(int s)
{
if (s == size)
{
return;
}
if (s <= data.Capacity())
{
size = s;
return;
}
// preserve a valid MemoryType and device flag
const MemoryType mt = data.GetMemoryType();
const bool use_dev = data.UseDevice();
data.Delete();
size = s;
data.New(s, mt);
data.UseDevice(use_dev);
}
inline void ComplexVector::SetSize(int s, MemoryType mt)
{
if (mt == data.GetMemoryType())
{
if (s == size)
{
return;
}
if (s <= data.Capacity())
{
size = s;
return;
}
}
const bool use_dev = data.UseDevice();
data.Delete();
if (s > 0)
{
data.New(s, mt);
size = s;
}
else
{
data.Reset();
size = 0;
}
data.UseDevice(use_dev);
}
inline void ComplexVector::NewMemoryAndSize(
const Memory<complex_t > &mem,
int s,
bool own_mem)
{
data.Delete();
size = s;
if (own_mem)
{
data = mem;
}
else
{
data.MakeAlias(mem, 0, s);
}
}
inline void ComplexVector::MakeRef(ComplexVector &base, int offset, int s)
{
data.Delete();
size = s;
data.MakeAlias(base.GetMemory(), offset, s);
}
inline void ComplexVector::MakeRef(ComplexVector &base, int offset)
{
data.Delete();
data.MakeAlias(base.GetMemory(), offset, size);
}
inline void ComplexVector::Destroy()
{
const bool use_dev = data.UseDevice();
data.Delete();
size = 0;
data.Reset();
data.UseDevice(use_dev);
}
inline complex_t &ComplexVector::operator()(int i)
{
MFEM_ASSERT(data && i >= 0 && i < size,
"index [" << i << "] is out of range [0," << size << ")");
return data[i];
}
inline const complex_t &ComplexVector::operator()(int i) const
{
MFEM_ASSERT(data && i >= 0 && i < size,
"index [" << i << "] is out of range [0," << size << ")");
return data[i];
}
inline void ComplexVector::Swap(ComplexVector &other)
{
mfem::Swap(data, other.data);
mfem::Swap(size, other.size);
}
/// Specialization of the template function Swap<> for class ComplexVector
template<> inline void Swap<ComplexVector>(ComplexVector &a, ComplexVector &b)
{
a.Swap(b);
}
inline ComplexVector::~ComplexVector()
{
data.Delete();
}
} // namespace mfem
#endif
+53 -28
View File
@@ -124,24 +124,21 @@ const real_t &DenseMatrix::Elem(int i, int j) const
void DenseMatrix::Mult(const real_t *x, real_t *y) const
{
HostRead();
kernels::Mult(height, width, Data(), x, y);
kernels::Mult(height, width, HostRead(), x, y);
}
void DenseMatrix::Mult(const real_t *x, Vector &y) const
{
MFEM_ASSERT(height == y.Size(), "incompatible dimensions");
y.HostReadWrite();
Mult(x, y.GetData());
Mult(x, y.HostWrite());
}
void DenseMatrix::Mult(const Vector &x, real_t *y) const
{
MFEM_ASSERT(width == x.Size(), "incompatible dimensions");
x.HostRead();
Mult(x.GetData(), y);
Mult(x.HostRead(), y);
}
void DenseMatrix::Mult(const Vector &x, Vector &y) const
@@ -149,9 +146,15 @@ void DenseMatrix::Mult(const Vector &x, Vector &y) const
MFEM_ASSERT(height == y.Size() && width == x.Size(),
"incompatible dimensions");
x.HostRead();
y.HostReadWrite();
Mult(x.GetData(), y.GetData());
Mult(x.HostRead(), y.HostWrite());
}
void DenseMatrix::AbsMult(const Vector &x, Vector &y) const
{
MFEM_ASSERT(height == y.Size() && width == x.Size(),
"incompatible dimensions");
kernels::AbsMult(height, width, HostRead(), x.HostRead(), y.HostWrite());
}
real_t DenseMatrix::operator *(const DenseMatrix &m) const
@@ -171,34 +174,21 @@ real_t DenseMatrix::operator *(const DenseMatrix &m) const
void DenseMatrix::MultTranspose(const real_t *x, real_t *y) const
{
HostRead();
real_t *d_col = Data();
for (int col = 0; col < width; col++)
{
real_t y_col = 0.0;
for (int row = 0; row < height; row++)
{
y_col += x[row]*d_col[row];
}
y[col] = y_col;
d_col += height;
}
kernels::MultTranspose(height, width, HostRead(), x, y);
}
void DenseMatrix::MultTranspose(const real_t *x, Vector &y) const
{
MFEM_ASSERT(width == y.Size(), "incompatible dimensions");
y.HostReadWrite();
MultTranspose(x, y.GetData());
MultTranspose(x, y.HostWrite());
}
void DenseMatrix::MultTranspose(const Vector &x, real_t *y) const
{
MFEM_ASSERT(height == x.Size(), "incompatible dimensions");
x.HostRead();
MultTranspose(x.GetData(), y);
MultTranspose(x.HostRead(), y);
}
void DenseMatrix::MultTranspose(const Vector &x, Vector &y) const
@@ -206,9 +196,16 @@ void DenseMatrix::MultTranspose(const Vector &x, Vector &y) const
MFEM_ASSERT(height == x.Size() && width == y.Size(),
"incompatible dimensions");
x.HostRead();
y.HostReadWrite();
MultTranspose(x.GetData(), y.GetData());
MultTranspose(x.HostRead(), y.HostWrite());
}
void DenseMatrix::AbsMultTranspose(const Vector &x, Vector &y) const
{
MFEM_ASSERT(height == x.Size() && width == y.Size(),
"incompatible dimensions");
kernels::AbsMultTranspose(height, width, HostRead(),
x.HostRead(), y.HostWrite());
}
void DenseMatrix::AddMult(const Vector &x, Vector &y, const real_t a) const
@@ -4408,4 +4405,32 @@ void BatchLUSolve(const DenseTensor &Mlu, const Array<int> &P, Vector &X)
BatchedLinAlg::LUSolve(Mlu, P, X);
}
#ifdef MFEM_USE_LAPACK
void BandedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
Array<int> &ipiv)
{
int LDAB = (2*KL) + KU + 1;
int N = AB.NumCols();
int NRHS = B.NumCols();
int info;
ipiv.SetSize(N);
MFEM_LAPACK_PREFIX(gbsv_)(&N, &KL, &KU, &NRHS, AB.GetData(), &LDAB,
ipiv.GetData(), B.GetData(), &N, &info);
MFEM_ASSERT(info == 0, "BandedSolve failed in LAPACK");
}
void BandedFactorizedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
bool transpose, Array<int> &ipiv)
{
int LDAB = (2*KL) + KU + 1;
int N = AB.NumCols();
int NRHS = B.NumCols();
char trans = transpose ? 'T' : 'N';
int info;
MFEM_LAPACK_PREFIX(gbtrs_)(&trans, &N, &KL, &KU, &NRHS, AB.GetData(), &LDAB,
ipiv.GetData(), B.GetData(), &N, &info);
MFEM_ASSERT(info == 0, "BandedFactorizedSolve failed in LAPACK");
}
#endif
} // namespace mfem
+14
View File
@@ -24,6 +24,7 @@ class DenseMatrix : public Matrix
{
friend class DenseTensor;
friend class DenseMatrixInverse;
friend class ComplexTypeDenseMatrix;
private:
Memory<real_t> data;
@@ -153,6 +154,9 @@ public:
/// Matrix vector multiplication.
void Mult(const Vector &x, Vector &y) const override;
/// Absolute-value matrix vector multiplication.
void AbsMult(const Vector &x, Vector &y) const override;
/// Multiply a vector with the transpose matrix.
void MultTranspose(const real_t *x, real_t *y) const;
@@ -165,6 +169,9 @@ public:
/// Multiply a vector with the transpose matrix.
void MultTranspose(const Vector &x, Vector &y) const override;
/// Multiply a vector with the absolute-value transpose matrix.
void AbsMultTranspose(const Vector &x, Vector &y) const override;
using Operator::Mult;
using Operator::MultTranspose;
@@ -1323,6 +1330,13 @@ void BatchLUFactor(DenseTensor &Mlu, Array<int> &P, const real_t TOL = 0.0);
dimension m x n. */
void BatchLUSolve(const DenseTensor &Mlu, const Array<int> &P, Vector &X);
#ifdef MFEM_USE_LAPACK
void BandedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
Array<int> &ipiv);
void BandedFactorizedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
bool transpose, Array<int> &ipiv);
#endif
// Inline methods
inline real_t &DenseMatrix::operator()(int i, int j)
+12
View File
@@ -2574,6 +2574,18 @@ void HypreParMatrix::EliminateBC(const Array<int> &ess_dofs,
#if defined(HYPRE_USING_GPU)
if (HypreUsingGPU())
{
#if defined(HYPRE_WITH_GPU_AWARE_MPI) || defined(HYPRE_USING_GPU_AWARE_MPI)
// hypre_GetGpuAwareMPI() was introduced in v2.31.0, however, its value
// is not checked in hypre_ParCSRCommHandleCreate_v2() before v2.33.0,
// instead only HYPRE_WITH_GPU_AWARE_MPI is checked.
#if MFEM_HYPRE_VERSION >= 23300
if (hypre_GetGpuAwareMPI())
#endif
{
// ensure int_buf_data has been computed before sending it
MFEM_STREAM_SYNC;
}
#endif
// Try to use device-aware MPI for the communication if available
comm_handle = hypre_ParCSRCommHandleCreate_v2(
11, comm_pkg, HYPRE_MEMORY_DEVICE, int_buf_data,
+9
View File
@@ -778,10 +778,19 @@ public:
of the matrix A. */
void AbsMult(real_t a, const Vector &x, real_t b, Vector &y) const;
/// @brief Computes y = |A| * x, using entry-wise absolute values of the matrix A.
void AbsMult(const Vector &x, Vector &y) const override
{ AbsMult(1.0, x, 0.0, y); }
/** @brief Computes y = a * |At| * x + b * y, using entry-wise absolute
values of the transpose of the matrix A. */
void AbsMultTranspose(real_t a, const Vector &x, real_t b, Vector &y) const;
/** @brief Computes y = |At| * x, using entry-wise absolute values of the
matrix A. */
void AbsMultTranspose(const Vector &x, Vector &y) const override
{ AbsMultTranspose(1.0, x, 0.0, y); }
/** @brief The "Boolean" analog of y = alpha * A * x + beta * y, where
elements in the sparsity pattern of the matrix are treated as "true". */
void BooleanMult(int alpha, const int *x, int beta, int *y)
+63
View File
@@ -188,6 +188,40 @@ void Mult(const int height, const int width, const TA *data, const TX *x, TY *y)
}
}
/** @brief Absolute-value matrix vector multiplication: y = |A| x, where the
matrix A is of size @a height x @a width with given @a data, while @a x and
@a y specify the data of the input and output vectors. */
template<typename TA, typename TX, typename TY>
MFEM_HOST_DEVICE inline
void AbsMult(const int height, const int width, const TA *data,
const TX *x, TY *y)
{
if (width == 0)
{
for (int row = 0; row < height; row++)
{
y[row] = 0.0;
}
return;
}
const TA *d_col = data;
TX x_col = x[0];
for (int row = 0; row < height; row++)
{
y[row] = x_col*std::fabs(d_col[row]);
}
d_col += height;
for (int col = 1; col < width; col++)
{
x_col = x[col];
for (int row = 0; row < height; row++)
{
y[row] += x_col*std::fabs(d_col[row]);
}
d_col += height;
}
}
/** @brief Matrix transpose vector multiplication: y = At x, where the matrix A
is of size @a height x @a width with given @a data, while @a x and @a y
specify the data of the input and output vectors. */
@@ -217,6 +251,35 @@ void MultTranspose(const int height, const int width, const TA *data,
}
}
/** @brief Absolute-value matrix transpose vector multiplication: y = |At| x,
where the matrix A is of size @a height x @a width with given @a data, while
@a x and @a y specify the data of the input and output vectors. */
template<typename TA, typename TX, typename TY>
MFEM_HOST_DEVICE inline
void AbsMultTranspose(const int height, const int width, const TA *data,
const TX *x, TY *y)
{
if (height == 0)
{
for (int row = 0; row < width; row++)
{
y[row] = 0.0;
}
return;
}
TY *y_off = y;
for (int i = 0; i < width; ++i)
{
TY val = 0.0;
for (int j = 0; j < height; ++j)
{
val += x[j] * std::fabs(data[i * height + j]);
}
*y_off = val;
y_off++;
}
}
/// Symmetrize a square matrix with given @a size and @a data: A -> (A+A^T)/2.
template<typename T>
MFEM_HOST_DEVICE inline
+7
View File
@@ -42,6 +42,13 @@ extern "C" void
MFEM_LAPACK_PREFIX(getri_)(int *N, real_t *A, int *LDA, int *IPIV, real_t *WORK,
int *LWORK, int *INFO);
extern "C" void
MFEM_LAPACK_PREFIX(gbsv_)(int *, int *, int *, int *, real_t *, int *, int *,
real_t *, int *, int *);
extern "C" void
MFEM_LAPACK_PREFIX(gbtrs_)(char *, int *, int *, int *, int *, real_t *, int *,
int *, real_t *, int *, int *);
extern "C" void
MFEM_LAPACK_PREFIX(syevr_)(char *JOBZ, char *RANGE, char *UPLO, int *N,
real_t *A, int *LDA, real_t *VL, real_t *VU, int *IL,
int *IU, real_t *ABSTOL, int *M, real_t *W,
+74
View File
@@ -645,18 +645,92 @@ void ConstrainedOperator::ConstrainedMult(const Vector &x, Vector &y,
}
}
void ConstrainedOperator::ConstrainedAbsMult(const Vector &x, Vector &y,
const bool transpose) const
{
const int csz = constraint_list.Size();
if (csz == 0)
{
if (transpose)
{
A->AbsMultTranspose(x, y);
}
else
{
A->AbsMult(x, y);
}
return;
}
z = x;
auto idx = constraint_list.Read();
// Use read+write access - we are modifying sub-vector of z
auto d_z = z.ReadWrite();
mfem::forall(csz, [=] MFEM_HOST_DEVICE (int i) { d_z[idx[i]] = 0.0; });
if (transpose)
{
A->AbsMultTranspose(z, y);
}
else
{
A->AbsMult(z, y);
}
auto d_x = x.Read();
// Use read+write access - we are modifying sub-vector of y
auto d_y = y.ReadWrite();
switch (diag_policy)
{
case DIAG_ONE:
mfem::forall(csz, [=] MFEM_HOST_DEVICE (int i)
{
const int id = idx[i];
d_y[id] = d_x[id];
});
break;
case DIAG_ZERO:
mfem::forall(csz, [=] MFEM_HOST_DEVICE (int i)
{
const int id = idx[i];
d_y[id] = 0.0;
});
break;
case DIAG_KEEP:
// Needs action of the operator diagonal on vector
mfem_error("ConstrainedOperator::AbsMult #1");
break;
default:
mfem_error("ConstrainedOperator::AbsMult #2");
break;
}
}
void ConstrainedOperator::Mult(const Vector &x, Vector &y) const
{
constexpr bool transpose = false;
ConstrainedMult(x, y, transpose);
}
void ConstrainedOperator::AbsMult(const Vector &x, Vector &y) const
{
constexpr bool transpose = false;
ConstrainedAbsMult(x, y, transpose);
}
void ConstrainedOperator::MultTranspose(const Vector &x, Vector &y) const
{
constexpr bool transpose = true;
ConstrainedMult(x, y, transpose);
}
void ConstrainedOperator::AbsMultTranspose(const Vector &x, Vector &y) const
{
constexpr bool transpose = true;
ConstrainedAbsMult(x, y, transpose);
}
void ConstrainedOperator::AddMult(const Vector &x, Vector &y,
const real_t a) const
{
+38 -6
View File
@@ -88,10 +88,22 @@ public:
/// Operator application: `y=A(x)`.
virtual void Mult(const Vector &x, Vector &y) const = 0;
/** @brief Action of the absolute-value operator: `y=|A|(x)`. The default
behavior in class Operator is to generate an error. If the Operator is a
composition of several operators, the composition unfold into a product
of absolute-value operators too. */
virtual void AbsMult(const Vector &x, Vector &y) const
{ MFEM_ABORT("Operator::AbsMult() is not overridden!"); }
/** @brief Action of the transpose operator: `y=A^t(x)`. The default behavior
in class Operator is to generate an error. */
virtual void MultTranspose(const Vector &x, Vector &y) const
{ mfem_error("Operator::MultTranspose() is not overridden!"); }
{ MFEM_ABORT("Operator::MultTranspose() is not overridden!"); }
/** @brief Action of the transpose absolute-value operator: `y=|A|^t(x)`.
The default behavior in class Operator is to generate an error. */
virtual void AbsMultTranspose(const Vector &x, Vector &y) const
{ MFEM_ABORT("Operator::AbsMultTranspose() is not overridden!"); }
/// Operator application: `y+=A(x)` (default) or `y+=a*A(x)`.
virtual void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const;
@@ -121,7 +133,7 @@ public:
behavior in class Operator is to generate an error. */
virtual Operator &GetGradient(const Vector &x) const
{
mfem_error("Operator::GetGradient() is not overridden!");
MFEM_ABORT("Operator::GetGradient() is not overridden!");
return const_cast<Operator &>(*this);
}
@@ -691,7 +703,7 @@ public:
const Vector &xB, const Vector &fxB,
int jokB, int *jcurB, real_t gammaB)
{
mfem_error("TimeDependentAdjointOperator::SUNImplicitSetupB() is not "
MFEM_ABORT("TimeDependentAdjointOperator::SUNImplicitSetupB() is not "
"overridden!");
return (-1);
}
@@ -709,7 +721,7 @@ public:
see the SUNDIALS User Guides. */
virtual int SUNImplicitSolveB(Vector &x, const Vector &b, real_t tol)
{
mfem_error("TimeDependentAdjointOperator::SUNImplicitSolveB() is not "
MFEM_ABORT("TimeDependentAdjointOperator::SUNImplicitSolveB() is not "
"overridden!");
return (-1);
}
@@ -930,6 +942,10 @@ public:
void Mult(const Vector & x, Vector & y) const override
{ P.Mult(x, Px); A.Mult(Px, APx); Rt.MultTranspose(APx, y); }
/// Operator-wise absolute-value application.
void AbsMult(const Vector & x, Vector & y) const override
{ P.AbsMult(x, Px); A.AbsMult(Px, APx); Rt.AbsMultTranspose(APx, y); }
/// Approximate diagonal of the RAP Operator.
/** Returns the diagonal of A, as returned by its AssembleDiagonal method,
multiplied be P^T.
@@ -950,6 +966,14 @@ public:
/// Application of the transpose.
void MultTranspose(const Vector & x, Vector & y) const override
{ Rt.Mult(x, APx); A.MultTranspose(APx, Px); P.MultTranspose(Px, y); }
/// Operator-wise absolute-value application of the transpose
void AbsMultTranspose(const Vector & x, Vector & y) const override
{
Rt.AbsMult(x, APx);
A.AbsMultTranspose(APx, Px);
P.AbsMultTranspose(Px, y);
}
};
@@ -1045,13 +1069,21 @@ public:
void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const override;
void AbsMult(const Vector &x, Vector &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override;
void AbsMultTranspose(const Vector &x, Vector &y) const override;
/** @brief Implementation of Mult or MultTranspose.
* TODO - Generalize to allow constraining rows and columns differently.
*/
TODO - Generalize to allow constraining rows and columns differently. */
void ConstrainedMult(const Vector &x, Vector &y, const bool transpose) const;
/** @brief Implementation of AbsMult or AbsMultTranspose.
TODO - Generalize to allow constraining rows and columns differently. */
void ConstrainedAbsMult(const Vector &x, Vector &y,
const bool transpose) const;
/// Destructor: destroys the unconstrained Operator, if owned.
~ConstrainedOperator() override { if (own_A) { delete A; } }
};
+8 -7
View File
@@ -624,7 +624,7 @@ void SLISolver::Mult(const Vector &b, Vector &x) const
}
r0 = std::max(nom*rel_tol, abs_tol);
if (nom <= r0)
if (Monitor(0, nom, r, x) || nom <= r0)
{
converged = true;
final_iter = 0;
@@ -665,18 +665,13 @@ void SLISolver::Mult(const Vector &b, Vector &x) const
nomold = nom;
bool done = false;
if (nom < r0)
if (Monitor(i, nom, r, x) || nom < r0)
{
converged = true;
final_iter = i;
done = true;
}
if (++i > max_iter)
{
done = true;
}
if (print_options.iterations || (done && print_options.first_and_last))
{
mfem::out << " Iteration : " << setw(3) << right << (i-1)
@@ -684,6 +679,11 @@ void SLISolver::Mult(const Vector &b, Vector &x) const
<< "\tConv. rate: " << cf << '\n';
}
if (++i > max_iter)
{
done = true;
}
if (done) { break; }
}
@@ -700,6 +700,7 @@ void SLISolver::Mult(const Vector &b, Vector &x) const
}
final_norm = nom;
Monitor(final_iter, final_norm, r, x, true);
}
void SLI(const Operator &A, const Vector &b, Vector &x,
+2 -2
View File
@@ -422,13 +422,13 @@ public:
void BooleanMultTranspose(const Array<int> &x, Array<int> &y) const;
/// y = |A| * x, using entry-wise absolute values of matrix A
void AbsMult(const Vector &x, Vector &y) const;
void AbsMult(const Vector &x, Vector &y) const override;
/// y = |At| * x, using entry-wise absolute values of the transpose of matrix A
/** If the matrix is modified, call ResetTranspose() and optionally
EnsureMultTranspose() to make sure this method uses the correct updated
transpose. */
void AbsMultTranspose(const Vector &x, Vector &y) const;
void AbsMultTranspose(const Vector &x, Vector &y) const override;
/// Compute y^t A x
real_t InnerProduct(const Vector &x, const Vector &y) const;
+219 -194
View File
@@ -11,19 +11,18 @@
// Implementation of data type vector
#include "kernels.hpp"
#include "vector.hpp"
#include "../general/forall.hpp"
#include "../general/reducers.hpp"
#include "../general/hash.hpp"
#include "vector.hpp"
#ifdef MFEM_USE_OPENMP
#include <omp.h>
#endif
#include <iostream>
#include <iomanip>
#include <cmath>
#include <ctime>
#include <limits>
namespace mfem
{
@@ -207,7 +206,7 @@ Vector &Vector::operator=(const Vector &v)
UseDevice(v.UseDevice());
#else
SetSize(v.Size());
bool vuse = v.UseDevice();
const bool vuse = v.UseDevice();
const bool use_dev = UseDevice() || vuse;
v.UseDevice(use_dev);
// keep 'data' where it is, unless 'use_dev' is true
@@ -249,8 +248,8 @@ Vector &Vector::operator*=(const Vector &v)
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
auto x = v.Read(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] *= x[i]; });
return *this;
}
@@ -271,8 +270,8 @@ Vector &Vector::operator/=(const Vector &v)
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
auto x = v.Read(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] /= x[i]; });
return *this;
}
@@ -292,8 +291,8 @@ Vector &Vector::operator-=(const Vector &v)
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
auto x = v.Read(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] -= x[i]; });
return *this;
}
@@ -313,8 +312,8 @@ Vector &Vector::operator+=(const Vector &v)
const bool use_dev = UseDevice() || v.UseDevice();
const int N = size;
const auto x = v.Read(use_dev);
auto y = ReadWrite(use_dev);
auto x = v.Read(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] += x[i]; });
return *this;
}
@@ -327,8 +326,8 @@ Vector &Vector::Add(const real_t a, const Vector &Va)
{
const int N = size;
const bool use_dev = UseDevice() || Va.UseDevice();
const auto x = Va.Read(use_dev);
auto y = ReadWrite(use_dev);
auto x = Va.Read(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] += a * x[i]; });
}
return *this;
@@ -340,7 +339,7 @@ Vector &Vector::Set(const real_t a, const Vector &Va)
const bool use_dev = UseDevice() || Va.UseDevice();
const int N = size;
auto x = Va.Read(use_dev);
const auto x = Va.Read(use_dev);
auto y = Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] = a * x[i]; });
return *this;
@@ -352,9 +351,9 @@ void Vector::SetVector(const Vector &v, int offset)
const bool use_dev = UseDevice() || v.UseDevice();
const int vs = v.Size();
const real_t *vp = v.Read(use_dev);
const auto vp = v.Read(use_dev);
// Use read+write access for *this - we only modify some of its entries
real_t *p = ReadWrite(use_dev) + offset;
auto p = ReadWrite(use_dev) + offset;
mfem::forall_switch(use_dev, vs, [=] MFEM_HOST_DEVICE (int i) { p[i] = vp[i]; });
}
@@ -364,8 +363,8 @@ void Vector::AddSubVector(const Vector &v, int offset)
const bool use_dev = UseDevice() || v.UseDevice();
const int vs = v.Size();
const real_t *vp = v.Read(use_dev);
real_t *p = ReadWrite(use_dev) + offset;
const auto vp = v.Read(use_dev);
auto p = ReadWrite(use_dev) + offset;
mfem::forall_switch(use_dev, vs, [=] MFEM_HOST_DEVICE (int i) { p[i] += vp[i]; });
}
@@ -385,6 +384,28 @@ void Vector::Reciprocal()
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] = 1.0/y[i]; });
}
void Vector::Abs()
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
y[i] = std::abs(y[i]);
});
}
void Vector::Pow(const real_t p)
{
const bool use_dev = UseDevice();
const int N = size;
auto y = ReadWrite(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
y[i] = std::pow(y[i], p);
});
}
void add(const Vector &v1, const Vector &v2, Vector &v)
{
MFEM_ASSERT(v.size == v1.size && v.size == v2.size,
@@ -394,8 +415,8 @@ void add(const Vector &v1, const Vector &v2, Vector &v)
const bool use_dev = v1.UseDevice() || v2.UseDevice() || v.UseDevice();
const int N = v.size;
// Note: get read access first, in case v is the same as v1/v2.
auto x1 = v1.Read(use_dev);
auto x2 = v2.Read(use_dev);
const auto x1 = v1.Read(use_dev);
const auto x2 = v2.Read(use_dev);
auto y = v.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { y[i] = x1[i] + x2[i]; });
#else
@@ -426,8 +447,8 @@ void add(const Vector &v1, real_t alpha, const Vector &v2, Vector &v)
const bool use_dev = v1.UseDevice() || v2.UseDevice() || v.UseDevice();
const int N = v.size;
// Note: get read access first, in case v is the same as v1/v2.
auto d_x = v1.Read(use_dev);
auto d_y = v2.Read(use_dev);
const auto d_x = v1.Read(use_dev);
const auto d_y = v2.Read(use_dev);
auto d_z = v.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -465,8 +486,8 @@ void add(const real_t a, const Vector &x, const Vector &y, Vector &z)
const bool use_dev = x.UseDevice() || y.UseDevice() || z.UseDevice();
const int N = x.size;
// Note: get read access first, in case z is the same as x/y.
auto xd = x.Read(use_dev);
auto yd = y.Read(use_dev);
const auto xd = x.Read(use_dev);
const auto yd = y.Read(use_dev);
auto zd = z.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -520,8 +541,8 @@ void add(const real_t a, const Vector &x,
const bool use_dev = x.UseDevice() || y.UseDevice() || z.UseDevice();
const int N = x.size;
// Note: get read access first, in case z is the same as x/y.
auto xd = x.Read(use_dev);
auto yd = y.Read(use_dev);
const auto xd = x.Read(use_dev);
const auto yd = y.Read(use_dev);
auto zd = z.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -550,8 +571,8 @@ void subtract(const Vector &x, const Vector &y, Vector &z)
const bool use_dev = x.UseDevice() || y.UseDevice() || z.UseDevice();
const int N = x.size;
// Note: get read access first, in case z is the same as x/y.
auto xd = x.Read(use_dev);
auto yd = y.Read(use_dev);
const auto xd = x.Read(use_dev);
const auto yd = y.Read(use_dev);
auto zd = z.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -589,8 +610,8 @@ void subtract(const real_t a, const Vector &x, const Vector &y, Vector &z)
const bool use_dev = x.UseDevice() || y.UseDevice() || z.UseDevice();
const int N = x.size;
// Note: get read access first, in case z is the same as x/y.
auto xd = x.Read(use_dev);
auto yd = y.Read(use_dev);
const auto xd = x.Read(use_dev);
const auto yd = y.Read(use_dev);
auto zd = z.Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -631,8 +652,8 @@ void Vector::median(const Vector &lo, const Vector &hi)
const bool use_dev = UseDevice() || lo.UseDevice() || hi.UseDevice();
const int N = size;
// Note: get read access first, in case *this is the same as lo/hi.
auto l = lo.Read(use_dev);
auto h = hi.Read(use_dev);
const auto l = lo.Read(use_dev);
const auto h = hi.Read(use_dev);
auto m = Write(use_dev);
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i)
{
@@ -652,9 +673,9 @@ void Vector::GetSubVector(const Array<int> &dofs, Vector &elemvect) const
const int n = dofs.Size();
elemvect.SetSize(n);
const bool use_dev = dofs.UseDevice() || elemvect.UseDevice();
const auto d_X = Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
auto d_y = elemvect.Write(use_dev);
auto d_X = Read(use_dev);
auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i)
{
const int dof_i = d_dofs[i];
@@ -664,7 +685,7 @@ void Vector::GetSubVector(const Array<int> &dofs, Vector &elemvect) const
void Vector::GetSubVector(const Array<int> &dofs, real_t *elem_data) const
{
data.Read(MemoryClass::HOST, size);
HostRead();
const int n = dofs.Size();
for (int i = 0; i < n; i++)
{
@@ -679,7 +700,7 @@ void Vector::SetSubVector(const Array<int> &dofs, const real_t value)
const int n = dofs.Size();
// Use read+write access for *this - we only modify some of its entries
auto d_X = ReadWrite(use_dev);
auto d_dofs = dofs.Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i)
{
const int j = d_dofs[i];
@@ -721,8 +742,8 @@ void Vector::SetSubVector(const Array<int> &dofs, const Vector &elemvect)
const int n = dofs.Size();
// Use read+write access for X - we only modify some of its entries
auto d_X = ReadWrite(use_dev);
auto d_y = elemvect.Read(use_dev);
auto d_dofs = dofs.Read(use_dev);
const auto d_y = elemvect.Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i)
{
const int dof_i = d_dofs[i];
@@ -740,7 +761,7 @@ void Vector::SetSubVector(const Array<int> &dofs, const Vector &elemvect)
void Vector::SetSubVector(const Array<int> &dofs, real_t *elem_data)
{
// Use read+write access because we overwrite only part of the data.
data.ReadWrite(MemoryClass::HOST, size);
HostReadWrite();
const int n = dofs.Size();
for (int i = 0; i < n; i++)
{
@@ -764,9 +785,9 @@ void Vector::AddElementVector(const Array<int> &dofs, const Vector &elemvect)
const bool use_dev = dofs.UseDevice() || elemvect.UseDevice();
const int n = dofs.Size();
auto d_y = elemvect.Read(use_dev);
const auto d_y = elemvect.Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
auto d_X = ReadWrite(use_dev);
auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i)
{
const int j = d_dofs[i];
@@ -783,7 +804,7 @@ void Vector::AddElementVector(const Array<int> &dofs, const Vector &elemvect)
void Vector::AddElementVector(const Array<int> &dofs, real_t *elem_data)
{
data.ReadWrite(MemoryClass::HOST, size);
HostReadWrite();
const int n = dofs.Size();
for (int i = 0; i < n; i++)
{
@@ -808,9 +829,9 @@ void Vector::AddElementVector(const Array<int> &dofs, const real_t a,
const bool use_dev = dofs.UseDevice() || elemvect.UseDevice();
const int n = dofs.Size();
const auto d_x = elemvect.Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
auto d_y = ReadWrite(use_dev);
auto d_x = elemvect.Read(use_dev);
auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i)
{
const int j = d_dofs[i];
@@ -835,7 +856,7 @@ void Vector::SetSubVectorComplement(const Array<int> &dofs, const real_t val)
Device::GetHostMemoryType());
auto d_data = ReadWrite(use_dev);
auto d_dofs_vals = dofs_vals.Write(use_dev);
auto d_dofs = dofs.Read(use_dev);
const auto d_dofs = dofs.Read(use_dev);
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i) { d_dofs_vals[i] = d_data[d_dofs[i]]; });
mfem::forall_switch(use_dev, N, [=] MFEM_HOST_DEVICE (int i) { d_data[i] = val; });
mfem::forall_switch(use_dev, n, [=] MFEM_HOST_DEVICE (int i) { d_data[d_dofs[i]] = d_dofs_vals[i]; });
@@ -844,7 +865,7 @@ void Vector::SetSubVectorComplement(const Array<int> &dofs, const real_t val)
void Vector::Print(std::ostream &os, int width) const
{
if (!size) { return; }
data.Read(MemoryClass::HOST, size);
HostRead();
for (int i = 0; 1; )
{
os << ZeroSubnormal(data[i]);
@@ -870,7 +891,7 @@ void Vector::Print(adios2stream &os,
const std::string& variable_name) const
{
if (!size) { return; }
data.Read(MemoryClass::HOST, size);
HostRead();
os.engine.Put(variable_name, &data[0] );
}
#endif
@@ -928,10 +949,7 @@ void Vector::PrintHash(std::ostream &os) const
void Vector::Randomize(int seed)
{
if (seed == 0)
{
seed = (int)time(0);
}
if (seed == 0) { seed = (int)time(0); }
srand((unsigned)seed);
@@ -947,20 +965,15 @@ real_t Vector::Norml2() const
// Scale entries of Vector on the fly, using algorithms from
// std::hypot() and LAPACK's drm2. This scaling ensures that the
// argument of each call to std::pow is <= 1 to avoid overflow.
if (size == 0)
{
return 0.0;
}
if (size == 0) { return 0.0; }
auto m_data = Read(UseDevice());
const auto m_data = Read(UseDevice());
using value_type = DevicePair<real_t, real_t>;
value_type res;
res.first = 0;
res.second = 0;
// first compute sum (|m_data|/scale)^2
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, value_type &r)
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, value_type &r)
{
real_t n = fabs(m_data[i]);
if (n > 0)
@@ -987,11 +1000,12 @@ real_t Vector::Normlinf() const
{
if (size == 0) { return 0; }
auto m_data = Read(UseDevice());
real_t res = 0;
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, real_t &r) { r = fmax(r, fabs(m_data[i])); },
const auto m_data = Read(UseDevice());
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r)
{
r = fmax(r, fabs(m_data[i]));
},
MaxReducer<real_t> {}, UseDevice(), vector_workspace());
return res;
}
@@ -1000,11 +1014,12 @@ real_t Vector::Norml1() const
{
if (size == 0) { return 0.0; }
auto m_data = Read(UseDevice());
real_t res = 0;
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, real_t &r) { r += fabs(m_data[i]); },
const auto m_data = Read(UseDevice());
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r)
{
r += fabs(m_data[i]);
},
SumReducer<real_t> {}, UseDevice(), vector_workspace());
return res;
}
@@ -1013,33 +1028,24 @@ real_t Vector::Normlp(real_t p) const
{
MFEM_ASSERT(p > 0.0, "Vector::Normlp");
if (p == 1.0)
{
return Norml1();
}
if (p == 2.0)
{
return Norml2();
}
if (p == 1.0) { return Norml1(); }
if (p == 2.0) { return Norml2(); }
if (p < infinity())
{
// Scale entries of Vector on the fly, using algorithms from
// std::hypot() and LAPACK's drm2. This scaling ensures that the
// argument of each call to std::pow is <= 1 to avoid overflow.
if (size == 0)
{
return 0.0;
}
if (size == 0) { return 0.0; }
auto m_data = Read(UseDevice());
using value_type = DevicePair<real_t, real_t>;
value_type res;
res.first = 0;
res.second = 0;
const auto m_data = Read(UseDevice());
// first compute sum (|m_data|/scale)^p
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, value_type &r)
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, value_type &r)
{
real_t n = fabs(m_data[i]);
if (n > 0)
@@ -1068,163 +1074,182 @@ real_t Vector::Normlp(real_t p) const
real_t Vector::operator*(const Vector &v) const
{
MFEM_ASSERT(size == v.size, "incompatible Vectors!");
if (size == 0) { return 0.0; }
const bool use_dev = UseDevice() || v.UseDevice();
const auto m_data = Read(use_dev), v_data = v.Read(use_dev);
auto m_data = Read(use_dev);
auto v_data = v.Read(use_dev);
if (use_dev)
{
// special path for OCCA and OpenMP
// If OCCA is enabled, it handles all selected backends
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
return occa::linalg::dot<real_t, real_t, real_t>(
OccaMemoryRead(data, size), OccaMemoryRead(v.data, size));
}
if (use_dev && DeviceCanUseOcca())
{
return occa::linalg::dot<real_t, real_t, real_t>(
OccaMemoryRead(data, size), OccaMemoryRead(v.data, size));
}
#endif
#ifdef MFEM_USE_OPENMP
if (Device::Allows(Backend::OMP_MASK))
const auto compute_dot = [&]()
{
real_t res = 0;
reduce(size, res, [=] MFEM_HOST_DEVICE (int i, real_t &r)
{
r += m_data[i] * v_data[i];
},
SumReducer<real_t> {}, use_dev, vector_workspace());
return res;
};
// Device backends have top priority
if (Device::Allows(Backend::DEVICE_MASK)) { return compute_dot(); }
// Special path for OpenMP
#ifdef MFEM_USE_OPENMP
if (use_dev && Device::Allows(Backend::OMP_MASK))
{
// By default, use a deterministic way of computing the dot product
#define MFEM_USE_OPENMP_DETERMINISTIC_DOT
#ifdef MFEM_USE_OPENMP_DETERMINISTIC_DOT
// By default, use a deterministic way of computing the dot product
static Vector th_dot;
#pragma omp parallel
static Vector th_dot;
#pragma omp parallel
{
const int nt = omp_get_num_threads();
#pragma omp master
th_dot.SetSize(nt);
const int tid = omp_get_thread_num();
const int stride = (size + nt - 1) / nt;
const int start = tid * stride;
const int stop = std::min(start + stride, size);
real_t my_dot = 0.0;
for (int i = start; i < stop; i++)
{
const int nt = omp_get_num_threads();
#pragma omp master
th_dot.SetSize(nt);
const int tid = omp_get_thread_num();
const int stride = (size + nt - 1) / nt;
const int start = tid * stride;
const int stop = std::min(start + stride, size);
real_t my_dot = 0.0;
for (int i = start; i < stop; i++)
{
my_dot += m_data[i] * v_data[i];
}
#pragma omp barrier
th_dot(tid) = my_dot;
my_dot += m_data[i] * v_data[i];
}
return th_dot.Sum();
#else
// The standard way of computing the dot product is non-deterministic
real_t prod = 0.0;
#pragma omp parallel for reduction(+ : prod)
for (int i = 0; i < size; i++)
{
prod += m_data[i] * v_data[i];
}
return prod;
#endif // MFEM_USE_OPENMP_DETERMINISTIC_DOT
#pragma omp barrier
th_dot(tid) = my_dot;
}
#endif // MFEM_USE_OPENMP
return th_dot.Sum();
#else
// The standard way of computing the dot product is non-deterministic
real_t prod = 0.0;
#pragma omp parallel for reduction(+ : prod)
for (int i = 0; i < size; i++)
{
prod += m_data[i] * v_data[i];
}
return prod;
#endif // MFEM_USE_OPENMP_DETERMINISTIC_DOT
}
#endif // MFEM_USE_OPENMP
// normal path for everything else (cuda, hip, debug, cpu)
real_t res = 0;
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, real_t &r) { r += m_data[i] * v_data[i]; },
SumReducer<real_t> {}, use_dev, vector_workspace());
return res;
// All other CPU backends
return compute_dot();
}
real_t Vector::Min() const
{
if (size == 0) { return infinity(); }
const bool use_dev = UseDevice();
auto m_data = Read(use_dev);
if (use_dev)
{
// special case for OCCA and OpenMP
const auto use_dev = UseDevice();
const auto m_data = Read(use_dev);
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
return occa::linalg::min<real_t,real_t>(OccaMemoryRead(data, size));
}
#endif
#ifdef MFEM_USE_OPENMP
if (Device::Allows(Backend::OMP_MASK))
{
real_t minimum = m_data[0];
#pragma omp parallel for reduction(min:minimum)
for (int i = 0; i < size; i++)
{
minimum = std::min(minimum, m_data[i]);
}
return minimum;
}
#endif
if (use_dev && DeviceCanUseOcca())
{
return occa::linalg::min<real_t,real_t>(OccaMemoryRead(data, size));
}
#endif
// normal path for everything else (cuda, hip, debug, cpu)
real_t res = infinity();
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, real_t &r) { r = fmin(r, m_data[i]); },
MinReducer<real_t> {}, use_dev, vector_workspace());
return res;
const auto compute_min = [&]()
{
real_t res = infinity();
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r)
{
r = fmin(r, m_data[i]);
},
MinReducer<real_t> {}, use_dev, vector_workspace());
return res;
};
// Device backends have top priority
if (Device::Allows(Backend::DEVICE_MASK)) { return compute_min(); }
// Special path for OpenMP
#ifdef MFEM_USE_OPENMP
if (use_dev && Device::Allows(Backend::OMP_MASK))
{
real_t minimum = m_data[0];
#pragma omp parallel for reduction(min:minimum)
for (int i = 0; i < size; i++)
{
minimum = std::min(minimum, m_data[i]);
}
return minimum;
}
#endif
// All other CPU backends
return compute_min();
}
real_t Vector::Max() const
{
if (size == 0) { return -infinity(); }
const bool use_dev = UseDevice();
auto m_data = Read(use_dev);
const auto use_dev = UseDevice();
const auto m_data = Read(use_dev);
if (use_dev)
{
// special cases where OCCA or OenMP are used
#ifdef MFEM_USE_OCCA
if (DeviceCanUseOcca())
{
return occa::linalg::max<real_t, real_t>(OccaMemoryRead(data, size));
}
#endif
#ifdef MFEM_USE_OPENMP
if (Device::Allows(Backend::OMP_MASK))
{
real_t maximum = m_data[0];
#pragma omp parallel for reduction(max : maximum)
for (int i = 0; i < size; i++)
{
maximum = fmax(maximum, m_data[i]);
}
return maximum;
}
#endif
if (use_dev && DeviceCanUseOcca())
{
return occa::linalg::max<real_t, real_t>(OccaMemoryRead(data, size));
}
#endif
// normal path for everything else (cuda, hip, debug, cpu)
real_t res = -infinity();
reduce(
size, res,
[=] MFEM_HOST_DEVICE(int i, real_t &r) { r = fmax(r, m_data[i]); },
MaxReducer<real_t> {}, use_dev, vector_workspace());
return res;
const auto compute_max = [&]()
{
real_t res = -infinity();
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r)
{
r = fmax(r, m_data[i]);
},
MaxReducer<real_t> {}, use_dev, vector_workspace());
return res;
};
// Device backends have top priority
if (Device::Allows(Backend::DEVICE_MASK)) { return compute_max(); }
// Special path for OpenMP
#ifdef MFEM_USE_OPENMP
if (use_dev && Device::Allows(Backend::OMP_MASK))
{
real_t maximum = m_data[0];
#pragma omp parallel for reduction(max : maximum)
for (int i = 0; i < size; i++)
{
maximum = fmax(maximum, m_data[i]);
}
return maximum;
}
#endif
// All other CPU backends
return compute_max();
}
real_t Vector::Sum() const
{
if (size == 0) { return 0.0; }
auto m_data = Read(UseDevice());
real_t res = 0;
reduce(
size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r) { r += m_data[i]; },
const auto m_data = Read(UseDevice());
reduce(size, res, [=] MFEM_HOST_DEVICE(int i, real_t &r)
{
r += m_data[i];
},
SumReducer<real_t> {}, UseDevice(), vector_workspace());
return res;
}
}
} // namespace mfem
+8
View File
@@ -80,6 +80,8 @@ inline real_t rand_real()
/// Vector data type.
class Vector
{
friend class ComplexVector;
protected:
Memory<real_t> data;
@@ -360,6 +362,12 @@ public:
/// (*this)(i) = 1.0 / (*this)(i)
void Reciprocal();
/// (*this)(i) = abs((*this)(i))
void Abs();
/// (*this)(i) = pow((*this)(i), p)
void Pow(const real_t p);
/// Swap the contents of two Vectors
inline void Swap(Vector &other);
+2 -2
View File
@@ -125,11 +125,11 @@ EXAMPLE_TEST_DIRS := examples
MINIAPP_SUBDIRS = common electromagnetics meshing navier performance tools \
toys nurbs gslib adjoint solvers shifted mtop parelag tribol autodiff dfem \
hooke multidomain dpg hdiv-linear-solver spde
hooke multidomain dpg hdiv-linear-solver spde diag-smoothers
MINIAPP_DIRS := $(addprefix miniapps/,$(MINIAPP_SUBDIRS))
MINIAPP_TEST_DIRS := $(filter-out %/common,$(MINIAPP_DIRS))
MINIAPP_USE_COMMON := $(addprefix miniapps/,electromagnetics meshing tools \
toys shifted dpg)
toys shifted dpg diag-smoothers)
EM_DIRS = $(EXAMPLE_DIRS) $(MINIAPP_DIRS)
+35 -27
View File
@@ -26,6 +26,12 @@
}\
}
#if defined(MFEM_USE_DOUBLE)
#define MFEM_NETCDF_REAL_T NC_DOUBLE
#elif defined(MFEM_USE_SINGLE)
#define MFEM_NETCDF_REAL_T NC_FLOAT
#endif
namespace mfem
{
@@ -135,18 +141,18 @@ public:
/// @brief Writes the mesh to an ExodusII file.
/// @param fpath The path to the file.
/// @param flags NC_CLOBBER will overwrite existing file.
void PrintExodusII(std::string fpath, int flags = NC_CLOBBER);
void PrintExodusII(const std::string &fpath, int flags = NC_CLOBBER);
/// @brief Static method for writing a mesh to an ExodusII file.
/// @param mesh The mesh to write to the file.
/// @param fpath The path to the file.
/// @param flags NetCDF file flags.
static void PrintExodusII(Mesh & mesh, std::string fpath,
static void PrintExodusII(Mesh & mesh, const std::string &fpath,
int flags = NC_CLOBBER);
protected:
/// @brief Closes any open file and creates a NetCDF file using selected flags.
void OpenExodusII(std::string fpath, int flags);
void OpenExodusII(const std::string &fpath, int flags);
/// @brief Closes any open file.
void CloseExodusII();
@@ -167,9 +173,9 @@ protected:
std::unordered_set<int> GenerateUniqueNodeIDs();
/// @brief Populates vectors with x, y, z coordinates from mesh.
void ExtractVertexCoordinates(std::vector<double> & coordx,
std::vector<double> & coordy,
std::vector<double> & coordz);
void ExtractVertexCoordinates(std::vector<real_t> &coordx,
std::vector<real_t> &coordy,
std::vector<real_t> &coordz);
/// @brief Writes node connectivity for a particular block.
/// @param block_id The block to write to the file.
@@ -187,7 +193,7 @@ protected:
/// @brief Writes the number of elements in the mesh.
void WriteNumOfElements();
/// @brief Writes the floating-point word size (4 == float; 8 == double).
/// @brief Writes the floating-point word size (sizeof(real_t)).
void WriteFloatingPointWordSize();
/// @brief Writes the API version.
@@ -291,7 +297,7 @@ private:
std::map<int, std::vector<int>> exodusII_side_ids_for_boundary_id;
};
void Mesh::PrintExodusII(const std::string fpath)
void Mesh::PrintExodusII(const std::string &fpath)
{
ExodusIIWriter::PrintExodusII(*this, fpath);
}
@@ -362,7 +368,7 @@ void ExodusIIWriter::WriteExodusIIMeshInformation()
WriteNodeSets();
}
void ExodusIIWriter::PrintExodusII(std::string fpath, int flags)
void ExodusIIWriter::PrintExodusII(const std::string &fpath, int flags)
{
OpenExodusII(fpath, flags);
@@ -374,7 +380,7 @@ void ExodusIIWriter::PrintExodusII(std::string fpath, int flags)
mfem::out << "Mesh successfully written to Exodus II file" << std::endl;
}
void ExodusIIWriter::PrintExodusII(Mesh & mesh, std::string fpath,
void ExodusIIWriter::PrintExodusII(Mesh &mesh, const std::string &fpath,
int flags)
{
ExodusIIWriter writer(mesh);
@@ -382,7 +388,7 @@ void ExodusIIWriter::PrintExodusII(Mesh & mesh, std::string fpath,
writer.PrintExodusII(fpath, flags);
}
void ExodusIIWriter::OpenExodusII(std::string fpath, int flags)
void ExodusIIWriter::OpenExodusII(const std::string &fpath, int flags)
{
CloseExodusII(); // Close any open files.
@@ -422,7 +428,7 @@ void ExodusIIWriter::WriteNumOfElements()
void ExodusIIWriter::WriteFloatingPointWordSize()
{
const int word_size = 8;
const int word_size = sizeof(real_t);
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_FLOATING_POINT_WORD_SIZE_LABEL,
NC_INT, 1,
&word_size);
@@ -430,13 +436,15 @@ void ExodusIIWriter::WriteFloatingPointWordSize()
void ExodusIIWriter::WriteAPIVersion()
{
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_API_VERSION_LABEL, NC_FLOAT, 1,
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_API_VERSION_LABEL, MFEM_NETCDF_REAL_T,
1,
&ExodusIILabels::EXODUS_API_VERSION);
}
void ExodusIIWriter::WriteDatabaseVersion()
{
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_DATABASE_VERSION_LABEL, NC_FLOAT, 1,
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_DATABASE_VERSION_LABEL,
MFEM_NETCDF_REAL_T, 1,
&ExodusIILabels::EXODUS_DATABASE_VERSION);
}
@@ -607,25 +615,25 @@ void ExodusIIWriter::WriteNodalCoordinates()
DefineDimension("num_nodes", num_nodes, &num_nodes_id);
// 3. Extract the nodal coordinates.
// NB: assume doubles (could be floats!); ndims = 1 (vector).
// NB: writes in format real_t (double or float); ndims = 1 (vector).
// https://docs.unidata.ucar.edu/netcdf-c/current/group__variables.html#gac7e8662c51f3bb07d1fc6d6c6d9052c8
std::vector<double> coordx(num_nodes);
std::vector<double> coordy(num_nodes);
std::vector<double> coordz(mesh.Dimension() == 3 ? num_nodes : 0);
std::vector<real_t> coordx(num_nodes);
std::vector<real_t> coordy(num_nodes);
std::vector<real_t> coordz(mesh.Dimension() == 3 ? num_nodes : 0);
ExtractVertexCoordinates(coordx, coordy, coordz);
// 4. Define and put the nodal coordinates.
DefineAndPutVar(ExodusIILabels::EXODUS_COORDX_LABEL, NC_DOUBLE, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDX_LABEL, MFEM_NETCDF_REAL_T, 1,
&num_nodes_id,
coordx.data());
DefineAndPutVar(ExodusIILabels::EXODUS_COORDY_LABEL, NC_DOUBLE, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDY_LABEL, MFEM_NETCDF_REAL_T, 1,
&num_nodes_id,
coordy.data());
if (mesh.Dimension() == 3)
{
DefineAndPutVar(ExodusIILabels::EXODUS_COORDZ_LABEL, NC_DOUBLE, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDZ_LABEL, MFEM_NETCDF_REAL_T, 1,
&num_nodes_id,
coordz.data());
}
@@ -770,9 +778,9 @@ void ExodusIIWriter::WriteNodeConnectivityForBlock(const int block_id)
}
void ExodusIIWriter::ExtractVertexCoordinates(std::vector<double> & coordx,
std::vector<double> & coordy,
std::vector<double> & coordz)
void ExodusIIWriter::ExtractVertexCoordinates(std::vector<real_t> & coordx,
std::vector<real_t> & coordy,
std::vector<real_t> & coordz)
{
if (mesh.GetNodes()) // Higher-order.
{
@@ -782,7 +790,7 @@ void ExodusIIWriter::ExtractVertexCoordinates(std::vector<double> & coordx,
sorted_node_ids.assign(unordered_node_ids.begin(), unordered_node_ids.end());
std::sort(sorted_node_ids.begin(), sorted_node_ids.end());
double coordinates[3];
real_t coordinates[3];
for (size_t i = 0; i < sorted_node_ids.size(); i++)
{
int node_id = sorted_node_ids[i];
@@ -802,7 +810,7 @@ void ExodusIIWriter::ExtractVertexCoordinates(std::vector<double> & coordx,
{
for (int ivertex = 0; ivertex < mesh.GetNV(); ivertex++)
{
double * coordinates = mesh.GetVertex(ivertex);
real_t *coordinates = mesh.GetVertex(ivertex);
coordx[ivertex] = coordinates[0];
coordy[ivertex] = coordinates[1];
@@ -1080,4 +1088,4 @@ void ExodusIIWriter::CheckNodalFESpaceIsSecondOrderH1() const
#endif
}
}
+128 -181
View File
@@ -2994,6 +2994,7 @@ void Mesh::DoNodeReorder(DSTable *old_v_to_v, Table *old_elem_vert)
const int num_edge_dofs = old_dofs.Size();
// Save the original nodes
Nodes->HostReadWrite(); // for "(*Nodes)() = "
const Vector onodes = *Nodes;
// vertex dofs do not need to be moved
@@ -3249,6 +3250,19 @@ int Mesh::GetPatchBdrAttribute(int i) const
return NURBSext->GetPatchBdrAttribute(i);
}
void Mesh::GetNURBSPatches(Array<NURBSPatch*> &patches)
{
MFEM_VERIFY(NURBSext, "Must be a NURBS mesh");
// This sets the data in NURBSPatch(es) from the control points (Nodes)
NURBSext->ConvertToPatches(*Nodes);
// Deep copy patches
NURBSext->GetPatches(patches);
// Among other things, this deletes patches in NURBSext
UpdateNURBS();
}
void Mesh::FinalizeTetMesh(int generate_edges, int refine, bool fix_orientation)
{
FinalizeCheck();
@@ -6273,7 +6287,7 @@ void Mesh::UpdateNURBS()
GenerateFaces();
}
void Mesh::LoadPatchTopo(std::istream &input, Array<int> &edge_to_knot)
void Mesh::LoadPatchTopo(std::istream &input, Array<int> &edge_to_ukv)
{
SetEmpty();
@@ -6313,20 +6327,20 @@ void Mesh::LoadPatchTopo(std::istream &input, Array<int> &edge_to_knot)
if (NumOfEdges > 0)
{
edge_vertex = new Table(NumOfEdges, 2);
edge_to_knot.SetSize(NumOfEdges);
edge_to_ukv.SetSize(NumOfEdges);
for (int j = 0; j < NumOfEdges; j++)
{
int *v = edge_vertex->GetRow(j);
input >> edge_to_knot[j] >> v[0] >> v[1];
input >> edge_to_ukv[j] >> v[0] >> v[1];
if (v[0] > v[1])
{
edge_to_knot[j] = -1 - edge_to_knot[j];
edge_to_ukv[j] = -1 - edge_to_ukv[j];
}
}
}
else
{
edge_to_knot.SetSize(0);
edge_to_ukv.SetSize(0);
}
skip_comment_lines(input, '#');
@@ -6338,196 +6352,129 @@ void Mesh::LoadPatchTopo(std::istream &input, Array<int> &edge_to_knot)
FinalizeTopology();
CheckBdrElementOrientation(); // check and fix boundary element orientation
/* Generate knot 2 edge mapping -- if edges are not specified in the mesh file
See data/two-squares-nurbs-autoedge.mesh for an example */
if (edge_to_knot.Size() == 0)
/* Generate edge to knotvector mapping if edges are not specified in the
mesh file. See miniapps/nurbs/meshes/two-squares-nurbs-autoedge.mesh
for an example */
if (edge_to_ukv.Size() == 0)
{
edge_vertex = new Table(NumOfEdges, 2);
edge_to_knot.SetSize(NumOfEdges);
constexpr int notset = -9999999;
edge_to_knot = notset;
Array<int> edges;
Array<int> oedge;
int knot = 0;
Array<int> ukv_to_rpkv;
GetEdgeToUniqueKnotvector(edge_to_ukv, ukv_to_rpkv);
}
}
Array<int> edge0, edge1;
int flip = 1;
if (Dimension() == 2)
void Mesh::GetEdgeToUniqueKnotvector(Array<int> &edge_to_ukv,
Array<int> &ukv_to_rpkv) const
{
const int dim = Dimension(); // topological (not physical) dimension
const int NP = NumOfElements; // number of patches
const int NPKV = NP * dim; // number of patch knotvectors
constexpr int notset = -9999999;
// Sign convention
auto sign = [](int i) { return -1 - i; };
auto unsign = [](int i) { return (i < 0) ? -1 - i : i; };
// Edge index -> dimension convention
auto edge_to_dim = [](int i) { return (i < 8) ? ((i & 1) ? 1 : 0) : 2; };
Array<int> v(2); // vertices of an edge
// 1D case is special: edge index = signed element index
// ukv_to_rpkv = Identity
if (dim == 1)
{
edge_to_ukv.SetSize(NP);
ukv_to_rpkv.SetSize(NP);
for (int i = 0; i < NP; i++)
{
edge0.SetSize(2);
edge1.SetSize(2);
edge0[0] = 0; edge1[0] = 2;
edge0[1] = 1; edge1[1] = 3;
flip = 1;
GetElementVertices(i, v);
// Sign is based on the edge's vertex indices
edge_to_ukv[i] = (v[1] > v[0]) ? i : sign(i);
ukv_to_rpkv[i] = i;
}
else if (Dimension() == 3)
return;
}
// Local (per-patch) variables
Array<int> edges, oedges;
// Edge index -> signed patch knotvector index (p*dim + d)
Array<int> edge_to_pkv(NumOfEdges);
edge_to_pkv.SetSize(NumOfEdges);
edge_to_pkv = notset;
// Initialize pkv_map as identity - this is the storage for the
// disjoint-set/union-find algorithm which will later be used
// to get the map pkv_to_rpkv
Array<int> pkv_map(NPKV);
for (int i = 0; i < NPKV; i++)
{
pkv_map[i] = i;
}
std::function<int(int)> get_root;
get_root = [&pkv_map, &get_root](int i) -> int
{
return (pkv_map[i] == i) ? i : get_root(pkv_map[i]);
};
auto unite = [&pkv_map, &get_root](int i, int j)
{
const int ri = get_root(i);
const int rj = get_root(j);
if (ri == rj) return;
// keep the lowest index
(ri < rj) ? pkv_map[rj] = ri : pkv_map[ri] = rj;
};
// Get edge_to_pkv (one edge can link to multiple pkv) and pkv_map
for (int p = 0; p < NP; p++)
{
GetElementEdges(p, edges, oedges);
// First loop checks for if edge has already been set
for (int i = 0; i < edges.Size(); i++)
{
edge0.SetSize(9);
edge1.SetSize(9);
const int edge = edges[i];
const int d = edge_to_dim(i);
const int pkv = p*dim+d;
edge0[0] = 0; edge1[0] = 2;
edge0[1] = 0; edge1[1] = 4;
edge0[2] = 0; edge1[2] = 6;
edge0[3] = 1; edge1[3] = 3;
edge0[4] = 1; edge1[4] = 5;
edge0[5] = 1; edge1[5] = 7;
edge0[6] = 8; edge1[6] = 9;
edge0[7] = 8; edge1[7] = 10;
edge0[8] = 8; edge1[8] = 11;
flip = -1;
}
/* Initial assignment of knots to edges. This is an algorithm that loops over the
patches and assigns knot vectors to edges. It starts with assigning knot vector 0
and 1 to the edges of the first patch. Then it uses: 1) patches can share edges
2) knot vectors on opposing edges in a patch are equal, to create edge_to_knot */
int e0, e1, v0, v1, df;
int p,j,k;
for (p = 0; p < GetNE(); p++)
{
GetElementEdges(p, edges, oedge);
const int *v = elements[p]->GetVertices();
for (j = 0; j < edges.Size(); j++)
// We've set this edge already - link this index to it
if (edge_to_pkv[edge] != notset)
{
int *vv = edge_vertex->GetRow(edges[j]);
const int *e = elements[p]->GetEdgeVertices(j);
if (oedge[j] == 1)
{
vv[0] = v[e[0]];
vv[1] = v[e[1]];
}
else
{
vv[0] = v[e[1]];
vv[1] = v[e[0]];
}
const int pkv_other = unsign(edge_to_pkv[edge]);
unite(pkv, pkv_other);
}
for (j = 0; j < edge1.Size(); j++)
else
{
e0 = edges[edge0[j]];
e1 = edges[edge1[j]];
v0 = edge_to_knot[e0];
v1 = edge_to_knot[e1];
df = flip*oedge[edge0[j]]*oedge[edge1[j]];
// Case 1: knot vector is not set
if ((v0 == notset) && (v1 == notset))
{
edge_to_knot[e0] = knot;
edge_to_knot[e1] = knot;
knot++;
}
// Case 2 & 3: knot vector on one of the two edges
// is set earlier (in another patch). We just have
// to copy it for the opposing edge.
else if ((v0 != notset) && (v1 == notset))
{
edge_to_knot[e1] = (df >= 0 ? -v0-1 : v0);
}
else if ((v0 == notset) && (v1 != notset))
{
edge_to_knot[e0] = (df >= 0 ? -v1-1 : v1);
}
GetEdgeVertices(edge, v);
// Sign is based on the edge's vertex indices
edge_to_pkv[edge] = (v[1] > v[0]) ? pkv : sign(pkv);
}
}
}
/* Verify correct assignment, make sure that corresponding edges
within patch point to same knot vector. If not assign the lowest number.
// Construct the pkv_to_rpkv map by finding the lowest/root index
Array<int> pkv_to_rpkv(NPKV);
ukv_to_rpkv.SetSize(NPKV);
for (int i = 0; i < NPKV; i++)
{
pkv_to_rpkv[i] = get_root(pkv_map[i]);
ukv_to_rpkv[i] = pkv_to_rpkv[i];
}
ukv_to_rpkv.Sort(); // ukv is just a renumbering of rpkv
ukv_to_rpkv.Unique();
We bound the while by GetNE() + 1 as this is probably the most unlucky
case. +1 to finish without corrections. Note that this is a check and
in general the initial assignment is correct. Then the while is performed
only once. Only on very tricky meshes it might need corrections.*/
int corrections;
int passes = 0;
do
{
corrections = 0;
for (p = 0; p < GetNE(); p++)
{
GetElementEdges(p, edges, oedge);
for (j = 0; j < edge1.Size(); j++)
{
e0 = edges[edge0[j]];
e1 = edges[edge1[j]];
v0 = edge_to_knot[e0];
v1 = edge_to_knot[e1];
v0 = ( v0 >= 0 ? v0 : -v0-1);
v1 = ( v1 >= 0 ? v1 : -v1-1);
if (v0 != v1)
{
corrections++;
if (v0 < v1)
{
edge_to_knot[e1] = (oedge[edge1[j]] >= 0 ? v0 : -v0-1);
}
else if (v1 < v0)
{
edge_to_knot[e0] = (oedge[edge0[j]] >= 0 ? v1 : -v1-1);
}
}
}
}
// Create inverse map
std::map<int, int> rpkv_to_ukv;
for (int i = 0; i < ukv_to_rpkv.Size(); i++)
{
rpkv_to_ukv[ukv_to_rpkv[i]] = i;
}
passes++;
}
while (corrections > 0 && passes < GetNE() + 1);
// Check the validity of corrections applied
if (corrections > 0)
{
mfem::err<<"Edge_to_knot mapping potentially incorrect"<<endl;
mfem::err<<" passes = "<<passes<<endl;
mfem::err<<" corrections = "<<corrections<<endl;
}
/* Renumber knotvectors, such that:
-- numbering is consecutive
-- starts at zero */
Array<int> cnt(NumOfEdges);
cnt = 0;
for (j = 0; j < NumOfEdges; j++)
{
k = edge_to_knot[j];
cnt[(k >= 0 ? k : -k-1)]++;
}
k = 0;
for (j = 0; j < cnt.Size(); j++)
{
cnt[j] = (cnt[j] > 0 ? k++ : -1);
}
for (j = 0; j < NumOfEdges; j++)
{
k = edge_to_knot[j];
edge_to_knot[j] = (k >= 0 ? cnt[k]:-cnt[-k-1]-1);
}
// Print knot to edge mapping
mfem::out<<"Generated edge to knot mapping:"<<endl;
for (j = 0; j < NumOfEdges; j++)
{
int *v = edge_vertex->GetRow(j);
k = edge_to_knot[j];
v0 = v[0];
v1 = v[1];
if (k < 0)
{
v[0] = v1;
v[1] = v0;
}
mfem::out<<(k >= 0 ? k:-k-1)<<" "<< v[0] <<" "<<v[1]<<endl;
}
// Terminate here upon failure after printing to have an idea of edge_to_knot.
if (corrections > 0 ) {mfem_error("Mesh::LoadPatchTopo");}
// Get edge_to_ukv = edge_to_pkv -> pkv_to_rpkv -> rpkv_to_ukv
edge_to_ukv.SetSize(NumOfEdges);
for (int i = 0; i < NumOfEdges; i++)
{
const int pkv = unsign(edge_to_pkv[i]);
const int rpkv = pkv_to_rpkv[pkv];
const int ukv = rpkv_to_ukv[rpkv];
edge_to_ukv[i] = (edge_to_pkv[i] < 0) ? sign(ukv) : ukv;
}
}
+42 -9
View File
@@ -39,6 +39,7 @@ namespace mfem
class GeometricFactors;
class FaceGeometricFactors;
class KnotVector;
class NURBSPatch;
class NURBSExtension;
class FiniteElementSpace;
class GridFunction;
@@ -472,7 +473,7 @@ protected:
const int *fine, int nfine, int op);
/// Read NURBS patch/macro-element mesh
void LoadPatchTopo(std::istream &input, Array<int> &edge_to_knot);
void LoadPatchTopo(std::istream &input, Array<int> &edge_to_ukv);
void UpdateNURBS();
@@ -587,9 +588,10 @@ protected:
void Loader(std::istream &input, int generate_edges = 0,
std::string parse_tag = "");
/** If NURBS mesh, write NURBS format. If NCMesh, write mfem v1.1 format.
If section_delimiter is empty, write mfem v1.0 format. Otherwise, write
mfem v1.2 format with the given section_delimiter at the end.
/** @brief If NURBS mesh, write NURBS format. If NCMesh, write mfem v1.1
format. If section_delimiter is empty, write mfem v1.0 format. Otherwise,
write mfem v1.2 format with the given section_delimiter at the end.
If @a comments is non-empty, it will be printed after the first line of
the file, and each line should begin with '#'. */
void Printer(std::ostream &os = mfem::out,
@@ -789,6 +791,29 @@ public:
/// Destroys Mesh.
virtual ~Mesh() { DestroyPointers(); }
/** Get the edge to unique knotvector map used by NURBS patch topology meshes
Various index maps are defined using the following indices:
edge: Edge index in the patch topology mesh
pkv: Patch knotvector index, equivalent to (p * dim + d) where
p is the patch index, dim is the topological dimension of
the patch, and d is the local dimension
rpkv: Root patch knotvector index; the lowest index pkv for all
equivalent pkv.
ukv: (signed) Unique knotvector index. Equivalent to rpkv reordered
from 0 to N-1, where N is the number of unique knotvectors +
sign, which indicates the orientation of the edge.
@param[in,out] edge_to_ukv Array<int> Map from edge index to (signed)
unique knotvector index. Will be resized
to the number of edges.
@param[in,out] ukv_to_rpkv Array<int> Map from (unsigned) unique
knotvector index to the (unsigned) root
patch knotvector index. Will be resized
to the number of unique knotvectors.
*/
void GetEdgeToUniqueKnotvector(Array<int> &edge_to_ukv,
Array<int> &ukv_to_rpkv) const;
/// @}
/** @anchor mfem_Mesh_named_ctors @name Named mesh constructors.
@@ -1435,6 +1460,12 @@ public:
/// Set the attribute of patch boundary element i, for a NURBS mesh.
void SetPatchBdrAttribute(int i, int attr);
/** Returns a deep copy of all patches. This method is not const
as it first sets the patches in NURBSext using control points
defined by Nodes. Caller gets ownership of the returned object,
and is responsible for deletion.*/
void GetNURBSPatches(Array<NURBSPatch*> &patches);
/// Returns the type of element i.
Element::Type GetElementType(int i) const;
@@ -2452,10 +2483,12 @@ public:
/// Print the mesh to the given stream using Netgen/Truegrid format.
virtual void PrintXG(std::ostream &os = mfem::out) const;
/// Print the mesh to the given stream using the default MFEM mesh format.
/// \see mfem::ofgzstream() for on-the-fly compression of ascii outputs. If
/// @a comments is non-empty, it will be printed after the first line of the
/// file, and each line should begin with '#'.
/** @brief Print the mesh to the given stream using the default MFEM mesh
format.
\see mfem::ofgzstream() for on-the-fly compression of ascii outputs. If
@a comments is non-empty, it will be printed after the first line of the
file, and each line should begin with '#'. */
virtual void Print(std::ostream &os = mfem::out,
const std::string &comments = "") const
{ Printer(os, "", comments); }
@@ -2507,7 +2540,7 @@ public:
#ifdef MFEM_USE_NETCDF
/// @brief Export a mesh to an Exodus II file.
void PrintExodusII(const std::string fpath);
void PrintExodusII(const std::string &fpath);
#endif
/** @brief Prints the mesh with boundary elements given by the boundary of
+3 -13
View File
@@ -802,21 +802,11 @@ struct BufferReader : BufferReaderBase
{
// Each "data block" is preceded by a header that is either UInt32 or
// UInt64. The rest of the data follows.
uint64_t data_size;
if (header_type == UINT32_HEADER)
{
uint32_t *data_size_32 = (uint32_t *)header_buf;
data_size = *data_size_32;
}
else
{
uint64_t *data_size_64 = (uint64_t *)header_buf;
data_size = *data_size_64;
}
MFEM_VERIFY(sizeof(F)*n == data_size, "AppendedData: wrong data size");
MFEM_VERIFY(sizeof(F)*n == ReadHeaderEntry(header_buf),
"AppendedData: wrong data size");
}
if (std::is_same<T, F>::value)
if (std::is_same_v<T, F>)
{
// Special case: no type conversions necessary, so can just memcpy
memcpy(dest, buf, sizeof(T)*n);
+159 -86
View File
@@ -9,8 +9,13 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "mesh_headers.hpp"
#include "../fem/fem.hpp"
#include "nurbs.hpp"
#include "point.hpp"
#include "segment.hpp"
#include "quadrilateral.hpp"
#include "hexahedron.hpp"
#include "../fem/gridfunc.hpp"
#include "../general/text.hpp"
#include <fstream>
@@ -33,6 +38,7 @@ KnotVector::KnotVector(istream &input)
knot.Load(input, NumOfControlPoints + Order + 1);
GetElements();
coarse = false;
}
KnotVector::KnotVector(int order, int NCP)
@@ -41,12 +47,13 @@ KnotVector::KnotVector(int order, int NCP)
NumOfControlPoints = NCP;
knot.SetSize(NumOfControlPoints + Order + 1);
NumOfElements = 0;
coarse = false;
knot = -1.;
}
KnotVector::KnotVector(int order, const Vector& intervals,
const Array<int>& continuity )
const Array<int>& continuity)
{
// NOTE: This may need to be generalized to support periodicity
// in the future.
@@ -86,6 +93,7 @@ KnotVector::KnotVector(int order, const Vector& intervals,
++NumOfElements;
}
}
coarse = false;
}
KnotVector &KnotVector::operator=(const KnotVector &kv)
@@ -143,7 +151,7 @@ void KnotVector::UniformRefinement(Vector &newknots, int rf) const
{
for (int m = 1; m < rf; ++m)
{
newknots(j) = m * h * (knot(i) + knot(i+1));
newknots(j) = ((1.0 - (m * h)) * knot(i)) + (m * h * knot(i+1));
j++;
}
}
@@ -332,7 +340,7 @@ void KnotVector::PrintFunctions(std::ostream &os, int samples) const
}
}
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Algorithm A2.2 p. 70
void KnotVector::CalcShape(Vector &shape, int i, real_t xi) const
{
@@ -359,7 +367,7 @@ void KnotVector::CalcShape(Vector &shape, int i, real_t xi) const
}
}
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Algorithm A2.3 p. 72
void KnotVector::CalcDShape(Vector &grad, int i, real_t xi) const
{
@@ -417,7 +425,7 @@ void KnotVector::CalcDShape(Vector &grad, int i, real_t xi) const
}
}
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Algorithm A2.3 p. 72
void KnotVector::CalcDnShape(Vector &gradn, int n, int i, real_t xi) const
{
@@ -537,11 +545,11 @@ void KnotVector::FindMaxima(Array<int> &ks, Vector &xi, Vector &u) const
int i = j - d;
if (isElement(i))
{
arg1 = 1e-16;
arg1 = std::numeric_limits<real_t>::epsilon() / 2_r;
CalcShape(shape, i, arg1);
max1 = shape[d];
arg2 = 1-(1e-16);
arg2 = 1_r - arg1;
CalcShape(shape, i, arg2);
max2 = shape[d];
@@ -579,9 +587,9 @@ void KnotVector::FindMaxima(Array<int> &ks, Vector &xi, Vector &u) const
}
}
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Algorithm A9.1 p. 369
void KnotVector::FindInterpolant(Array<Vector*> &x)
void KnotVector::FindInterpolant(Array<Vector*> &x, bool reuse_inverse)
{
int order = GetOrder();
int ncp = GetNCP();
@@ -589,29 +597,93 @@ void KnotVector::FindInterpolant(Array<Vector*> &x)
// Find interpolation points
Vector xi_args, u_args;
Array<int> i_args;
FindMaxima(i_args,xi_args, u_args);
FindMaxima(i_args, xi_args, u_args);
// Assemble collocation matrix
Vector shape(order+1);
DenseMatrix A(ncp,ncp);
A = 0.0;
#ifdef MFEM_USE_LAPACK
// If using LAPACK, we use banded matrix storage (order + 1 nonzeros per row).
// Find banded structure of matrix.
int KL = 0; // Number of subdiagonals
int KU = 0; // Number of superdiagonals
for (int i = 0; i < ncp; i++)
{
CalcShape(shape, i_args[i], xi_args[i]);
for (int p = 0; p < order+1; p++)
{
A(i,i_args[i] + p) = shape[p];
const int col = i_args[i] + p;
if (col < i)
{
KL = std::max(KL, i - col);
}
else if (i < col)
{
KU = std::max(KU, col - i);
}
}
}
// Solve problems
A.Invert();
const int LDAB = (2*KL) + KU + 1;
const int N = ncp;
fact_AB.SetSize(LDAB, N);
#else
// Without LAPACK, we store and invert a DenseMatrix (inefficient).
if (!reuse_inverse)
{
A_coll_inv.SetSize(ncp, ncp);
A_coll_inv = 0.0;
}
#endif
Vector shape(order+1);
if (!reuse_inverse) // Set collocation matrix entries
{
for (int i = 0; i < ncp; i++)
{
CalcShape(shape, i_args[i], xi_args[i]);
for (int p = 0; p < order+1; p++)
{
const int j = i_args[i] + p;
#ifdef MFEM_USE_LAPACK
fact_AB(KL+KU+i-j,j) = shape[p];
#else
A_coll_inv(i,j) = shape[p];
#endif
}
}
}
// Solve the system
#ifdef MFEM_USE_LAPACK
const int NRHS = x.Size();
DenseMatrix B(N, NRHS);
for (int j=0; j<NRHS; ++j)
{
for (int i=0; i<N; ++i) { B(i, j) = (*x[j])[i]; }
}
if (reuse_inverse)
{
BandedFactorizedSolve(KL, KU, fact_AB, B, false, fact_ipiv);
}
else
{
BandedSolve(KL, KU, fact_AB, B, fact_ipiv);
}
for (int j=0; j<NRHS; ++j)
{
for (int i=0; i<N; ++i) { (*x[j])[i] = B(i, j); }
}
#else
if (!reuse_inverse) { A_coll_inv.Invert(); }
Vector tmp;
for (int i= 0; i < x.Size(); i++)
for (int i = 0; i < x.Size(); i++)
{
tmp = *x[i];
A.Mult(tmp,*x[i]);
A_coll_inv.Mult(tmp, *x[i]);
}
#endif
}
int KnotVector::findKnotSpan(real_t u) const
@@ -1413,7 +1485,7 @@ void NURBSPatch::DegreeElevate(int t)
}
}
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
void NURBSPatch::DegreeElevate(int dir, int t)
{
if (dir >= kv.Size() || dir < 0)
@@ -1431,8 +1503,8 @@ void NURBSPatch::DegreeElevate(int dir, int t)
KnotVector &oldkv = *kv[dir];
oldkv.GetElements();
NURBSPatch *newpatch = new NURBSPatch(this, dir, oldkv.GetOrder() + t,
oldkv.GetNCP() + oldkv.GetNE()*t);
auto *newpatch = new NURBSPatch(this, dir, oldkv.GetOrder() + t,
oldkv.GetNCP() + oldkv.GetNE()*t);
NURBSPatch &newp = *newpatch;
KnotVector &newkv = *newp.GetKV(dir);
@@ -1984,7 +2056,7 @@ NURBSExtension::NURBSExtension(const NURBSExtension &orig)
activeDof(orig.activeDof),
patchTopo(new Mesh(*orig.patchTopo)),
own_topo(true),
edge_to_knot(orig.edge_to_knot),
edge_to_ukv(orig.edge_to_ukv),
knotVectors(orig.knotVectors.Size()), // knotVectors are copied in the body
knotVectorsCompr(orig.knotVectorsCompr.Size()),
weights(orig.weights),
@@ -2025,7 +2097,7 @@ NURBSExtension::NURBSExtension(std::istream &input, bool spacing)
{
// Read topology
patchTopo = new Mesh;
patchTopo->LoadPatchTopo(input, edge_to_knot);
patchTopo->LoadPatchTopo(input, edge_to_ukv);
own_topo = true;
CheckPatches();
@@ -2227,7 +2299,7 @@ NURBSExtension::NURBSExtension(NURBSExtension *parent, int newOrder)
patchTopo = parent->patchTopo;
own_topo = false;
parent->edge_to_knot.Copy(edge_to_knot);
parent->edge_to_ukv.Copy(edge_to_ukv);
NumOfKnotVectors = parent->GetNKV();
knotVectors.SetSize(NumOfKnotVectors);
@@ -2285,7 +2357,7 @@ NURBSExtension::NURBSExtension(NURBSExtension *parent,
patchTopo = parent->patchTopo;
own_topo = false;
parent->edge_to_knot.Copy(edge_to_knot);
parent->edge_to_ukv.Copy(edge_to_ukv);
NumOfKnotVectors = parent->GetNKV();
MFEM_VERIFY(mOrders.Size() == NumOfKnotVectors, "invalid newOrders array");
@@ -2344,7 +2416,7 @@ NURBSExtension::NURBSExtension(Mesh *mesh_array[], int num_pieces)
own_topo = true;
parent->own_topo = false;
parent->edge_to_knot.Copy(edge_to_knot);
parent->edge_to_ukv.Copy(edge_to_ukv);
parent->GetOrders().Copy(mOrders);
mOrder = parent->GetOrder();
@@ -2377,70 +2449,61 @@ NURBSExtension::NURBSExtension(Mesh *mesh_array[], int num_pieces)
}
NURBSExtension::NURBSExtension(const Mesh *patch_topology,
const Array<const NURBSPatch*> p)
const Array<const NURBSPatch*> &patches_)
{
// Basic topology checks
MFEM_VERIFY(patches_.Size() > 0, "Must have at least one patch");
MFEM_VERIFY(patches_.Size() == patch_topology->GetNE(),
"Number of patches must equal number of elements in patch_topology");
// Copy patch_topology mesh and NURBSPatch(es)
patchTopo = new Mesh( *patch_topology );
patchTopo->GetEdgeVertexTable();
own_topo = 1;
patches.Reserve(p.Size());
Array<int> edges;
Array<int> oedges;
Array<int> kvs(3);
edge_to_knot.SetSize(patch_topology->GetNEdges());
NumOfKnotVectors = 0;
NumOfElements = 0;
for (int ielem = 0; ielem < patch_topology->GetNE(); ++ielem)
patches.SetSize(patches_.Size());
for (int p = 0; p < patches.Size(); p++)
{
patches.Append(new NURBSPatch(*p[ielem]));
NURBSPatch& patch = *patches[ielem];
int num_patch_elems = 1;
for (int ikv = 0; ikv < patch.GetNKV(); ++ikv)
{
kvs[ikv] = knotVectors.Size();
knotVectors.Append(new KnotVector(*patch.GetKV(ikv)));
num_patch_elems *= patch.GetKV(ikv)->GetNE();
++NumOfKnotVectors;
}
NumOfElements += num_patch_elems;
patch_topology->GetElementEdges(ielem, edges, oedges);
for (int iedge = 0; iedge < edges.Size(); ++iedge)
{
if (iedge < 8)
{
if (iedge & 1)
{
edge_to_knot[edges[iedge]] = kvs[1];
}
else
{
edge_to_knot[edges[iedge]] = kvs[0];
}
}
else
{
edge_to_knot[edges[iedge]] = kvs[2];
}
}
patches[p] = new NURBSPatch(*patches_[p]);
}
GenerateOffsets();
CountBdrElements();
NumOfActiveElems = NumOfElements;
activeElem.SetSize(NumOfElements);
activeElem = true;
Array<int> ukv_to_rpkv;
patchTopo->GetEdgeToUniqueKnotvector(edge_to_ukv, ukv_to_rpkv);
own_topo = true;
CheckPatches(); // This is checking the edge_to_ukv mapping
// Set number of unique (not comprehensive) knot vectors
NumOfKnotVectors = ukv_to_rpkv.Size();
knotVectors.SetSize(NumOfKnotVectors);
knotVectors = NULL;
// Assign the unique knot vectors from patches
for (int i = 0; i < NumOfKnotVectors; i++)
{
// pkv = p*dim + d for an arbitrarily chosen patch p,
// in its reference direction d
const int pkv = ukv_to_rpkv[i];
const int p = pkv / Dimension();
const int d = pkv % Dimension();
knotVectors[i] = new KnotVector(*patches[p]->GetKV(d));
}
CreateComprehensiveKV();
SetOrdersFromKnotVectors();
GenerateOffsets();
CountElements();
CountBdrElements();
NumOfActiveElems = NumOfElements;
activeElem.SetSize(NumOfElements);
activeElem = true;
GenerateActiveVertices();
InitDofMap();
GenerateElementDofTable();
GenerateActiveBdrElems();
GenerateBdrElementDofTable();
weights.SetSize(GetNDof());
CheckPatches();
ConnectBoundaries();
}
NURBSExtension::~NURBSExtension()
@@ -2481,7 +2544,7 @@ void NURBSExtension::Print(std::ostream &os, const std::string &comments) const
}
const int version = kvSpacing.Size() > 0 ? 11 : 10; // v1.0 or v1.1
patchTopo->PrintTopo(os, edge_to_knot, version, comments);
patchTopo->PrintTopo(os, edge_to_ukv, version, comments);
if (patches.Size() == 0)
{
os << "\nknotvectors\n" << NumOfKnotVectors << '\n';
@@ -2936,7 +2999,7 @@ void NURBSExtension::CheckPatches()
for (int i = 0; i < edges.Size(); i++)
{
edges[i] = edge_to_knot[edges[i]];
edges[i] = edge_to_ukv[edges[i]];
if (oedge[i] < 0)
{
edges[i] = -1 - edges[i];
@@ -2954,7 +3017,7 @@ void NURBSExtension::CheckPatches()
edges[8] != edges[11])))
{
mfem::err << "NURBSExtension::CheckPatch (patch = " << p
<< ")\n Inconsistent edge-to-knot mapping!\n";
<< ")\n Inconsistent edge-to-knotvector mapping!";
mfem_error();
}
}
@@ -2971,7 +3034,7 @@ void NURBSExtension::CheckBdrPatches()
for (int i = 0; i < edges.Size(); i++)
{
edges[i] = edge_to_knot[edges[i]];
edges[i] = edge_to_ukv[edges[i]];
if (oedge[i] < 0)
{
edges[i] = -1 - edges[i];
@@ -4596,7 +4659,7 @@ void NURBSExtension::KnotInsert(Array<Vector *> &kv)
// Flip vector
int size = pkvc[d]->Size();
int ns = ceil(size/2.0);
int ns = static_cast<int>(ceil(size/2.0));
for (int j = 0; j < ns; j++)
{
real_t tmp = apb - pkvc[d]->Elem(j);
@@ -4656,7 +4719,7 @@ void NURBSExtension::KnotRemove(Array<Vector *> &kv, real_t tol)
// Flip vector
int size = pkvc[d]->Size();
int ns = ceil(size/2.0);
int ns = static_cast<int>(ceil(size/2.0));
for (int j = 0; j < ns; j++)
{
real_t tmp = apb - pkvc[d]->Elem(j);
@@ -4878,6 +4941,16 @@ void NURBSExtension::GetElementIJK(int elem, Array<int> & ijk)
el_to_IJK.GetRow(elem, ijk);
}
void NURBSExtension::GetPatches(Array<NURBSPatch*> &patches_copy)
{
const int NP = patches.Size();
patches_copy.SetSize(NP);
for (int p = 0; p < NP; p++)
{
patches_copy[p] = new NURBSPatch(*GetPatch(p));
}
}
void NURBSExtension::SetPatchToElements()
{
const int np = GetNP();
@@ -4982,7 +5055,7 @@ ParNURBSExtension::ParNURBSExtension(MPI_Comm comm, NURBSExtension *parent,
own_topo = true;
parent->own_topo = false;
parent->edge_to_knot.Copy(edge_to_knot);
parent->edge_to_ukv.Copy(edge_to_ukv);
parent->GetOrders().Copy(mOrders);
mOrder = parent->GetOrder();
@@ -5045,7 +5118,7 @@ ParNURBSExtension::ParNURBSExtension(NURBSExtension *parent,
own_topo = parent->own_topo;
parent->own_topo = false;
Swap(edge_to_knot, parent->edge_to_knot);
Swap(edge_to_ukv, parent->edge_to_ukv);
NumOfKnotVectors = parent->NumOfKnotVectors;
Swap(knotVectors, parent->knotVectors);

Some files were not shown because too many files have changed in this diff Show More