Compare commits

...
Author SHA1 Message Date
Ido Akkerman 62682fa0f2 Switch schur approximation to mass matrix 2025-12-03 11:08:08 +01:00
Ido Akkerman 4d59f9de54 Remove comments 2025-12-03 10:24:38 +01:00
Ido Akkerman c3f6ecbd0b Merge remote-tracking branch 'origin/master' into add-vfe-grad 2025-12-03 09:18:01 +01:00
Tzanio Kolev 9708cc522e Merge pull request #5135 from mfem/gitlab-ci-extra-time
Gitlab CI: allocate more time for the dane-baseline pipeline
2025-12-02 12:49:54 -08:00
Tzanio Kolev a546e320fd Merge pull request #4996 from mfem/contact-miniapp
Contact miniapp
2025-12-02 11:28:18 -08:00
Veselin Dobrev fbcd9cf4d3 In Gitlab CI, change the allocation time for 'dane-baseline' to 1 hour
The previous commit had the change in the wrong place.
2025-12-02 09:56:59 -08:00
Veselin Dobrev f60799af0c In Gitlab CI, increase the time allocation for dane-baseline to 1 hour 2025-12-02 08:04:47 -08:00
Tzanio Kolev 74ff518840 Merge pull request #4840 from mfem/qf-project-gf-fallback
Add fallback for QuadratureFunction::ProjectGridFunction
2025-12-01 20:03:00 -08:00
Tzanio Kolev 4492be9ebb Merge pull request #4028 from mfem/table-array
Use Array<int> instead of Memory<int> in Table
2025-12-01 20:02:00 -08:00
Tzanio Kolev b62e215620 Merge pull request #5121 from tomstitt/claude/review-integ-kernels-01LSxNDE62DLqsm9LRtXfbcJ
Fix array shape in `SmemPADiffusionDiagonal2D`
2025-12-01 19:59:41 -08:00
Tzanio Kolev 8497c5129b Merge pull request #5127 from mfem/fix-clang-cuda
Fix CLANG + CUDA builds
2025-12-01 19:59:19 -08:00
Tzanio Kolev 12cda5ec86 Merge pull request #5113 from mfem/mfem-bounds-const
Fix const-ness for bounding related methods
2025-12-01 19:58:52 -08:00
Ido Akkerman 44995d0f98 Switch back to CG + adding a clean tau definition 2025-12-01 16:30:53 +01:00
Veselin Dobrev d90a23f195 Fix the build with PUMI: use 'std::swap' instead of just 'swap'.
In class DenseSymmetricMatrix:
* Fix a warning from -Wextra about implicit copy-ctor.
* Explicitly use defaulted copy/move ctor/assignment.
* A small fix in the SetSize method.
* Rewrap doxygen comments to 80 chars
2025-11-26 15:32:23 -08:00
Tzanio Kolev 0c5fa5c148 Merge branch 'master' into contact-miniapp 2025-11-25 19:50:00 -08:00
Will Pazner 838dd47259 TMOP W caching in new files 2025-11-25 14:00:31 -08:00
Will Pazner c78863a437 Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	fem/tmop/tmop_pa_da3.cpp
#	fem/tmop/tmop_pa_tc2.cpp
#	fem/tmop/tmop_pa_tc3.cpp
#	general/array.hpp
#	linalg/densemat.hpp
2025-11-25 14:00:20 -08:00
Ketan Mittal 2985487736 Merge branch 'master' into mfem-bounds-const 2025-11-25 13:25:11 -08:00
Tzanio Kolev 2caa75e35a Merge pull request #5072 from mfem/mtop-gpu
[MTOP] solver with GPU support
2025-11-25 12:16:20 -08:00
Tzanio Kolev adf110b3a7 Merge pull request #4981 from mfem/multivector-dev
`ParticleVector`
2025-11-25 12:15:53 -08:00
Tzanio Kolev 46fd4bdb93 Merge pull request #5117 from mfem/mexp-nurbs-m
Fix mesh-explorer material view for 3D NURBS meshes
2025-11-25 12:10:55 -08:00
John Camier ab8080f413 Merge branch 'master' into claude/review-integ-kernels-01LSxNDE62DLqsm9LRtXfbcJ 2025-11-25 10:38:47 -06:00
camierjs 7f27e4605e Use MFEM_ABORT_KERNEL instead 2025-11-25 08:21:08 -08:00
camierjs dbab49b725 Avoid taking address of pure virtual function 2025-11-24 17:22:51 -08:00
Mittal, Ketan 916e23ddc4 formatting 2025-11-24 09:23:33 -08:00
Mittal, Ketan d45bdbcea2 update CHANGELOG 2025-11-24 09:16:57 -08:00
Mittal, Ketan 9d9c7387a7 Merge branch 'master' of https://github.com/mfem/mfem into multivector-dev 2025-11-24 09:10:14 -08:00
Tzanio Kolev 266ff0cdf6 Merge branch 'master' into contact-miniapp 2025-11-23 17:52:07 -08:00
Tzanio Kolev fa8c861b75 Merge pull request #5022 from mfem/dfem-assemble-matrix
dFEM assemble implementations
2025-11-21 10:39:17 -08:00
Socratis Petrides 5453c63c91 cmakelist fix 2025-11-21 09:52:16 -08:00
Ido Akkerman 8bd806147c Corrected comments 2025-11-21 15:56:57 +01:00
Claude 5db9a79022 Fix smem array dimension bug in SmemPADiffusionDiagonal2D
Fixed array dimension mismatch similar to the bugs fixed in PRs #5108
and #4931. The QD array was declared as [MD1][MQ1] but accessed as
[Q][D], causing incorrect memory layout when MQ1 > MD1.

Changed:
  MFEM_SHARED real_t QD[3][NBZ][MD1][MQ1];
To:
  MFEM_SHARED real_t QD[3][NBZ][MQ1][MD1];

This matches the access pattern QD0[qx][dy] where qx < Q1D and dy < D1D,
and the pointer cast real_t (*)[MD1] which expects the second dimension
to be MD1.
2025-11-19 23:00:05 +00:00
camierjs e1edcb321d Fix nvcc access errors and VectorCoefficient partially overridden warnings 2025-11-19 12:14:42 -06:00
John Camier c0d9732d1b Merge branch 'master' into mtop-gpu 2025-11-19 09:45:55 -08:00
Julian Andrej 354859c91d size_t -> int for map key 2025-11-18 08:29:48 -08:00
Socratis Petrides 6bc2b2b5aa another doc fix 2025-11-17 19:53:55 -08:00
Socratis Petrides dd2e531218 fix doxygen comment 2025-11-17 19:12:31 -08:00
Socratis Petrides 769c396613 fixing comments in ip 2025-11-17 18:59:11 -08:00
Socratis Petrides b7a5940f00 forgotten makefile changes 2025-11-17 18:55:03 -08:00
Socratis Petrides cab5e5f99c added the new direct solver interface to contact driver 2025-11-17 18:53:13 -08:00
Socratis Petrides 0226a21a76 added generic direct solver interf 2025-11-17 18:52:32 -08:00
Socratis Petrides 9302c4229a remove leftover comment 2025-11-17 15:12:16 -08:00
Dylan Copeland 096605f532 Revert empty line. 2025-11-17 15:03:29 -08:00
Dylan Copeland 69b4689048 Fix a bug so that "m" option works for 3D NURBS meshes. 2025-11-17 14:58:21 -08:00
Julian Andrej c3ee799e24 use fold expression 2025-11-17 13:57:57 -08:00
Julian Andrej 8056ba5610 try again 2025-11-17 13:49:22 -08:00
Julian Andrej 7dd05fe015 guard against MSVC N=0 case 2025-11-17 12:58:08 -08:00
Socratis Petrides 861cb9cb6b Merge branch 'master' into contact-miniapp 2025-11-17 11:59:29 -08:00
Socratis Petrides 458254a1f6 set relax_type depending on Hypre version 2025-11-17 11:59:00 -08:00
Socratis Petrides 4defb5c401 removing no longer needed c++ standard spec 2025-11-17 11:58:09 -08:00
Mittal, Ketan aecd9648af merge with master and resolve conflicts 2025-11-17 11:37:43 -08:00
Mittal, Ketan 1e592a7442 change reordering logic 2025-11-17 11:35:47 -08:00
Julian Andrej 78608e4fcd remove prints 2025-11-17 11:09:15 -08:00
Julian Andrej d0f4f9baaf deduction simplification 2025-11-17 11:07:04 -08:00
Joseph Signorelli bc7ea43317 Move ordering to linalg 2025-11-17 12:17:53 -06:00
Tzanio Kolev 9cc28d8772 Merge branch 'master' into mfem-bounds-const 2025-11-17 05:25:06 -08:00
Tzanio Kolev 1877a40a85 Merge pull request #4995 from mfem/amgf
AMGF Solver
2025-11-17 04:45:34 -08:00
Tzanio Kolev 99785763af Merge pull request #5078 from mfem/user-inner-prod
User-Defined Inner Products for IterativeSolvers
2025-11-17 04:41:45 -08:00
Ido Akkerman 7f95f32b32 Add tests to makefile 2025-11-17 12:58:41 +01:00
Ido Akkerman c2805439e4 Force no vis 2025-11-17 12:49:21 +01:00
Ido Akkerman 05feb66563 Fix typo 2025-11-17 11:42:27 +01:00
Ido Akkerman 677ece5024 Add test cases 2025-11-17 11:24:58 +01:00
Ido Akkerman 6d91303675 Add manufactured solution. 2025-11-17 11:24:39 +01:00
Ido Akkerman 10ab241b3c Add manufactured solution. 2025-11-17 11:22:15 +01:00
Tzanio Kolev e3b4edcae1 Merge pull request #5114 from mfem/dof-table-mem-leak-fix
Fix memory leak in `TensorBasisElement::GetTensorDofToQuad`
2025-11-15 10:02:33 -08:00
Tzanio Kolev 34307512d5 Merge pull request #5108 from mfem/fix-smem-pa-diffusion-apply-2d-indexing
Fix indexing in `SmemPADiffusionApply2D`
2025-11-15 10:01:53 -08:00
Vladimir Z Tomov cef3517d60 updated contact/README with more details for the build. 2025-11-14 18:00:22 -08:00
Socratis Petrides 555d5771a6 Merge branch 'amgf' into contact-miniapp 2025-11-14 09:55:00 -08:00
Socratis Petrides e1d4c6cd40 fix petsc include 2025-11-14 09:54:34 -08:00
Socratis Petrides 147204ee70 readme edits 2025-11-13 17:28:15 -08:00
Socratis Petrides 4b8832190d cmake fixes 2025-11-13 17:23:55 -08:00
Socratis Petrides 69159b13a6 Merge branch 'amgf' into contact-miniapp 2025-11-13 16:54:05 -08:00
Ketan Mittal edbdf0628d Merge branch 'master' into mfem-bounds-const 2025-11-13 14:16:53 -08:00
Tzanio Kolev 7a9253d13c Merge branch 'master' into dfem-assemble-matrix 2025-11-13 13:18:20 -08:00
Victor DeCaria 1f8f2f1826 slight optimization 2025-11-13 12:55:05 -07:00
Victor DeCaria 1c9b9beb1e change logic for storing dof2quad tables 2025-11-13 12:44:57 -07:00
John Camier 20dcc1f991 Merge branch 'master' into mtop-gpu 2025-11-13 11:03:48 -08:00
Socratis Petrides 13252e4734 Fix formatting in CHANGELOG 2025-11-13 11:00:14 -08:00
Julian Andrej 53a5c2f687 remove unused variables 2025-11-13 10:40:27 -08:00
Julian Andrej 6b279d6e32 initialize qp only once each thread 2025-11-13 10:22:21 -08:00
John Camier 032bf0f3dd Merge branch 'master' into fix-smem-pa-diffusion-apply-2d-indexing 2025-11-13 09:36:46 -08:00
John CamiercamierjsVladimir Z TomovMittal, Ketan <mittal3@llnl.gov>
50368046bc [TMOP] Simplify kernels (#3658)
* Simplify TMOP kernels, fix unit tests to run --all tests with adjusted tolerance

* make style

* Split TMOP h3s file with metrics

* TMOP kernel MFEM_HOST_DEVICE fix

* Cleanup TMOP CUDA kernels from base class

* Added TMOP PA metrics directory

* meld toward master

* [tmop] struct to class friends

* Simplify tmop file names

* make style

* Cleanup

* Style and vscode gitignore

* WIP resolve conflicts

* 2024 headers

* Tmop pass

* All tmop tests

* make style

* Add astyle to clang format

* make style

* Fix class visibility

* Include cleanup

* real_t pass

* style

* MFEM_REGISTER_KERNELS for TMOPAssembleGradPA_001

* make style

* add config files

* Update config

* metric_t

* wip with T

* wip

* wip T Specialization

* c++20, fmt make_format_args

* print types and values

* wip Kernel<decltype(M)>

* wip

* Working with metric_t, int, int

* C++20 ok

* C++17 cleaned

* Rename tmop files

* Sync TMOP kernels with dispatch

* make style

* Cleanup metrics

* Use TMOPKernel

* 3D metrics standalone

* Chdir assemble

* tmop 2d/3d directories

* TMOP assemble using specializations

* All TMOP kernel specializations

* MFEM_REPORT_KERNELS

* make style

* Sync with master

* Sync with master

* make style

* make style

* Removed 2d/3d TMOP sub-directories

* CMake TMOP file list update

* makefile directories order

* With style

* Re-enable vscode gitignore

* Fix merge conflicts

* make style

* Sync

* Meld toward master

* Changes toward master

* make style

* Meld back fem tmop files

* Fix TMOP_Integrator friends

* PA tests fix & history bump

* Cleanup test tmop and fix energy2 metric data

* Update copyright 2010-2025

* 2D energy metrics

* 3D energy metrics

* make style

* Simplify metric registration

* TMOP fem kernels with double buffering

* grad3, grad3_coef

* grad3_coef, grad3, mult3_coefs, mult3

* TMOP sm kernels tools

* Rename kernels smem and use regs

* Grad3 w/ vector reg grad

* Kernel register cleanup

* Add MAX_TMOP_1D and HIP tmop ctests

* Add kernels_foreach

* Add kernels foreach

* Prefix foreach_thread

* Kernels regs w/ foreach threads

* Swap Y and X in forward only

* Backward kernels_regs

* Use simplified grad3d

* Wip D1D Q1D

* Runtime D1D Q1D

* Remove T1D

* Cleanup

* AddKernelSpecializations

* Sync with SetMaxOf

* Rename to LoadDofs and use deduced templated parameters

* Grad2d & factorization

* Eval3d for grad3 coef

* Eval2d for grad2 coef

* Cleanup TMOP_SetupGradPA_C0_2D

* Use Bld and B

* Use other accessors

* TMOPAddMultPA3D

* TMOP_AddMultPA_C0_2D

* TMOP_AddMultGradPA_3D

* TMOP_AddMultGradPA_2D

* TMOP_AddMultGradPA_C0_3D

* AddMultGradPA_C0_2D

* TMOP_AssembleDiagonalPA_2D

* Wip TMOP_AssembleDiagonalPA_C0_3D

* TMOP_AssembleDiagonalPA_3D

* TMOP_MinDetJpr_3D

* TMOP_EnergyPA_C0_2D

* TMOPEnergyPA3D

* TMOP_TcIdealShapeGivenSize_3D

* TMOP_DatcSize_3D

* Remove MAX_TMOP_1D

* Remove smem kernels

* TMOP cleanup

* TMOP - solve for displacements #4694 changes

* Cleanup and move verifications

* Rename TMOP Assemble kernels

* Move kernel regs to TMOP pa

* make style

* Meld back toward master

* Meld back to master

* Use static constexpr

* Temporary branch-history

* Help msvc with namespaces

* MSVC inner static constexpr

* Move regs to mfem namespace

* MSVC all static constexpr

* TMOP_AssembleDiagPA_C0_3D w/o regs

* Avoid set but unused variable

* MSVC TMOP_AssembleDiagPA_C0_3D ternary test try

* MSVC MFEM_TMOP_REGISTER_MDQ_KERNEL

* Switch to MFEM_TMOP_MDQ_REGISTER

* MSVC help with static constexpr

* MSVC conversions try

* MSVC as_regs2d_ref

* MSVC Explicitly bind as reference

* MSCV with reinterpret_cast

* MSVC avoiding required l-values

* MSVC avoid explicit ref bindings

* MSVC avoid explicit ref bindings 2D

* Cleanup

* Enable MFEM_TMOP_PA_DEVICE with makefile

* TMOP tests w/o Kernel Specializations

* TMOP re-enable kernels specializations

* TMOP PA tests tolerances

* TMOP tests adjustments

* Fix transposed eval regs access

* MSVC remove not allowed dllimport definitions

* MSVC linalg vector warning fix

* MSVC avoiding definition of dllimport function not allowed

* Re-enable DetKernels specializations

* Sync latest TMOP changes

* TMOP PA tests normalization wip

* Sync TMOP tests

* Remove debug file

* Meld back toward master

* Add missing tmop make source dir

* tmop shadowing, CMake & make mpi tests

* TMOP periodic tests, shadowing fix

* TMOP pa mpi tests, fix shadowing

* TMOP tighten Square01 + Combo tests

* TMOP MSVC include ordering

* Revert TMOP MPI debug device tests

* Add TMOP_DatcSize_2D

* Use mfem::future for tensor

* Move TMOP PA specific kernels to sync'ed fem kernels

* makefile source dirs fix

* use explicit namespace to avoid clash (swap)

* Revert to MFEM_FOREACH_THREAD
Use scalar/vector regs types

* Sync kernels

* Sync kernels

* Avoid applying non-zero offset to null pointer runtime error

* Remove debug include

* TMOP rename coef to limit

* Comments.

* minor

* changelog

* Replace TMOP's MFEM_FOREACH_THREAD with MFEM_FOREACH_THREAD_DIRECT

* add some missing metric IDs

* Revert branch-history

* Add missing MFEM_SYNC_THREAD in kernels
Verify TMOP isfinite energy

* UseDevice for local vectors

* make style

* No grids in TMOP_DatcSize kernels

* Remove isfinite assertions
Cleanup unused header files
Add 3D energy finite verifications

* Filter out TMOP PA tests

---------

Co-authored-by: camierjs <camierjs@Io>
Co-authored-by: Vladimir Z Tomov <tomov2@llnl.gov>
Co-authored-by: Mittal, Ketan <mittal3@llnl.gov>
2025-11-13 08:47:32 -08:00
Ido Akkerman 30f18c02c8 Correct silly mistakes 2025-11-13 14:33:41 +01:00
Ido Akkerman 6a9e01507e Change from std::cout to cout 2025-11-13 13:57:03 +01:00
Ido Akkerman 8a0f382ccf Restore lininteg function 2025-11-13 13:56:14 +01:00
Ido Akkerman 2e552eec91 Add comments to VectorDG class 2025-11-13 12:26:55 +01:00
Ido Akkerman b145001e80 Corrected small typos, comments, const keywords and removed 2 superflous functions 2025-11-13 12:11:10 +01:00
Ido Akkerman fe7e5bbd8d Add H1 capability -- forgot to copy from defunct PR 2025-11-13 12:10:26 +01:00
Mittal, Ketan 05d5294709 fix const-ness for bounding related methods 2025-11-12 23:09:44 -08:00
Julian Andrej 5ed263e9b4 try threadblocks 2025-11-12 16:56:36 -08:00
camierjs 5450915b0b CHANGELOG mtop miniapp entry 2025-11-12 10:31:08 -08:00
camierjs 06d6bdbad0 Merge branch 'mtop-gpu' of github.com:mfem/mfem into mtop-gpu 2025-11-12 09:49:38 -08:00
camierjs c2955a8d52 Rename to GetSolutionVector & GetAdjointSolutionVector 2025-11-12 09:49:13 -08:00
John Camier 5b1de08790 Merge branch 'master' into mtop-gpu 2025-11-11 20:44:47 -08:00
Tzanio Kolev 19733980de Merge pull request #4692 from mfem/fix-issue-4455
Fix Mesh::MakeSimplicial for surface in 3D
2025-11-11 17:36:02 -08:00
Will Pazner 7e98c14f9b Fix copy and move semantics in DenseSymmetricMatrix 2025-11-11 11:28:20 -08:00
Julian Andrej 4528a608ea Merge branch 'master' into dfem-assemble-matrix 2025-11-11 11:24:58 -08:00
Socratis PetridesandWill Pazner 6f8d55d35f MFEM_VERIFY
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-11-11 09:59:35 -08:00
Tom Stitt 459f93d71c flip indexing for DQ0/DQ1 so we don't overrun xy-block smem when MQ1>MD1 2025-11-11 09:53:13 -08:00
Ketan Mittal 71699fe790 Merge branch 'master' into multivector-dev 2025-11-10 18:09:47 -08:00
Socratis Petrides 25bf3e0235 style 2025-11-10 16:32:08 -08:00
Socratis Petrides 763ea10fd1 remove deprecated warning 2025-11-10 16:31:45 -08:00
Socratis Petrides 4439841f79 adjusting the contact driverr to the amgf changes 2025-11-10 16:28:53 -08:00
Socratis Petrides d834fd8cb3 Merge branch 'amgf' into contact-miniapp 2025-11-10 16:11:15 -08:00
Socratis Petrides 7f79fc8adc reviewers comments 2025-11-10 16:11:03 -08:00
Socratis Petrides 1b2ad14923 Merge branch 'amgf' into contact-miniapp 2025-11-10 14:45:40 -08:00
Socratis Petrides a6c6d63aa6 fix doxygen complaint 2025-11-10 14:45:24 -08:00
Socratis Petrides 95fc0b414f fix readme line length 2025-11-10 14:44:24 -08:00
Socratis Petrides 4f0b12acd6 Merge branch 'amgf' into contact-miniapp 2025-11-10 14:32:02 -08:00
Socratis Petrides 56689d3c0e fixing changelog line length 2025-11-10 14:31:44 -08:00
Socratis Petrides 04453a12c7 merge with amgf 2025-11-10 13:27:51 -08:00
Socratis Petrides 4a5f5f8528 Merge branch 'master' into amgf 2025-11-10 13:18:18 -08:00
Socratis Petrides e30c8b53d8 fixing comments 2025-11-10 13:17:55 -08:00
camierjs b497159d50 Remove used private field 2025-11-10 13:00:02 -08:00
camierjs 60e0bef7d3 Merge branch 'master' into mtop-gpu 2025-11-10 12:05:06 -08:00
camierjs f6d6148108 Fix comments 2025-11-10 12:04:58 -08:00
camierjs 7cda588460 Remove unused declarations 2025-11-10 12:03:52 -08:00
Tzanio Kolev 73afff37cc Merge pull request #5099 from farscape-project/restricted
Add {ND,RT}_R{1,2}D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] to FEC factory
2025-11-10 12:01:54 -08:00
camierjs c8d2c38b57 Merge branch 'master' into mtop-gpu 2025-11-10 08:31:17 -08:00
camierjs 2a24edf180 Use '-1' for 'all' BCs, add verifications
Add sample runs in mtop example
Remove unused code in mtop example
2025-11-10 08:31:01 -08:00
Julian Andrej 555ce60a73 style 2025-11-10 08:00:57 -08:00
Julian Andrej 76de06bf32 remove vectordivergence from gpu tests 2025-11-10 07:57:57 -08:00
Ido Akkerman 49cadecc82 Switched to CG solver 2025-11-10 13:24:43 +01:00
Ido Akkerman 62934ed7ac Improved unit test 2025-11-10 13:24:15 +01:00
Ido Akkerman b641c0c997 Corrected some comments 2025-11-10 13:23:43 +01:00
Tzanio Kolev 087a29249a Merge pull request #5096 from mfem/guard-raja-gpu-in-forall
raja+gpu forall guard
2025-11-09 14:09:11 -08:00
John Camier a5fd83b1fb Merge branch 'master' into mtop-gpu 2025-11-09 07:36:56 -08:00
Tzanio Kolev f2e841d7f9 Merge pull request #5084 from mfem/complex-gf-save-dev
Adding [Par]ComplexGridFuction::Save
2025-11-08 16:54:05 -08:00
Tzanio Kolev dcff876e94 Merge pull request #4979 from mfem/array-vector-improvements-dev
Improvements to Vector + Array
2025-11-08 16:53:41 -08:00
Will Pazner 4d042959e8 Add convenience function Mesh::IsMixedMesh 2025-11-08 14:03:37 -08:00
Tzanio Kolev 19e4c53acd Merge branch 'master' into qf-project-gf-fallback 2025-11-07 13:34:56 -08:00
Tzanio Kolev 0555b6cc97 Merge pull request #4333 from mfem/fix-caliper-scope
Fix `MFEM_PERF_SCOPE` with latest Caliper
2025-11-06 06:25:45 -08:00
Tzanio Kolev ea329bca35 Merge pull request #5088 from mfem/fix-boris-dev
Fix Boris rotation term using pre-rotation momentum [fix-boris-dev]
2025-11-05 09:54:33 -08:00
Tzanio Kolev 78cd3d3838 Merge branch 'master' into multivector-dev 2025-11-05 09:53:34 -08:00
John Camier 2726959a2d Merge branch 'master' into fix-caliper-scope 2025-11-05 09:06:37 -08:00
Nuno Nobre 2fef8ca1f0 Add {ND,RT}_R{1,2}D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] to FEC factory 2025-11-05 15:27:18 +00:00
Joseph Signorelli 7da0f4c226 TestResize --> TestSetNumParticles for clarity 2025-11-05 08:24:08 -06:00
Joseph Signorelli 3681b6f6a2 Merge branch 'multivector-dev' of github.com:mfem/mfem into multivector-dev 2025-11-05 08:19:08 -06:00
Joseph Signorelli 021bf157a7 Documentation updates 2025-11-05 08:16:16 -06:00
Joseph Signorelli 0463749eb9 update_data --> keep_data 2025-11-05 08:16:07 -06:00
Mittal, Ketan b8443bf867 formatting fixes 2025-11-04 17:09:54 -08:00
Mittal, Ketan 793fd42653 minor 2025-11-04 16:59:11 -08:00
Joseph Signorelli 7e6034c4e2 potential fix to mem leak? 2025-11-04 18:10:00 -06:00
Mark L. Stowell 0f5b6cc0f5 Merge pull request #3525 from mfem/sjg/mesh-vis-dev
Mesh and DataCollection upgrades
2025-11-04 14:58:15 -08:00
Mittal, Ketan db4f95d252 Merge branch 'multivector-dev' of https://github.com/mfem/mfem into multivector-dev 2025-11-04 14:45:21 -08:00
Mittal, Ketan 40bed3db8f make style 2025-11-04 14:45:06 -08:00
Ketan Mittal 2f6c8b720c Merge branch 'master' into multivector-dev 2025-11-04 14:43:25 -08:00
Mittal, Ketan 44b8778af1 add update data to SetNumParticles and GrowSize 2025-11-04 14:42:43 -08:00
Mittal, Ketan 4311f1ab04 make style 2025-11-04 14:07:01 -08:00
Mittal, Ketan cc1a5774e1 add class documentation 2025-11-04 14:01:50 -08:00
Mittal, Ketan c1b7a529d6 resolve some documentation mix-up due to earlier merge 2025-11-04 13:43:11 -08:00
Mittal, Ketan 6dca8a1c54 merge upstream changes 2025-11-04 13:38:34 -08:00
Mittal, Ketan b0b21633b7 fix line wraps and add some checks 2025-11-04 13:36:54 -08:00
Joseph Signorelli 71aa4a366d Add update_data flag for SetOrdering + SetVDim 2025-11-04 15:21:29 -06:00
Joseph Signorelli 218880f7e1 add wrong_comp_count check to unit test 2025-11-04 15:14:23 -06:00
Joseph Signorelli 0c61b567a5 Merge branch 'array-vector-improvements-dev' into multivector-dev 2025-11-04 14:41:08 -06:00
Joseph Signorelli 55fb86f566 Merge branch 'master' into multivector-dev 2025-11-04 14:39:22 -06:00
Joseph Signorelli 94f2670dc0 style 2025-11-04 14:36:29 -06:00
Joseph Signorelli 749a04e7b1 MultiVector --> ParticleVector 2025-11-04 14:35:46 -06:00
Joseph Signorelli c6744e3f39 multivector --> particlevector file names 2025-11-04 14:21:22 -06:00
Joseph Signorelli 8a35a7843a style 2025-11-04 14:11:53 -06:00
Joseph Signorelli 967e811442 Keep existing data for SetVDim, with unit test 2025-11-04 14:11:24 -06:00
Joseph Signorelli 260f7c345b Add Get/SetComponents w/ unit test 2025-11-04 13:55:02 -06:00
Tzanio Kolev 058e3414b0 Merge branch 'master' into fix-caliper-scope 2025-11-04 11:51:58 -08:00
Joseph Signorelli 1bfe259d2a add test_ordering to cmakelists 2025-11-04 13:47:56 -06:00
Joseph Signorelli a4b5eb8bda Add unit test for getting + setting vector values 2025-11-04 13:47:27 -06:00
Joseph Signorelli 0fce771a08 rm file header duplication 2025-11-04 13:46:48 -06:00
Joseph Signorelli 331151d4f6 Create new file for test ordering 2025-11-04 13:46:22 -06:00
Julian Andrej 78b4b00359 remove unnecessary header 2025-11-04 11:31:09 -08:00
Mark L. Stowell e756067456 Merge branch 'master' into sjg/mesh-vis-dev 2025-11-04 11:23:49 -08:00
Veselin Dobrev aaf2d99de8 Merge pull request #5095 from mfem/product-coeff-gpu
Fix GPU test failure
2025-11-04 10:17:23 -08:00
Julian Andrej 9c110c6acb Merge branch 'master' into dfem-assemble-matrix
# Conflicts:
#	tests/unit/dfem/test_mass.cpp
2025-11-04 08:36:38 -08:00
Tzanio Kolev bcf5153192 Merge pull request #5001 from mfem/feature-gslib-fetch
Enable CMake fetching of GSLIB (and renaming of FETCH_TPLS, HYPRE_FETCH, and METIS_FETCH)
2025-11-04 07:53:01 -08:00
Tzanio Kolev 93066142ab Merge pull request #4993 from mfem/dfem-boundary-integrator
dFEM Boundary Integrator
2025-11-04 07:46:12 -08:00
Joseph Signorelli eabce59758 Default vdim to 1 2025-11-04 08:09:02 -06:00
Joseph Signorelli 779a4e6c0b Move ordering to general in its own file 2025-11-04 08:07:26 -06:00
Joseph Signorelli acfa01c46c Add warning to Vector::DeleteAt for unique indices 2025-11-04 07:41:42 -06:00
Tom Stitt 47c39d5102 followup to #4923 so mfem built with raja gpu support can be used in a library that isn't using the device compiler 2025-11-03 14:33:43 -08:00
Tzanio Kolev 4c803a8214 Merge pull request #5073 from mfem/sundials-add-finalize
Add Finalize routine to Sundials class for precise control of cleanup
2025-11-03 09:40:06 -08:00
Joseph Signorelli 43231cb143 Revert "Move Ordering to multivector.hpp/cpp" and move Ordering::Reorder to fespace.cpp/hpp
This reverts commit 769d2914c8.
2025-11-03 11:19:34 -06:00
John Camier d235e23b6a Merge branch 'master' into mtop-gpu 2025-11-02 08:49:58 -08:00
Tzanio Kolev f460b547ea Merge pull request #4861 from mfem/vector-pa-kernels
Vector pa kernels
2025-11-01 12:00:56 -07:00
Tzanio Kolev fa13480fcf Merge pull request #5027 from mfem/batch-linalg-pivot
Change batched linalg to always use 1-based pivot indexing.
2025-11-01 11:59:38 -07:00
Tzanio Kolev 48683a7f02 Merge branch 'master' into dfem-boundary-integrator 2025-11-01 11:51:05 -07:00
Tom Stitt 97cb5abb35 switch to CALI_CXX_MARK_SCOPE 2025-10-31 14:48:45 -07:00
Tom Stitt f09ca5df3c Merge remote-tracking branch 'origin/master' into fix-caliper-scope 2025-10-31 13:47:11 -07:00
Joseph Signorelli 0db3e561a3 style 2025-10-31 14:43:34 -05:00
Joseph SignorelliandAndrew Ho 229c9d9fbe Use new scan wrappers
Co-authored-by: Andrew Ho <ho37@llnl.gov>
2025-10-31 13:40:30 -05:00
Julian Andrej 58c6ac70c2 another fallback case 2025-10-31 09:20:53 -07:00
Will Pazner 7c38e8ca72 Add HostRead to unit test 2025-10-30 12:26:21 -07:00
Julian Andrej fdd909b132 typos and missing params 2025-10-30 08:34:43 -07:00
Julian Andrej 9222808f7c documentation 2025-10-30 08:23:20 -07:00
Julian Andrej 792c5b7617 revert to random input vector 2025-10-30 07:42:18 -07:00
Mark L. Stowell c18a0e470b Merge branch 'master' into complex-gf-save-dev 2025-10-30 00:06:41 -07:00
John Camier b9dc31be2b Merge branch 'master' into vector-pa-kernels 2025-10-29 19:36:56 -07:00
John Camier 09e4b7296e Merge branch 'master' into mtop-gpu 2025-10-29 19:36:36 -07:00
Stowell, Mark L. 6f848e00f8 More file closures 2025-10-29 15:13:30 -07:00
Stowell, Mark L. ce3c13781b Improving documentation 2025-10-29 14:41:06 -07:00
Stowell, Mark L. 41378e2fa3 Closing files in unit test 2025-10-29 14:40:36 -07:00
Veselin Dobrev f19768cc70 Merge pull request #4952 from mfem/hughcars/race-condition-fix
Fix OpenMP race conditions
2025-10-29 14:03:50 -07:00
Julian Andrej 6e67b690d9 change fallback mechanism 2025-10-29 11:29:48 -07:00
Tzanio Kolev 5e40324e5d Merge branch 'master' into vector-pa-kernels 2025-10-29 11:24:34 -07:00
Tzanio Kolev a39d48748d Merge pull request #5063 from mfem/product-coeff-gpu
Implement Coefficient::Project for sum, product, and ratio coefficients
2025-10-29 11:21:41 -07:00
Tzanio Kolev 56a122e86a Merge pull request #5086 from mfem/artv3/qfunction-temp-memory
QuadratureFunction: add constructor that takes device memory type
2025-10-29 11:21:25 -07:00
Stowell, Mark L. 49d57f4f6f Correcting for loop range 2025-10-29 11:10:15 -07:00
Julian Andrej 09bff27a2c remove clang format 2025-10-29 08:30:57 -07:00
Julian Andrej 8001d0b75a make style 2025-10-28 16:49:45 -07:00
Julian Andrej 2bebd0e4a4 relax norms properly 2025-10-28 16:47:16 -07:00
Stowell, Mark L. 525ef99753 Merge branch 'complex-gf-save-dev' of github.com:mfem/mfem into complex-gf-save-dev 2025-10-28 16:20:43 -07:00
Stowell, Mark L. e6180f56ee Documentation updates and improved error reporting 2025-10-28 16:20:24 -07:00
Mark L. Stowell 132de1fa57 Merge branch 'master' into complex-gf-save-dev 2025-10-28 14:55:21 -07:00
Stowell, Mark L. e1a35d929c Adding unit test (and bug fixes that it found) 2025-10-28 14:27:33 -07:00
Julian Andrej 3154767cc5 relax tolerance to 2*eps 2025-10-28 13:10:35 -07:00
Chris Vogl ac84807be1 merged in master with conflict resolution in FindHypre 2025-10-28 12:48:35 -07:00
Chris Vogl 9601a02547 added deprecation warning for FETCH_TPLS as requested by @kmittal2 2025-10-28 12:36:10 -07:00
Chris Vogl b9bebd0921 add changes suggested by @nmnombre and @k-collie to support default and enivornment C flags for GSLIB 2025-10-28 12:24:27 -07:00
Andrew Ho d276ad6aff Merge branch 'master' into array-vector-improvements-dev 2025-10-28 11:48:12 -07:00
Julian Andrej f4f4889a4c fix memory leak 2025-10-28 11:42:28 -07:00
Mark L. Stowell f3aeae7b04 Merge branch 'master' into sjg/mesh-vis-dev 2025-10-28 11:23:14 -07:00
Julian Andrej 7850d9d0a9 integration rule 2025-10-28 10:57:20 -07:00
John Camier 2944bc5ea3 Merge branch 'master' into product-coeff-gpu 2025-10-28 10:19:40 -07:00
John Camier 2eb244fb21 Merge branch 'master' into vector-pa-kernels 2025-10-27 18:19:18 -07:00
John Camier decc2f695d Merge branch 'master' into mtop-gpu 2025-10-27 18:18:56 -07:00
Julian Andrej 83330e2639 revert debug device 2025-10-27 16:17:01 -07:00
Julian Andrej 5889c2155e use existing matrix comparison test - gpu still failing 2025-10-27 15:38:10 -07:00
Andrew Ho c817c3cdbe Merge branch 'master' into batch-linalg-pivot 2025-10-27 13:17:02 -07:00
Julian Andrej a24b5edcf0 move AddIntegrator to public for nvcc 2025-10-27 08:24:03 -07:00
Julian Andrej b47f3e52a3 add fallback return for dispatch 2025-10-27 08:09:16 -07:00
Julian Andrej 16694dd1c7 Merge branch 'dfem-assemble-matrix' of github.com:mfem/mfem into dfem-assemble-matrix 2025-10-27 08:04:01 -07:00
Julian Andrej e8c3dc1e8f explicit casts 2025-10-27 08:03:51 -07:00
Julian Andrej 8a6ac41d69 slight norm comparison discrepancy remaining on rt-2d-q3 2025-10-27 08:03:43 -07:00
Ido Akkerman 9270365ff8 Fix pedantic bug 2025-10-27 14:51:15 +01:00
Ido Akkerman f270144550 Removed debug functions 2025-10-27 13:57:43 +01:00
Ido Akkerman 09014fcef6 Add comments to new functions 2025-10-27 13:51:02 +01:00
Ido Akkerman 4d945bc74f Improved stokes description 2025-10-27 13:44:49 +01:00
Ido Akkerman 0da0e1da2f Adding vector_diffusion and stokes test cases to makefile. Resulting files are added to gitignore and clean-exec 2025-10-27 13:36:22 +01:00
Rushan Zhang 93f3b73790 fix mistake with Boris 2025-10-26 18:55:34 -04:00
Tzanio Kolev 4870ecd351 Merge branch 'master' into dfem-boundary-integrator 2025-10-26 11:41:08 -07:00
Tzanio KolevandJohn Camier 5ed9f34dd8 Update tests/unit/dfem/test_mass.cpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-10-26 11:40:47 -07:00
Tzanio Kolev 41c8a8e65e Merge pull request #5051 from mfem/cmake-gpu
Change CMake CUDA_ARCH behavior
2025-10-26 11:39:33 -07:00
Tzanio Kolev a2025d7693 Merge pull request #5018 from mfem/parallel-scan
Additional parallel scan wrappers
2025-10-26 11:39:08 -07:00
camierjs ae629dda70 Fix typos 2025-10-25 11:53:33 -07:00
John Camier 70439dcb84 Merge branch 'master' into dfem-boundary-integrator 2025-10-25 11:27:22 -07:00
John Camier 1485fedec9 Merge branch 'master' into product-coeff-gpu 2025-10-25 11:26:27 -07:00
camierjs 2bad979cac Add GPU tags for dFEM diffusion, divergence and lvector parallel unit tests 2025-10-24 18:05:58 -07:00
Sohail Reddy 1ead635557 checked variable name 2025-10-24 17:13:29 -07:00
Sohail Reddy c903edd931 Merge branch 'master' into user-inner-prod 2025-10-24 17:01:51 -07:00
Sohail Reddy d8cd8f4f37 added doc for inner product operators 2025-10-24 17:01:11 -07:00
Arturo Vargas 4c9769a927 add contructor that takes device memory type 2025-10-24 11:13:33 -07:00
Julian Andrej b83a4184b5 guard weight instantiation 2025-10-24 10:00:51 -07:00
Julian Andrej 042762701b reorganize dimensional dispatch 2025-10-24 10:00:38 -07:00
Julian Andrej 3b6532e7bb style 2025-10-24 10:00:26 -07:00
Andrew Ho 8731dd171e old comment 2025-10-24 09:23:34 -07:00
Andrew Ho 45b1b3c535 Merge remote-tracking branch 'base/parallel-scan' into parallel-scan 2025-10-24 09:18:59 -07:00
Andrew Ho d450cc563c missing header 2025-10-24 09:17:45 -07:00
Andrew Ho 4083fcc4b4 Merge branch 'master' into parallel-scan 2025-10-24 09:10:12 -07:00
John Camier f27a7c5612 Merge branch 'master' into vector-pa-kernels 2025-10-24 08:57:15 -07:00
John Camier 9e058cbfe9 Merge branch 'master' into mtop-gpu 2025-10-24 08:56:55 -07:00
Mark L. Stowell 1027c6a9a8 Merge branch 'master' into complex-gf-save-dev 2025-10-24 08:06:34 -07:00
Stowell, Mark L. 7843bfe46a Fixing CI errors 2025-10-24 08:05:44 -07:00
Chris VoglandNuno Nobre 71aa2f0985 Update config/cmake/modules/FindGSLIB.cmake
adding a make clean command to GSLIB fetching to ensure it is rebuilt when CMake configuration is changed... also added -O2 -fPIC flags when building a shared library

Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2025-10-23 23:45:33 -07:00
Andrew Ho 35f0d5bfe1 complex densemat pivot fix 2025-10-23 18:59:24 -07:00
Tzanio Kolev 70a3355e90 Merge pull request #4431 from mfem/woptim/leverage-radiuss-ci
Leverage radiuss CI + add tioga
2025-10-23 18:26:33 -07:00
Stowell, Mark L. a6b1eff541 Ignoring new output files 2025-10-23 17:51:51 -07:00
Stowell, Mark L. 893c28bff3 Testing Save in parallel 2025-10-23 17:45:35 -07:00
Stowell, Mark L. 207fcd2fb6 Adding Save to ParComplexGridFunction 2025-10-23 17:45:19 -07:00
Stowell, Mark L. d312108f14 Testing serial Save function 2025-10-23 17:44:57 -07:00
Stowell, Mark L. 409aa1cba7 Adding Save to ComplexGridFunction 2025-10-23 17:44:39 -07:00
Veselin Dobrev 454a215175 More tweaks in Gitlab CI 2025-10-23 16:21:29 -07:00
Tom Stitt 8af606b848 forgot static 2025-10-23 13:30:16 -07:00
Tom Stitt 9a2bbec272 one more change 2025-10-23 13:29:48 -07:00
Sohail Reddy 57f4be30c5 Made InnerProductOperator::Dot virtual 2025-10-23 10:13:12 -07:00
Tom Stitt 9b9a6cfe34 missing one 2025-10-22 21:37:36 -07:00
Andrew Ho dbea0c3a0b Merge branch 'master' into batch-linalg-pivot 2025-10-21 11:40:47 -07:00
Sohail Reddy 96a1cf8c93 updated vectors in WeightedInnerProduct to use different weighting operators 2025-10-21 10:18:11 -07:00
John Camier 8fbd815ad9 Merge branch 'master' into dfem-assemble-matrix 2025-10-21 10:09:46 -07:00
Andrew HoandJohn Camier a447a33528 Update tests/unit/general/test_scan.cpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-10-21 08:34:02 -07:00
Andrew HoandJohn Camier 79f2ce2540 Update tests/unit/general/test_scan.cpp
Co-authored-by: John Camier <camierjs@gmail.com>
2025-10-21 08:33:54 -07:00
Sohail Reddy a109493821 fixed MPI variable initialization 2025-10-20 12:57:57 -07:00
Veselin Dobrev 32ce005ca0 Try to fix Gitlab CI 2025-10-20 09:42:21 -07:00
John Camier 3f0e146fb6 Merge branch 'master' into hughcars/race-condition-fix 2025-10-20 09:36:06 -07:00
John Camier 87c5e59228 Merge branch 'master' into parallel-scan 2025-10-20 09:35:18 -07:00
John Camier 9e91bfb376 Merge branch 'master' into cmake-gpu 2025-10-20 09:09:33 -07:00
Sohail Reddy 253a5b4b79 Added operator weighted inner product. Moved inner product defintions to solvers.hpp/cpp 2025-10-19 13:37:39 -07:00
John Camier b9105ca3e8 Merge branch 'master' into product-coeff-gpu 2025-10-19 10:31:42 -07:00
camierjs 90379a4792 mtop_test_iso_elasticity mesh file directory and output numbers 2025-10-19 10:24:33 -07:00
camierjs 5f33d92ca8 Move mtop data meshes to miniapp folder 2025-10-19 09:30:40 -07:00
John Camier fdb4368212 Merge branch 'master' into vector-pa-kernels 2025-10-19 09:27:54 -07:00
John Camier c04e52bc6a Merge branch 'master' into mtop-gpu 2025-10-19 09:27:35 -07:00
Sohail Reddy 1808ca1f3a Added class for user-defined inner products. Updated IterativeSolver to use custom inner products. 2025-10-19 01:14:50 -07:00
Socratis Petrides f492c5c70d add README 2025-10-18 19:15:39 -07:00
Socratis Petrides 474d9811dd minor edits in the driver 2025-10-17 21:19:39 -07:00
Socratis Petrides 9a2ef1263f contact-miniapp: squash all changes since branching from master 2025-10-17 18:39:55 -07:00
Veselin Dobrev c34de48041 In Gitlab CI, use a longer timeout for the Matrix pipeline 2025-10-17 16:34:20 -07:00
Will Pazner 6602df4fd2 In QuadratureInterpolator::SupportsFESpace, return false for mixed meshes or variable orders
In QuadratureFunction::ProjectGridFunction unit test, test (element) QuadratureSpace also.
2025-10-17 14:58:46 -07:00
Will Pazner e9ca0960f7 Add fix for Mesh::MakeHigherOrderSimplicial_ on surface meshes 2025-10-17 09:15:45 -07:00
Will Pazner 8c236973c7 Merge remote-tracking branch 'origin/master' into fix-issue-4455
# Conflicts:
#	tests/unit/mesh/test_mesh.cpp
2025-10-17 09:15:35 -07:00
Will Pazner 8d0a2d4336 Use star-surf in 'MakeSimplicial Surface Mesh' unit test 2025-10-17 08:57:35 -07:00
Veselin Dobrev 0694668c24 In Gitlab CI, update the radiuss/radiuss-shared-ci ref to the latest
release tag, v2025.09.1.
2025-10-16 20:20:33 -07:00
Veselin Dobrev 3492b20c70 Try to fix reporting in Gitlab CI 2025-10-16 19:47:47 -07:00
Socratis Petrides 033b29edfc unit test fix
adjusting tol
loosen the iteration requirement
fix int64
mem leak fix
one more fix for BigInt
another fix for int64
2025-10-16 19:27:25 -07:00
Veselin Dobrev 5fec899806 Try to fix reporting in Gitlab CI 2025-10-16 16:13:26 -07:00
Veselin Dobrev fc010f0423 In Gitlab CI, test why reporting does not work 2025-10-16 13:44:02 -07:00
Socratis Petrides 6bffbc1d2e Merge branch 'master' into amgf 2025-10-16 12:45:47 -07:00
Socratis Petrides ec6937c286 Merge branch 'amgf' of github.com:mfem/mfem into amgf 2025-10-16 12:45:38 -07:00
Socratis Petrides acc1b3a28c adding a small test for amgf 2025-10-16 12:44:52 -07:00
Adrien Bernede a9427e2663 Merge branch 'master' into woptim/leverage-radiuss-ci 2025-10-15 22:07:09 -07:00
Tzanio Kolev 72deb3b12c Merge branch 'master' into woptim/leverage-radiuss-ci 2025-10-15 16:09:08 -07:00
Will Pazner 63d983ed6a Add unit test for QuadratureFunction::ProjectGridFunction 2025-10-15 14:54:25 -07:00
Will Pazner 59a95cc8e6 In QuadratureFunction::ProjectGridFunction, fall back earlier
GetElementRestriction or GetFaceRestriction may fail in unsupported cases.

In the case of FaceQuadratureSpace, fall back on non-tensor meshes since
ElementDofOrdering::NATIVE is not supported in the restriction operator.
2025-10-15 14:52:36 -07:00
Jan Nikl 440d552e85 Merge branch 'master' into qf-project-gf-fallback 2025-10-15 10:04:34 -07:00
Julian Andrej 67d31a13fd Merge branch 'dfem-boundary-integrator' of github.com:mfem/mfem into dfem-boundary-integrator 2025-10-15 08:29:49 -07:00
Julian Andrej 0320e5a476 fix for ordering in tensor transform 2025-10-15 08:29:17 -07:00
John Camier 91e56806bd Merge branch 'master' into mtop-gpu 2025-10-15 07:46:42 -07:00
Tzanio Kolev 041b4ab958 Merge branch 'master' into dfem-boundary-integrator 2025-10-15 06:53:16 -07:00
Cody Balos 3374851918 whitespace 2025-10-14 16:48:32 -07:00
Cody Balos 46a44c5241 add Sundials::Finalize routine to allow for precise control of cleanup 2025-10-14 16:43:44 -07:00
camierjs 641692ac4a Merge branch 'master' into vector-pa-kernels 2025-10-14 16:28:30 -07:00
camierjs f97bfdea7e Cleanup and remove unused changes 2025-10-14 16:26:10 -07:00
Andrew Ho 159c89542b formatting 2025-10-14 15:40:08 -07:00
camierjs 3bfa5f0e37 [MTOP] solver with GPU support 2025-10-14 15:37:14 -07:00
Andrew Ho 37fc5a3d5c Merge branch 'master' into cmake-gpu 2025-10-14 11:42:48 -07:00
Andrew Ho 6fad311e11 Merge branch 'master' into parallel-scan 2025-10-14 11:41:16 -07:00
Dylan Copeland fcdc9c43d5 Fix uninitialized variables. 2025-10-14 11:24:34 -07:00
Julian Andrej d32a63171f trying to figure out attribute arrays, 3d boundary jacobians failing 2025-10-14 09:53:43 -07:00
Veselin Dobrev 4fdf2cdda1 In Gitlab CI, remove the explicit suppression of CUSPARSE warnings 2025-10-13 19:58:42 -07:00
Veselin Dobrev 0af0b9b420 Merge branch 'master' into woptim/leverage-radiuss-ci 2025-10-13 19:57:33 -07:00
Julian Andrej 3c0f1d783e fix 1d case 2025-10-13 15:04:56 -07:00
Julian Andrej cd55bcc67b fix attempt 2025-10-13 13:44:46 -07:00
Julian Andrej 689288e18d fix unit tests, failing 2d derivatives 2025-10-13 09:29:07 -07:00
John Camier 01afc1685a Merge branch 'master' into vector-pa-kernels 2025-10-13 08:12:57 -07:00
Veselin Dobrev 4501988a3b Try to fix Gitlab CI 2025-10-10 21:54:49 -07:00
Veselin Dobrev cd82063118 Try to fix Gitlab CI 2025-10-10 20:57:43 -07:00
Veselin Dobrev cac3f72e9e Try to fix Gitlab CI 2025-10-10 19:56:54 -07:00
Veselin Dobrev d93a030431 Updates in Gitlab CI 2025-10-10 19:04:07 -07:00
Veselin Dobrev 8cfd5e0a07 Update the mfem-uberenv commit hash 2025-10-10 17:29:47 -07:00
Veselin Dobrev 63110738d4 In Gitlab CI replace Lassen with Matrix 2025-10-10 17:25:53 -07:00
John Camier c05d58a9c2 Merge branch 'master' into vector-pa-kernels 2025-10-10 08:01:57 -07:00
Ido Akkerman 0dc3a9d319 Fix unit test cmake 2025-10-10 14:20:17 +02:00
Ido Akkerman d8b7cb5306 Fix pedantic compile error 2025-10-10 14:18:21 +02:00
Ido Akkerman a606f5ba18 Correct test errors 2025-10-10 13:58:48 +02:00
Ido Akkerman 525f77ee6e Cleaner stokes example 2025-10-10 13:40:33 +02:00
Ido Akkerman a8970a1e4a Add visualization options 2025-10-10 11:52:08 +02:00
Ido Akkerman 8a2cac6a8e Style 2025-10-10 11:51:25 +02:00
Ido Akkerman 106dd2f486 Remove comments and old code 2025-10-10 11:50:33 +02:00
Ido Akkerman 969d166925 Style 2025-10-10 11:50:15 +02:00
Ido Akkerman 2b1b3fe8a3 Add options to nurbs_vector_diffusion miniapp 2025-10-10 11:28:43 +02:00
Ido Akkerman 5bdfff1e0c Improve Face vd bdr vertices check for nurbs 2025-10-10 11:27:11 +02:00
Will Pazner 2647e9ef3d Implement Coefficient::Project for sum, product, and ratio coefficients
Add unit test to compare with Coefficient::Eval
2025-10-09 23:03:58 -07:00
Ido Akkerman 34017cf732 Adding fixes -- keeping the old stuf for now 2025-10-09 14:42:39 +02:00
Ido Akkerman b7cebb221c Add Hcurl and HDiv tests 2025-10-09 14:42:15 +02:00
Andrew Ho 0efc76834a Merge branch 'master' into batch-linalg-pivot 2025-10-08 16:09:48 -07:00
Andrew Ho 1311729f79 needs to be uncommented to actually work 2025-10-08 16:08:26 -07:00
Andrew Ho 526f73688e Merge branch 'master' into dfem-boundary-integrator 2025-10-08 11:34:42 -07:00
Andrew Ho aa0a33f51b workaround for umpire and BLT using the deprecated CMake CUDA integration 2025-10-08 10:22:08 -07:00
Ido Akkerman 17945a7bec Remove unused variable 2025-10-08 13:27:14 +02:00
Ido Akkerman 20bbb7bfc6 make style 2025-10-08 13:22:25 +02:00
Ido Akkerman 2a05231d2e Improved miniapps 2025-10-08 13:19:32 +02:00
Ido Akkerman 826cca9066 Correct the Projections 2025-10-08 12:34:06 +02:00
Andrew Ho 34abf22be5 Merge branch 'master' into batch-linalg-pivot 2025-10-06 11:38:43 -07:00
Andrew Ho aa35d62e28 fix compatibility with CMAKE_HIP_ARCHITECTURES 2025-10-06 10:31:50 -07:00
Tzanio Kolev bfa80426bb Merge branch 'master' into amgf 2025-10-03 09:28:03 -07:00
Ido Akkerman d54ed209a6 Remove comments 2025-10-03 17:31:56 +02:00
Ido Akkerman 4ec01674c3 Remove unused variables 2025-10-03 17:26:44 +02:00
John Camier 0d6eced899 Merge branch 'master' into vector-pa-kernels 2025-10-03 08:18:05 -07:00
Ido Akkerman f5a67eb43b Split scalar and vector FE integrators 2025-10-03 17:14:10 +02:00
Ido Akkerman a8b34de0b7 Remove redundant coefficient 2025-10-03 16:15:14 +02:00
Ido Akkerman a2e81d705f Remove redundant coefficient 2025-10-03 16:08:42 +02:00
Ido Akkerman 716177e0dc correct comment 2025-10-03 12:00:45 +02:00
Ido Akkerman a192464f46 make style 2025-10-03 11:44:20 +02:00
Ido Akkerman 2041f5ac41 Use placeholder projection 2025-10-03 11:19:24 +02:00
Ido Akkerman 1ed723dc52 Add bilinear integ functions for Nitsche BC 2025-10-03 10:17:14 +02:00
Ido Akkerman 7c9d707ccf Remove debug statement 2025-10-03 10:16:17 +02:00
Ido Akkerman dea8141ced Add linear integ function for Nitsche BC 2025-10-03 09:56:02 +02:00
Ido Akkerman 6b0867e850 Change deprecated function call 2025-10-03 08:45:34 +02:00
Ido Akkerman 3ca21e3179 Add scaling to NormalTrace Integ 2025-10-03 08:29:48 +02:00
Ido Akkerman a9df973873 Add scaling to VectorFEDiv bilinearinteg 2025-10-03 08:23:25 +02:00
Ido Akkerman 91f521652c Add vector case to DGDirichlet 2025-10-03 08:15:41 +02:00
Ido Akkerman 455f80d082 Style 2025-10-03 08:14:51 +02:00
John Camier e336658f8c Merge branch 'master' into hughcars/race-condition-fix 2025-10-02 15:31:59 -07:00
Andrew Ho b276939d69 Merge branch 'master' into parallel-scan 2025-10-02 13:23:25 -07:00
Veselin Dobrev 324ab0684e Merge branch 'master' into woptim/leverage-radiuss-ci 2025-10-02 12:33:03 -07:00
Veselin Dobrev c5efd6a06a Gitlab CI: replace Ruby with Dane 2025-10-02 11:57:54 -07:00
Andrew Ho 626cb41c1b Merge branch 'master' into cmake-gpu 2025-10-02 11:47:59 -07:00
Veselin Dobrev 0c167ab5e7 Merge pull request #4684 from mfem/leverage-radiuss-ci--updates
Updates for the branch `woptim/leverage-radiuss-ci`
2025-10-02 11:33:47 -07:00
Hugh Carson 944217e7f3 Merge branch 'master' into hughcars/race-condition-fix 2025-10-02 12:20:20 -04:00
Hugh Carson bd31eba056 Revert "Introduce DenseMatrix::NewMemoryAndSize and use to reallocate buffer"
This reverts commit 94b80fcd3c.
2025-10-02 12:18:59 -04:00
Ido Akkerman 107b5bede4 Change definition order 2025-10-02 14:47:26 +02:00
Ido Akkerman 2802e9af1b Add missing function 2025-10-02 14:47:08 +02:00
Ido Akkerman 0f09bf2cf7 Add Vector FE diffusion bilinear integrator 2025-10-02 14:41:15 +02:00
Ido Akkerman 3ceefdada4 Add vector coefficient based on a gridfucntion 2025-10-02 14:36:18 +02:00
Ido Akkerman 7e22247237 Add some access functions to densetensor class 2025-10-02 14:30:06 +02:00
Ido Akkerman 3892469c5c Add gradient to vector NURBS FE 2025-10-02 14:23:21 +02:00
Ido Akkerman 752b7a4af3 Add vectorfe grad test cases 2025-10-02 14:08:38 +02:00
Andrew Ho eea5d48c4e Change CUDA_ARCH so it behaves like CMAKE_CUDA_ARCHITECTURES while still supporting the previous usage. 2025-10-01 11:08:02 -07:00
John Camier 5637c98f50 Merge branch 'master' into vector-pa-kernels 2025-09-30 11:35:05 -07:00
John Camier fe7cce14b7 Merge branch 'master' into vector-pa-kernels 2025-09-29 07:45:27 -07:00
Hugh Carson 94b80fcd3c Introduce DenseMatrix::NewMemoryAndSize and use to reallocate buffer 2025-09-26 10:12:51 -04:00
John Camier 3d6c5ebd71 Merge branch 'master' into vector-pa-kernels 2025-09-25 15:58:31 -07:00
John Camier 6e2148a698 Merge branch 'master' into hughcars/race-condition-fix 2025-09-25 15:58:14 -07:00
Andrew Ho 20caca8333 Merge branch 'master' into parallel-scan 2025-09-25 15:24:03 -07:00
Andrew Ho 37ebeed499 Merge branch 'master' into batch-linalg-pivot 2025-09-25 15:23:38 -07:00
Joseph Signorelli 1f6afa0f87 Use up-to-date scan 2025-09-25 09:34:41 -05:00
Joseph Signorelli a18d36e841 Merge branch 'master' into array-vector-improvements-dev 2025-09-25 09:24:32 -05:00
Socratis Petrides a1b9ad1403 update comment 2025-09-24 17:09:28 -07:00
Socratis Petrides 47c85401cd adding comments 2025-09-24 16:58:35 -07:00
Socratis Petrides 2aa40b53e4 Changelog 2025-09-24 16:57:42 -07:00
Tzanio Kolev 792af80eac Merge branch 'master' into dfem-assemble-matrix 2025-09-24 08:12:52 -07:00
Andrew Ho dd21f5469c Merge branch 'master' into batch-linalg-pivot 2025-09-23 19:46:10 -07:00
Andrew Ho 1ee9cbcc43 Merge branch 'master' into parallel-scan 2025-09-23 16:30:40 -07:00
Andrew Ho 9c56bc4d9e Merge remote-tracking branch 'base/batch-linalg-pivot' into batch-linalg-pivot 2025-09-23 14:40:16 -07:00
Andrew Ho 229e1b1bc6 also need to change the base LUFactors class to 1-based 2025-09-23 14:39:32 -07:00
Andrew Ho c2b508c5fc Merge branch 'master' into batch-linalg-pivot 2025-09-23 14:05:13 -07:00
Andrew Ho 7df869974d Change batched linalg to always use 1-based pivot indexing.
This makes the native backend consistent with vendors and MAGMA
2025-09-23 14:02:03 -07:00
Julian Andrej 593622ae88 host/device bug 2025-09-23 09:48:57 -07:00
Julian Andrej 256821fb6c maybe unused annotations 2025-09-23 09:48:47 -07:00
Julian Andrej cffcd4ea89 hypre gpu fix 2025-09-23 09:48:39 -07:00
John Camier 30859f614d Merge branch 'master' into sjg/mesh-vis-dev 2025-09-23 08:00:27 -07:00
Julian Andrej 66dfc9352f nvcc fixes 2025-09-22 11:58:55 -07:00
Julian Andrej 4cd6c1fbfb add matrix assemble and amg option to minsurface miniapp 2025-09-22 10:06:04 -07:00
camierjs b73ae540e1 Fix DO_NOT_DOCUMENT cond 2025-09-22 07:50:13 -07:00
camierjs 56615c8fc3 Fix VectorDiffusionIntegrator AddMultPA definitions 2025-09-21 20:42:39 -07:00
camierjs 5e99a705c2 Merge branch 'master' into vector-pa-kernels 2025-09-21 20:42:38 -07:00
John Camier 8d4819b143 Merge branch 'master' into hughcars/race-condition-fix 2025-09-21 20:13:30 -07:00
camierjs 9d1fe4ba6b Fix FormLinearSystem unit test merge duplicate 2025-09-21 10:58:22 -07:00
camierjs 8147aeba7d Merge branch 'master' into vector-pa-kernels 2025-09-21 10:54:17 -07:00
John Camier 78afa5d313 Merge branch 'master' into table-array 2025-09-21 10:31:08 -07:00
John Camier ce3f215d75 Merge branch 'master' into hughcars/race-condition-fix 2025-09-21 10:15:00 -07:00
Julian Andrej 47ab067ead Merge branch 'master' into dfem-assemble-matrix 2025-09-19 10:17:29 -07:00
Julian Andrej 72682e3f46 assemble methods and unit tests 2025-09-19 09:16:26 -07:00
Andrew Ho 8edf90f72a Merge branch 'master' into parallel-scan 2025-09-17 10:30:30 -07:00
Andrew Ho 28590b2026 Added conditional copy/compaction wrappers
Speed up CMake unit test building
2025-09-15 15:43:25 -07:00
Andrew Ho 5429e98bb3 Merge branch 'master' into parallel-scan 2025-09-15 11:48:03 -07:00
Julian Andrej 0e1e7ac5c9 guard more logical expression 2025-09-14 11:54:15 -07:00
Julian Andrej 1c03e51342 doc 2025-09-14 11:52:11 -07:00
Julian Andrej dea236d9ff more warnings 2025-09-14 11:50:46 -07:00
Julian Andrej 90e9ec2d50 Merge branch 'dfem-boundary-integrator' of github.com:mfem/mfem into dfem-boundary-integrator 2025-09-14 11:18:01 -07:00
Julian Andrej b9da92c5a1 logical expression warnings 2025-09-14 11:17:50 -07:00
Chris VoglandNuno Nobre d84095af57 specify MPI flag for fetched GSLIB build
Co-authored-by: Nuno Nobre <nuno.nobre@stfc.ac.uk>
2025-09-03 14:57:05 -07:00
Chris Vogl 8dba8024a1 added MFEM_ prefix to fetch variables; added GSLIB default; changed naming 2025-09-02 16:05:18 -07:00
Chris Vogl 41151d5fe9 added functionality to fetch GSLIB 2025-09-02 16:05:18 -07:00
Hugh Carson 3dfe67a219 Merge branch 'master' into sjg/mesh-vis-dev 2025-09-02 12:11:44 -04:00
Hugh Carson 36e2c896e4 Fix race condition in DofToQuad 2025-09-02 12:10:27 -04:00
Hugh Carson 1fa6c1ded9 Add thread safe buffer version of operator() for DenseTensor 2025-09-02 12:10:27 -04:00
Tzanio Kolev 419de5c398 Merge branch 'master' into dfem-boundary-integrator 2025-08-31 15:33:20 -07:00
Socratis Petrides 1e62733c26 include memory 2025-08-25 15:15:33 -07:00
Socratis Petrides 7ab83365aa another fix for serial builts 2025-08-25 14:57:40 -07:00
Socratis Petrides 60100a336e fix serial built 2025-08-25 13:48:47 -07:00
Will Pazner d9d4b5bb72 Rename member variable to fix shadow warning 2025-08-23 15:13:05 -07:00
Will Pazner 5c04ba55f6 Fix failing TMOP tests
Keep a cached copy of GeomToPerfGeomJac for use on device.

Be sure to call HostRead before DenseMatrix::Det.
2025-08-23 15:06:03 -07:00
Socratis Petrides e09479465b derived amgf class 2025-08-22 15:16:45 -07:00
Socratis Petrides c4da5827a2 removing op handle. Doing the RAP by type cast 2025-08-21 22:19:05 -07:00
Socratis Petrides 8df19f39ef first iteration on FilteredSolver 2025-08-21 18:37:04 -07:00
Joseph Signorelli 405c674d22 Update unit tests to check host + device 2025-08-21 15:24:19 -07:00
Joseph Signorelli 83362c7b4d Add HostReadWrite to beginnning of Array 2025-08-21 15:13:50 -07:00
Will Pazner 240443abfb Vector::DeleteAt on device using InclusiveScan 2025-08-21 14:42:17 -07:00
Andrew Ho c189f50ab4 Added GPU-accelerated parallel scan 2025-08-20 11:15:53 -07:00
Socratis Petrides 7d7d4cf6f4 class sketch 2025-08-19 17:45:35 -07:00
Joseph Signorelli f43c6771c6 Merge branch 'master' into multivector-dev 2025-08-19 16:42:31 -07:00
Will Pazner 061a92067f Merge branch 'master' into array-vector-improvements-dev 2025-08-19 16:33:02 -07:00
Joseph SignorelliandWill Pazner 0ddb02c7e7 fix type
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-08-19 15:57:45 -07:00
Joseph SignorelliandWill Pazner 427406d1b8 Update Vector::Reserve
Co-authored-by: Will Pazner <11493037+pazner@users.noreply.github.com>
2025-08-19 13:36:42 -07:00
Joseph Signorelli 09327924ee Merge branch 'master' into multivector-dev 2025-08-19 12:15:01 -07:00
Joseph Signorelli 5026d6ca8a Merge branch 'master' into array-vector-improvements-dev 2025-08-19 12:14:21 -07:00
Joseph Signorelli e659e823e9 Add newlines to end of files 2025-08-19 12:06:43 -07:00
Joseph Signorelli 04d7e8a62f add newlines to bottom of test files 2025-08-19 12:03:34 -07:00
Veselin Dobrev 1bf4c4f7fe Merge branch 'master' into table-array 2025-08-15 12:23:56 -07:00
Joseph Signorelli 715ab0a328 Fix typo causing doc fail 2025-08-13 16:35:47 -07:00
Joseph Signorelli 160100e0b3 style 2025-08-13 16:27:19 -07:00
Joseph Signorelli 997942b44e Add MultiVector w/ unit tests 2025-08-13 16:22:23 -07:00
Joseph Signorelli a1bb9cfe9f style 2025-08-13 16:22:23 -07:00
Joseph Signorelli 2489dce8a0 Add Vector::Reserve 2025-08-13 16:22:23 -07:00
Joseph Signorelli 4ea338fe53 Add Vector::DeleteAt w/ unit test 2025-08-13 16:22:23 -07:00
Joseph Signorelli eb5a0eb132 Add Array::DeleteAt w/ unit test. 2025-08-13 16:22:23 -07:00
Joseph Signorelli 236ba45fb9 Add Ordering::Reorder w/ unit test 2025-08-13 16:11:52 -07:00
Joseph Signorelli e23975a2d8 Add test_multivector.cpp 2025-08-13 16:10:41 -07:00
Joseph Signorelli 769d2914c8 Move Ordering to multivector.hpp/cpp 2025-08-13 16:02:37 -07:00
Joseph Signorelli 5c87a6c665 Create new files multivector.cpp/hpp 2025-08-13 15:47:17 -07:00
Joseph Signorelli 915853cee0 style 2025-08-13 14:51:59 -07:00
Joseph Signorelli a52599d4cc Add Vector::Reserve 2025-08-13 14:49:39 -07:00
Joseph Signorelli 0d5b13c4aa Add Vector::DeleteAt w/ unit test 2025-08-13 14:44:55 -07:00
Joseph Signorelli dda6b0dbe1 Add Array::DeleteAt w/ unit test. 2025-08-13 14:39:29 -07:00
Veselin Dobrev 32efc9be3c Merge pull request #4978 from mfem/table-array-additions
Proposed addtions and changes to PR #4028 (branch table-array)
2025-08-13 12:16:20 -07:00
Veselin Dobrev 2675f36788 Update formatting 2025-08-13 10:57:22 -07:00
Veselin Dobrev 3372b2d1a8 Proposed addtions and changes to PR #4028 (branch table-array)
* Add mfem::swap for the classes Memory, Array, Array2D, and Vector.
* These mfem::swap functions are for use by the standard library.
* Re-define mfem::Swap to use mfem::swap, or, if is not defined, std::swap
* Remove some calls to Memory::Reset after Memory::Delete since the latter
  calls the former
* Update/improve the definitions of the Vector move-constructor and move-
  assignment; in the copy-assignment, skip the virtual calls to
  v.UseDevice(bool) when they are not needed.
* In DenseMatrix::SetSize, update the size of the data Array to match
  height x width.
* Propagate the parameter use_dev in Table::{ReadI,ReadJ} to the respective
  Array::Read calls.
* In Table::Size_of_connections, return J.Size() instead of I[size].
* Some small Doxygen tweaks.
2025-08-12 14:51:58 -07:00
Veselin Dobrev 8a278c70d6 Merge branch 'master' into table-array 2025-08-12 14:44:38 -07:00
Will Pazner 7fea470192 Modify Array::Copy to match semantics of Vector
Also revert Array copy assignment to use Array::Copy
2025-08-11 15:24:25 -07:00
Will Pazner 6bab6a283f Use same copy assignment semantics in Array as in Vector 2025-08-09 10:25:57 -07:00
Will PaznerandVeselin Dobrev 5a7edc9548 Fix Array move constructor and assignment operator
Co-authored-by: Veselin Dobrev <dobrev@llnl.gov>
2025-08-09 08:28:40 -07:00
Veselin Dobrev bdb627a5ae Merge branch 'master' into table-array
Resolved conflicts:
   general/array.hpp
2025-08-07 11:24:27 -07:00
Will Pazner 30d975fb9b Don't need specializations of mfem::Swap for Array or Array2D 2025-08-05 15:39:29 -07:00
Will Pazner dd68bbe9fd Use explicitly defaulted copy and move operations in Array2D 2025-08-05 15:38:43 -07:00
Will Pazner d1efdf93be Use default move constructor and assignment in Array<T> 2025-08-05 15:38:14 -07:00
Will Pazner e95d59cdf1 Implement mfem::Swap in terms of std::swap 2025-08-05 15:37:49 -07:00
Will Pazner b8c1b5c2fa Use explicitly defaulted move and copy member functions in DenseMatrix 2025-08-05 15:37:13 -07:00
Will Pazner a30af2ed43 Variable name consistency in Memory
Documentation was 'other', code was 'orig'. Changed to 'other'.
2025-08-05 15:36:39 -07:00
Julian Andrej c579f28b3f boundary integrator 2025-07-30 12:02:59 -07:00
Julian Andrej 947694ae63 add tensor::weight 2025-07-30 11:57:50 -07:00
Julian Andrej e97704d91c transformation bugfix for n x m tensor from memory 2025-07-30 11:57:21 -07:00
Will Pazner f5da02ce45 Fix bug in DenseTensor::NewMemoryAndSize 2025-07-29 13:11:13 -07:00
Julian Andrej 1f8f7e44bc remove dead code 2025-07-28 11:45:31 -07:00
John Camier 1aba4e0bb8 Merge branch 'master' into vector-pa-kernels 2025-07-04 08:02:46 -07:00
camierjs 03cf0edb5b Sync kernels 2025-07-02 11:04:40 -07:00
camierjs 0838384b78 Merge branch 'master' into vector-pa-kernels 2025-07-02 07:40:25 -07:00
Veselin Dobrev 3e321e6c9e Merge branch 'master' into woptim/leverage-radiuss-ci 2025-07-01 16:35:45 -07:00
John Camier 9c20900150 Merge branch 'master' into vector-pa-kernels 2025-07-01 09:51:22 -07:00
camierjs 52014e215e Sync kernels 2025-06-24 10:35:10 -07:00
camierjs 2daa072b55 Rename to GradTranspose3d kernel 2025-06-23 17:39:54 -07:00
John Camier ed31cb7bbd Merge branch 'master' into vector-pa-kernels 2025-06-23 16:11:19 -07:00
John Camier 7811775ab1 Merge branch 'master' into vector-pa-kernels 2025-06-23 07:46:38 -07:00
John Camier 5edabb64b7 Merge branch 'master' into vector-pa-kernels 2025-06-18 10:22:30 -07:00
camierjs 319b870ae1 Use future namespace for tensor
Fix couple of warnings
2025-06-16 08:47:39 -07:00
camierjs ba70885008 Merge branch 'master' into vector-pa-kernels 2025-06-16 08:46:29 -07:00
John Camier 2e706883d0 Merge branch 'master' into vector-pa-kernels 2025-06-13 08:49:04 -07:00
camierjs 614530599f Merge branch 'master' into vector-pa-kernels 2025-05-28 11:37:33 -07:00
camierjs acbe258939 CMake remove header 2025-05-19 15:25:09 -07:00
camierjs 7dba46021e Move kernel methods to fem/kernels.hpp 2025-05-19 15:13:20 -07:00
camierjs 189c201792 Fix shadowing 2025-05-19 11:17:18 -07:00
camierjs 0742475bc7 test_pa_kernels tags 2025-05-19 11:00:11 -07:00
camierjs 716c201fd3 GPU vector kernels fix 2025-05-19 10:35:40 -07:00
camierjs f9972a57c0 Simplify back SmemPAVectorMassApply2D 2025-05-19 08:27:44 -07:00
camierjs b3e9ddd846 WIN32 cmath 2025-05-18 15:56:41 -07:00
camierjs f3443f92c3 MSVC fix 2025-05-18 15:19:15 -07:00
camierjs c50905e492 MAX_T1D for WIN32 2025-05-18 14:52:38 -07:00
camierjs 7a37cc2448 MSVC header guards 2025-05-18 13:03:19 -07:00
camierjs 82c2e01678 Cleanup 2025-05-18 12:45:11 -07:00
camierjs 2ead488818 Pre SmemPAVectorMassApply2D cleanup 2025-05-18 12:33:12 -07:00
camierjs 190c4a091d WIP SmemPAVectorMassApply2D layouts 2025-05-18 11:13:18 -07:00
camierjs 6bb5f8ae0b Change pa data layout 2025-05-18 09:45:29 -07:00
camierjs 3b88332c98 Merge branch 'master' into vector-pa-kernels 2025-05-17 15:13:20 -07:00
camierjs ea25a9bb7b Remove PAVectorDiffusionApply kernels, fuse SDIM != DIM diffusion vector into same kernel 2025-05-17 15:12:57 -07:00
camierjs 62dd1aee3b Cleanup 2025-05-15 17:31:52 -07:00
camierjs e41c782974 Use explicit type names for registers 2025-05-15 15:02:12 -07:00
camierjs eb6843443f Simplify regs_t kernels types on CPU 2025-05-15 12:53:45 -07:00
camierjs 51c28ade7b Fix grad ii vs. jj matrix coeff
VectorDiagonalPA tests
2025-05-13 18:31:21 -07:00
camierjs 686feffc1b VectorDiffusionAddMultPA registered 2025-05-13 15:54:42 -07:00
camierjs 795a121f4e SmemPAVectorDiffusionApply3D 2025-05-13 15:33:01 -07:00
camierjs 6a04555556 SmemPAVectorDiffusionApply2D 2025-05-13 15:19:18 -07:00
camierjs 580dc6f50c Simplify VectorDiffusionIntegrator::AssemblePA 2025-05-13 12:26:17 -07:00
camierjs d27fa842b1 Pre AssemblePA 2D vector diffusion
simplify
2025-05-13 12:11:26 -07:00
camierjs 3fb13a3614 WIP 3D vector diffusion with sym mcoeff 2025-05-13 11:52:45 -07:00
camierjs 3c2f98bf64 test_vector_pa_integrator all tests passed 2025-05-13 10:59:12 -07:00
Hugh Carson f4bbc69ae6 Merge remote-tracking branch 'origin/master' into sjg/mesh-vis-dev 2025-05-13 12:18:24 -04:00
Hugh Carson 484249d94c Move coefficient fields to ParaViewDataCollection 2025-05-13 12:16:26 -04:00
camierjs 2a3c1c5bde VectorDiffusionIntegrator 2D cleanup 2025-05-13 08:55:04 -07:00
camierjs 031cd67c76 VectorDiffusionIntegrator 2D mcoeff 2025-05-12 18:13:00 -07:00
camierjs 7f7e5f61e2 All c, d 2025-05-12 17:49:24 -07:00
camierjs 582773213e WIP 0 1 2025-05-12 17:45:20 -07:00
camierjs 3086daecee WIP PAVectorDiffusionApply2D mcoeff 2025-05-12 17:16:27 -07:00
camierjs c707a4d3f6 jj=ii 2025-05-12 16:00:57 -07:00
camierjs ba4befcfb9 VectorDiffusionIntegrator wip Matrix coefficient 2025-05-12 13:59:05 -07:00
camierjs 72d28e0cdb VectorDiffusionIntegrator AssemblePA vector coefficients 2025-05-12 10:06:39 -07:00
camierjs 1c04593e79 VectorMassIntegrator cleanup 2025-05-12 07:19:29 -07:00
camierjs 0892f190cf VectorMassIntegrator 3D MatrixFunctionCoefficient 2025-05-12 07:15:40 -07:00
camierjs b08d12b5f1 VectorMassIntegrator 2D with MatrixFunctionCoefficients 2025-05-11 18:07:51 -07:00
camierjs d2e283419a VectorMassIntegrator PA specializations 2025-05-09 18:09:21 -07:00
camierjs 3949decb69 PAVectorMassApply2D with Eval2d 2025-05-09 17:09:18 -07:00
camierjs 93dbe1404a PAVectorMassApply3D with Eval3d 2025-05-09 16:32:21 -07:00
camierjs 68ae66c3ed vecmass PA setup 2025-05-09 15:51:24 -07:00
Tzanio Kolev 1206e9c575 Merge branch 'master' into qf-project-gf-fallback 2025-05-03 13:27:56 -07:00
Will Pazner 86b16c8989 Add fallback for QuadratureFunction::ProjectGridFunction
The (slower) fallback will be used when QuadratureInterpolator is not supported
for the finite element space.
2025-04-29 10:35:44 -07:00
Hugh Carson feda9f0125 make style 2025-04-22 14:36:58 -04:00
Hugh Carson 003b4175df Merge remote-tracking branch 'origin/master' into sjg/mesh-vis-dev 2025-04-22 14:20:21 -04:00
Hugh Carson fe83a192e9 Merge branch 'master' into sjg/mesh-vis-dev 2025-04-15 14:01:33 -04:00
Adrien M. BERNEDE 746fbca164 Merge branch 'woptim/leverage-radiuss-ci' into leverage-radiuss-ci--updates 2025-03-12 21:50:32 +01:00
Tzanio Kolev e7f71da564 Merge branch 'master' into woptim/leverage-radiuss-ci 2025-03-12 13:13:56 -07:00
Adrien M. BERNEDE 7d9048f9f0 Merge branch 'woptim/leverage-radiuss-ci' into leverage-radiuss-ci--updates 2025-03-07 17:49:19 +01:00
Adrien M. BERNEDE 3b2fe7ce7f Merge branch 'master' into woptim/leverage-radiuss-ci 2025-03-07 17:49:03 +01:00
Adrien M. BERNEDE dd64dcc10b Merge branch 'woptim/leverage-radiuss-ci' into leverage-radiuss-ci--updates 2025-02-19 14:23:21 +01:00
Adrien M. BERNEDE f89e89e9fa Merge branch 'master' into woptim/leverage-radiuss-ci 2025-02-19 14:21:35 +01:00
Adrien M. BERNEDE 2e27af385c Proposal to avoid .tioga_job_command override 2025-02-17 16:41:37 +01:00
Adrien M. BERNEDE f1025ecdeb Explain REGISTRY_TOKEN 2025-02-14 16:10:47 +01:00
Adrien M. BERNEDE 09b5dcab6d Merge branch 'woptim/leverage-radiuss-ci' into leverage-radiuss-ci--updates 2025-02-14 16:00:41 +01:00
Adrien M. BERNEDE 22e1f000b9 Update tests/gitlab readme and reproducibility files 2025-02-14 15:58:55 +01:00
Adrien M. BERNEDE 8884c31461 Update .gitlab/README.md 2025-02-14 15:26:51 +01:00
Adrien M. BERNEDE 04b868a47b Merge branch 'woptim/leverage-radiuss-ci' into leverage-radiuss-ci--updates 2025-02-13 17:39:01 +01:00
Adrien M. BERNEDE 8b99a6a847 Point at merge commit in mfem-uberenv 2025-02-11 11:35:33 +01:00
Adrien M. BERNEDE 4868222660 Update mfem-uberenv 2025-02-10 20:41:04 +01:00
Francis Giraldeau 82aa3a135c Fix Mesh::MakeSimplicial for surface in 3D
The Z dimension of a surface mesh was lost when converting it to simplex.

Add a test checking all vertices after the conversion. We create a special mesh for this specific case.

Close #4455
2025-02-08 23:01:09 -05:00
Veselin Dobrev f4ef788e40 Address some feedback 2025-02-06 14:42:15 -08:00
Veselin Dobrev 8f333155cc Updates for the branch 'woptim/leverage-radiuss-ci' 2025-02-05 13:08:25 -08:00
Hugh Carson 06d9613f59 Merge remote-tracking branch 'origin/master' into sjg/mesh-vis-dev 2025-01-21 14:59:56 -05:00
Adrien M. BERNEDE 5499eb8939 Merge branch 'master' into woptim/leverage-radiuss-ci 2025-01-16 15:34:19 +01:00
Adrien M. BERNEDE 5ccdd4b53d Merge branch 'master' into woptim/leverage-radiuss-ci 2025-01-13 12:21:54 +01:00
Will Pazner d0a994fa02 Add Array<T>::NewMemoryAndSize
Analogous to Vector::NewMemoryAndSize.

Implement DenseTensor::NewMemoryAndSize in terms of Array<T>::NewMemoryAndSize.
2024-11-22 10:50:12 -08:00
Will Pazner 6e4f6d11be Merge remote-tracking branch 'origin/master' into table-array 2024-11-22 10:41:03 -08:00
Adrien Bernede c50a55f05c Merge branch 'master' into woptim/leverage-radiuss-ci 2024-11-04 15:29:28 +01:00
Adrien Bernede 1c146ff523 Merge branch 'master' into woptim/leverage-radiuss-ci 2024-10-21 14:52:48 +02:00
Adrien M. BERNEDE 00d3b1ca49 Merge branch 'master' into woptim/leverage-radiuss-ci 2024-10-08 10:44:33 +02:00
Adrien M. BERNEDE 0f13bc76f5 Merge branch 'master' into woptim/leverage-radiuss-ci 2024-09-26 11:41:29 +02:00
Hugh Carson 0c2f087bb9 Merge branch 'master' into sjg/mesh-vis-dev 2024-09-13 13:08:42 -04:00
Adrien Bernede b377eac8cd Merge branch 'master' into woptim/leverage-radiuss-ci 2024-09-12 11:13:57 +02:00
Adrien M. BERNEDE 175a4302c3 Run baseline sub-pipeline as soon as machine is verified 2024-08-08 23:36:29 +02:00
Adrien M. BERNEDE e31f662f8b Fix typo 2024-08-08 18:10:43 +02:00
Adrien M. BERNEDE 903ddbf97d Update build_and_test script to populate Gitlab registry as build cache 2024-08-08 18:01:31 +02:00
Adrien M. BERNEDE 9c44c45a7e Deactivate rocm 5.7 job on tioga 2024-08-06 18:09:27 +02:00
Adrien M. BERNEDE 4da44357d0 Reduce allocation duration on ruby 2024-08-05 15:50:30 +02:00
Adrien M. BERNEDE bf55c1cc46 Only one node for CI jobs, so that /dev/shm exists for any sub job 2024-08-05 12:56:56 +02:00
Adrien M. BERNEDE 96d188ac12 Merge branch 'master' into woptim/leverage-radiuss-ci 2024-08-05 11:30:11 +02:00
Adrien M. BERNEDE c0a1e91237 Revert "TEMP: pci queue not working on tioga"
This reverts commit 788c1a15da.
2024-08-05 10:43:26 +02:00
Adrien M. BERNEDE 788c1a15da TEMP: pci queue not working on tioga 2024-08-02 14:49:26 +02:00
Adrien M. BERNEDE bb72f2c0c6 mfem-uberenv: Attempt at enforcing coherent rocm compiler in rocm stack 2024-07-30 16:53:26 +02:00
Adrien M. BERNEDE edeb7ab0f9 Attempt to use cray-libsci through modules in Spack 2024-07-30 16:19:14 +02:00
Adrien M. BERNEDE 55f954e6c3 mfem-uberenv: fix 2024-07-30 15:03:35 +02:00
Adrien M. BERNEDE 587ef0ef47 mfem-uberenv: Adding cray-libsci as sole blas / lapack provider 2024-07-30 14:57:00 +02:00
Adrien M. BERNEDE 23a18a46b3 Fix logic to prevent pushing to autotest repo 2024-07-29 12:36:53 +02:00
Adrien M. BERNEDE 8e00d7c191 Fix logic to prevent pushing to autotest repo 2024-07-29 12:20:30 +02:00
Adrien M. BERNEDE b242fde7ec Fix reproducer 2024-07-29 12:03:08 +02:00
Adrien M. BERNEDE a09c9ef266 Fix reproducer 2024-07-29 11:37:44 +02:00
Adrien Bernede a2e5ffc9db Fix reproducer 2024-07-26 18:15:27 +02:00
Adrien Bernede 6bebe4caf9 Update mfem uberenv with missing MFEM package patches 2024-07-26 14:55:05 +02:00
Adrien M. BERNEDE 97e554fb13 Update mfem-uberenv: do not use spack compiler 2024-07-25 17:25:20 +02:00
Adrien M. BERNEDE 3a30286567 Revert unecessary changes, fix alloc command using xargs in shared ci 2024-07-23 16:53:14 +02:00
Adrien M. BERNEDE bdc919f084 fix hope 2024-07-23 12:36:38 +02:00
Adrien M. BERNEDE 8cc6821cad hope 2024-07-23 12:33:52 +02:00
Adrien M. BERNEDE f2851bbb61 ... 2024-07-23 12:24:08 +02:00
Adrien M. BERNEDE 49e19c2414 ??? 2024-07-23 12:15:36 +02:00
Adrien M. BERNEDE 73cbdc75e1 New attempt 2024-07-23 12:08:07 +02:00
Adrien M. BERNEDE 81b91619ae Apply improved variable handling 2024-07-23 11:55:25 +02:00
Adrien M. BERNEDE 070ab14951 Looking for a fix (4) 2024-07-22 18:01:43 +02:00
Adrien M. BERNEDE bbbb50ff5b Looking for a fix (ter) 2024-07-22 17:32:04 +02:00
Adrien M. BERNEDE 204c9c60d1 Looking for a fix (bis) 2024-07-22 17:25:19 +02:00
Adrien M. BERNEDE e7f8898546 Looking for a fix 2024-07-22 17:16:46 +02:00
Adrien M. BERNEDE 9c64303464 Remove extra quotes 2024-07-22 16:02:51 +02:00
Adrien M. BERNEDE 28bf478b68 Prevent early variable expansion 2024-07-22 15:55:29 +02:00
Adrien M. BERNEDE 9b0edad337 Fix script name 2024-07-22 15:40:22 +02:00
Adrien M. BERNEDE d6cbd76c46 Fix no jobs imported from radiuss-spack-configs 2024-07-22 14:06:06 +02:00
Adrien M. BERNEDE 2c8a7548cb Complete moving to Shared CI 2024-07-22 13:02:54 +02:00
Adrien M. BERNEDE 33ce5f1765 WIP: moving to RADIUSS Shared CI 2024-07-19 09:18:21 +02:00
Adrien M. BERNEDE 84ce97a114 Merge branch 'master' into woptim/ci-tioga 2024-07-18 09:34:39 +02:00
Adrien M. BERNEDE c508781257 Remove Corona from CI machines, replaced by Tioga 2024-07-18 09:33:35 +02:00
Adrien M. BERNEDE df994a4890 Update mfem-uberenv: add external installs for rocsparse and hipsparse 2024-07-17 16:01:56 +02:00
Adrien M. BERNEDE db63f71f40 Increase number of nodes for ruby 2024-07-17 16:01:03 +02:00
Adrien M. BERNEDE c4b05bd83e Update mfem-uberenv: fix spack configs path 2024-07-17 11:55:16 +02:00
Adrien M. BERNEDE 6155e148a7 Do not use upstream spack 2024-07-17 11:30:02 +02:00
Adrien M. BERNEDE b8e0bd0692 Update Uberenv with Tioga configs 2024-07-17 10:49:16 +02:00
Adrien M. BERNEDE 675ecb994f Use Python3 to run uberenv 2024-07-17 10:42:08 +02:00
Adrien M. BERNEDE e3d5c35614 Update Uberenv, spack and spack configs 2024-07-17 10:36:38 +02:00
Adrien M. BERNEDE 953a063626 Merge branch 'master' into woptim/ci-tioga 2024-07-17 10:17:36 +02:00
Adrien M. BERNEDE 535e39bbb1 Update build_and_test script with required changes to work well with new mfem-uberenv 2024-07-16 17:02:37 +02:00
Adrien M. BERNEDE cfa41e0d9c Enforce quotes is corona job command 2024-07-15 12:11:23 +02:00
Adrien M. BERNEDE 0ee13cc107 Fix SPECS in corona and tioga CI 2024-07-15 12:07:26 +02:00
Adrien M. BERNEDE 4635107f2b Attempt at forcing quotes 2024-07-15 11:34:46 +02:00
Adrien M. BERNEDE e5c1b4412c Unify allocation process between corona and tioga 2024-07-12 16:47:12 +02:00
Adrien M. BERNEDE a3cd66cf84 Fix: remove nonexistent partition 2024-07-12 16:13:02 +02:00
Adrien M. BERNEDE 7abddc527a Get corona and tioga jobs to run by default (ON_<MACHINE> var) 2024-07-12 15:59:06 +02:00
Adrien Bernede 6afec618be Merge branch 'master' into woptim/ci-tioga 2024-07-12 12:24:39 +02:00
Adrien Bernede fff5a50f21 Merge branch 'master' into woptim/ci-tioga 2024-06-25 15:04:49 +02:00
Will Pazner 8c8b79fdd9 Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	general/table.cpp
#	general/table.hpp
2024-06-24 11:08:08 -07:00
Will Pazner 326acce08c Replace double with real_t in BatchLUFactor 2024-06-02 14:03:46 -07:00
Will Pazner 3d78f073ae Merge remote-tracking branch 'origin/master' into table-array
# Conflicts:
#	general/table.hpp
#	linalg/densemat.cpp
#	linalg/densemat.hpp
2024-06-02 13:25:08 -07:00
Tom Stitt 257ab6837b fix MFEM_PERF_SCOPE with latest caliper 2024-05-31 13:09:42 -07:00
Adrien M. BERNEDE e8b229bb54 Merge branch 'woptim/gitlab-updates' into woptim/ci-tioga 2024-05-23 11:26:12 +02:00
Adrien M. BERNEDE f7db5cbc24 Merge branch 'woptim/gitlab-updates' into woptim/ci-tioga 2024-05-23 11:19:07 +02:00
Adrien M. BERNEDE 1a1df8de48 Update relationship between jobs for correct ordering 2024-05-21 11:32:43 +02:00
Adrien M. BERNEDE 3f7fddb188 A first attempt at adding CI on tioga 2024-05-21 11:24:02 +02:00
Sebastian Grimberg 94ccc4f364 Fix for single precision builds 2024-05-06 12:20:30 -07:00
Sebastian Grimberg d86da2d2c5 Merge branch 'master' into sjg/mesh-vis-dev 2024-05-06 12:17:30 -07:00
Sebastian Grimberg 46d81f4742 Merge branch 'master' into sjg/mesh-vis-dev 2024-04-04 11:31:21 -07:00
Will Pazner 9f1c1b811a Merge branch 'densemat-array' into table-array 2024-01-30 12:24:41 -08:00
Will Pazner 843365752f Fix Array copy/move bug in DeviceConformingProlongationOperator 2024-01-02 13:29:39 -08:00
Will Pazner cc21c6b859 Add move assignment operator to Array 2024-01-02 11:40:37 -08:00
Will Pazner d788f190ea Use Array<double> in DenseMatrix and DenseTensor 2023-12-13 21:39:36 -08:00
Will Pazner d0bb1a86ed Delete specialization of mfem::Swap for Table 2023-12-13 21:11:08 -08:00
Will Pazner fafec06958 Use std::move in mfem::Swap 2023-12-13 21:10:49 -08:00
Will Pazner e52a2b4dcd Use Array<int> instead of Memory<int> in Table
Also make some small Doxygen edits
2023-12-13 14:13:38 -08:00
Sebastian Grimberg 81786e6671 Merge branch 'master' into sjg/mesh-vis-dev 2023-12-12 18:17:37 -08:00
Sebastian Grimberg 70a1fd7679 Revert addition of length scale for mesh coordinates in mesh output 2023-11-06 07:58:57 -08:00
Sebastian Grimberg feefc7d62a Merge branch 'master' into sjg/mesh-vis-dev 2023-11-06 07:52:01 -08:00
Sebastian Grimberg 3061b78689 Merge branch 'master' into sjg/mesh-vis-dev 2023-09-29 07:27:12 -07:00
Sebastian Grimberg a4f4e00ba0 Merge branch 'master' into sjg/mesh-vis-dev 2023-09-06 11:30:19 -07:00
Sebastian Grimberg ba938e8a97 Fix from merge 2023-09-06 11:30:17 -07:00
Sebastian Grimberg 1cc174cd7f Merge branch 'master' into sjg/mesh-vis-dev 2023-08-17 12:17:56 -07:00
Sebastian Grimberg db840eb015 Merge branch 'master' into sjg/mesh-vis-dev 2023-07-30 18:19:40 -07:00
Hugh Carson 77d32658fa Remove changes to test_operator.cpp 2023-07-21 13:58:34 -04:00
Hugh Carson d55cacbd6e make style 2023-07-21 10:53:15 -04:00
Hugh Carson c42e6e7687 Merge remote-tracking branch 'origin/master' into sjg/mesh-vis-dev 2023-07-21 10:34:59 -04:00
Sebastian Grimberg a96d4cdf33 Merge branch 'master' into sjg/mesh-vis-dev 2023-06-26 15:37:39 -07:00
Sebastian Grimberg ec402f6b24 Merge branch 'master' into sjg/mesh-vis-dev 2023-06-15 14:37:50 -07:00
Sebastian Grimberg 2a6f96bb90 Merge branch 'master' into sjg/mesh-vis-dev 2023-05-16 18:40:05 -07:00
Sebastian Grimberg 6398fa198b Merge branch 'master' into sjg/mesh-vis-dev 2023-05-02 17:45:44 -07:00
Sebastian Grimberg f1f1b420b7 Merge branch 'master' into sjg/mesh-vis-dev 2023-04-18 11:10:18 -07:00
hughcars ad989ab425 Merge branch 'master' into sjg/mesh-vis-dev 2023-03-30 15:04:25 -04:00
Sebastian Grimberg 7caa15acde Address PR comments 2023-03-15 16:23:40 -07:00
Sebastian Grimberg 04009fedb0 make style 2023-03-06 10:19:59 -08:00
Sebastian Grimberg 098db1b70c Mesh and DataCollection upgrades 2023-03-03 17:28:25 -08:00
272 changed files with 30397 additions and 11566 deletions
+17 -10
View File
@@ -19,9 +19,15 @@ CMakeFiles/
# Clangd server cache
*.cache*
#vscode settings
/.vscode/
# Backup files
*~
# clangd index
/.cache/
# Default install location
/mfem/
@@ -79,6 +85,7 @@ examples/sol_u.*
examples/sol_p.*
examples/sol_r.*
examples/sol_i.*
examples/sol_z.*
examples/ex6p-checkpoint.*
examples/order.*
examples/ex9.mesh
@@ -269,10 +276,8 @@ miniapps/meshing/refined.mesh
miniapps/meshing/bounding-box*
miniapps/meshing/jacobian-determinant*
miniapps/mtop/parheat
miniapps/mtop/ParHeat/*
miniapps/mtop/seqheat
miniapps/mtop/SeqHeat/*
miniapps/mtop/ParaView/
miniapps/mtop/mtop_test_iso_elasticity
miniapps/autodiff/paradiff
miniapps/autodiff/seqadiff
@@ -304,8 +309,11 @@ miniapps/nurbs/nurbs_printfunc
miniapps/nurbs/nurbs_patch_ex1
miniapps/nurbs/nurbs_curveint
miniapps/nurbs/nurbs_surface
miniapps/nurbs/nurbs_stokes
miniapps/nurbs/nurbs_vector_diffusion
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/*.mesh
miniapps/nurbs/sol_?.gf
miniapps/nurbs/sol.*
miniapps/nurbs/mode_*
@@ -313,16 +321,12 @@ miniapps/nurbs/Example1*
miniapps/nurbs/Example3*
miniapps/nurbs/Example5*
miniapps/nurbs/Solenoidal*
miniapps/nurbs/Stokes*
miniapps/nurbs/VectorDiffusion*
miniapps/nurbs/ParaView
miniapps/nurbs/sin-fit.mesh
miniapps/nurbs/ex5.mesh
miniapps/nurbs/exsol.mesh
miniapps/nurbs/CurveInt
miniapps/nurbs/nurbs_naca_cmesh
miniapps/nurbs/naca-cmesh.mesh
miniapps/nurbs/glvis_naca-cmesh.mesh
miniapps/nurbs/Naca_cmesh
miniapps/nurbs/*-Surface.mesh
miniapps/performance/ex1
miniapps/performance/ex1p
@@ -414,6 +418,9 @@ miniapps/tribol/contact-patch-test
miniapps/diag-smoothers/abs-l1-jacobi
miniapps/diag-smoothers/mg-abs-l1-jacobi
miniapps/contact/contact
miniapps/contact/ParaView
# Unit test binary and outputs
tests/unit/output_meshes
tests/unit/unit_tests
+529 -70
View File
@@ -9,91 +9,550 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# DESCRIPTION:
###############################################################################
# General GitLab pipelines configurations for supercomputers and Linux clusters
# at Lawrence Livermore National Laboratory (LLNL). This entire pipeline is
# LLNL-specific!
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# The pipeline is divided into stages. Usually, jobs in a given stage wait for
# the preceding stages to complete before to start. However, we sometimes use
# the "needs" keyword and express the DAG of jobs for more efficiency.
# - We use setup and setup_baseline phases to download content outside of mfem
# directory.
# - Allocate/Release is where Dane resource are allocated/released once for all.
# - Build and Test is where we build and MFEM for multiple toolchains.
# - Baseline_checks gathers baseline-type test suites execution
# - Baseline_publish, only available on master, allows to update baseline
# results
stages:
- sub-pipelines
# at Lawrence Livermore National Laboratory (LLNL).
# This entire pipeline is LLNL-specific
#
# Important note: This file is a template provided by llnl/radiuss-shared-ci.
# Remains to set variable values, change the reference to the radiuss-shared-ci
# repo, opt-in and out optional features. The project can then extend it with
# additional stages.
#
# In addition, each project should copy over and complete:
# - .gitlab/custom-jobs-and-variables.yml
# - .gitlab/subscribed-pipelines.yml
#
# The jobs should be specified in a file local to the project,
# - .gitlab/jobs/${CI_MACHINE}.yml
# or generated (see LLNL/Umpire for an example).
###############################################################################
# MAP OF GITLAB CI
#######################
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# File dependencies: direct, through jobs, through variables
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# .gitlab-ci.yml
# ├── .build-and-test [job]
# │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ ├── .custom_job [job]
# │ │ ├── .reproducer_vars [job]
# │ │ ├── .report_job_success [job]
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ │ ├── .report_job_failure [job]
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ │ └── JOB_CMD [var]
# │ │ └── tests/gitlab/build_and_test [script]
# │ │ └── tests/gitlab/get_mfem_uberenv [script]
# │ ├── <radiuss-shared-ci>/pipelines/matrix.yml [conditional]
# │ │ ├── .on_matrix [job]
# │ │ ├── .matrix_reproducer_init [job]
# │ │ ├── .matrix_reproducer_vars [job]
# │ │ ├── .matrix_reproducer_job [job]
# │ │ ├── .matrix_job_command [job]
# │ │ └── .job_on_matrix [job]
# │ ├── <radiuss-shared-ci>/pipelines/dane.yml [conditional]
# │ │ ├── .on_dane [job]
# │ │ ├── .dane_reproducer_init [job]
# │ │ ├── .dane_reproducer_vars [job]
# │ │ ├── .dane_reproducer_job [job]
# │ │ ├── .dane_job_command [job]
# │ │ ├── .job_on_dane [job]
# │ │ ├── allocate_resources [job]
# │ │ └── release_resources [job]
# │ ├── <radiuss-shared-ci>/pipelines/tioga.yml [conditional]
# │ │ ├── .on_tioga [job]
# │ │ ├── .tioga_reproducer_init [job]
# │ │ ├── .tioga_reproducer_vars [job]
# │ │ ├── .tioga_reproducer_job [job]
# │ │ ├── .tioga_job_command [job]
# │ │ ├── .job_on_tioga [job]
# │ │ ├── allocate_resources [job]
# │ │ └── release_resources [job]
# │ ├── <artifact>/matrix-jobs.yml [conditional, from 'generate-job-lists']
# │ │ ├── .gitlab/jobs/matrix.yml
# │ │ │ ├── .matrix_reproducer_vars [job]
# │ │ │ ├── setup [job]
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ │ ├── opt_mpi_cuda_gcc [job]
# │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc [job]
# │ │ └── .gitlab/jobs/matrix-reports.yml [used conditionally]
# │ │ ├── report_job_success
# │ │ └── report_job_failure
# │ ├── <artifact>/dane-jobs.yml [conditional, from 'generate-job-lists']
# │ │ ├── .gitlab/jobs/dane.yml
# │ │ │ ├── .dane_reproducer_vars [job]
# │ │ │ ├── setup [job]
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ │ ├── debug_ser_gcc_10 [job]
# │ │ │ ├── debug_par_gcc_10 [job]
# │ │ │ ├── opt_ser_gcc_10 [job]
# │ │ │ ├── opt_par_gcc_10 [job]
# │ │ │ ├── opt_par_gcc_10_sundials [job]
# │ │ │ ├── opt_par_gcc_10_petsc [job]
# │ │ │ └── opt_par_gcc_10_pumi [job]
# │ │ └── .gitlab/jobs/dane-reports.yml [used conditionally]
# │ │ ├── report_job_success
# │ │ └── report_job_failure
# │ └── <artifact>/tioga-jobs.yml [conditional, from 'generate-job-lists']
# │ ├── .gitlab/jobs/tioga.yml
# │ │ ├── .tioga_reproducer_vars [job]
# │ │ ├── setup [job]
# │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ └── cce_16_0_1 [job]
# │ └── .gitlab/jobs/tioga-reports.yml [used conditionally]
# │ ├── report_job_success
# │ └── report_job_failure
# └── .gitlab/subscribed-pipelines.yml
# ├── .machine-check [job]
# ├── generate-job-lists [job]
# ├── dane-up-check [job]
# ├── dane-build-and-test [job]
# ├── dane-baseline [job]
# │ └── .gitlab/dane-baseline.yml
# │ ├── .on_dane [job]
# │ ├── baselinecheck_mfem_intel_dane [job]
# │ │ └── .gitlab/scripts/baseline [script]
# │ ├── cleanup [job]
# │ ├── report_baseline [job]
# │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ ├── baselinepublish_mfem_dane [job]
# │ │ └── .gitlab/scripts/rebaseline [script]
# │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ └── <same as above: see .gitlab-ci.yml/.build-and-test>
# │ └── .gitlab/configs/setup-baseline.yml
# │ └── setup_baseline [job]
# ├── tioga-up-check [job]
# ├── tioga-build-and-test [job]
# ├── matrix-up-check [job]
# └── matrix-build-and-test [job]
#
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# File tree hierarchy with file contents highlights
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# In addition to the files in the MFEM repo, the Gitlab CI uses files from the
# radiuss/radiuss-shared-ci project, see below, after the <mfem-root> tree.
#
# <mfem root>
# ├── .gitlab-ci.yml [this file]
# │ ├── <jobs>
# │ │ └── .build-and-test
# │ ├── <included files>
# │ │ ├── .gitlab/subscribed-pipelines.yml
# │ │ ├── .gitlab/custom-jobs-and-variables.yml [by ".build-and-test"]
# │ │ ├── <artifact> [by ".build-and-test"]
# │ │ │ ├── artifact: '${CI_MACHINE}-jobs.yml'
# │ │ │ └── job: 'generate-job-lists'
# │ │ └── <external> [by ".build-and-test"]
# │ │ ├── project: 'radiuss/radiuss-shared-ci'
# │ │ ├── ref: 'v2025.09.1'
# │ │ └── file: 'pipelines/${CI_MACHINE}.yml'
# │ └── <defined variables>
# │ ├── CUSTOM_CI_BUILDS_DIR
# │ ├── USER_CI_TOP_DIR
# │ ├── SHARED_REPOS_DIR
# │ ├── AUTOTEST_ROOT
# │ ├── MFEM_DATA_DIR
# │ ├── AUTOTEST
# │ ├── AUTOTEST_COMMIT
# │ ├── REBASELINE
# │ ├── GITHUB_PROJECT_NAME
# │ └── GITHUB_PROJECT_ORG
# ├── .gitlab
# │ ├── configs
# │ │ └── setup-baseline.yml
# │ │ ├── <jobs>
# │ │ │ └── setup_baseline
# │ │ └── <used variables>
# │ │ ├── MACHINE_NAME
# │ │ ├── REBASELINE
# │ │ ├── AUTOTEST
# │ │ ├── AUTOTEST_COMMIT
# │ │ ├── BUILD_ROOT
# │ │ ├── TPLS_REPO
# │ │ ├── TESTS_REPO
# │ │ ├── AUTOTEST_ROOT
# │ │ └── AUTOTEST_REPO
# │ ├── jobs
# │ │ ├── matrix-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_matrix
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── matrix.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── .matrix_reproducer_vars
# │ │ │ │ ├── setup
# │ │ │ │ ├── opt_mpi_cuda_gcc
# │ │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc
# │ │ │ ├── <used jobs>
# │ │ │ │ ├── .reproducer_vars
# │ │ │ │ ├── .on_matrix
# │ │ │ │ └── .job_on_matrix
# │ │ │ ├── <included and used files>
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ │ └── <defined variables>
# │ │ │ └── SPEC
# │ │ ├── dane-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_dane
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── dane.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── .dane_reproducer_vars
# │ │ │ │ ├── setup
# │ │ │ │ ├── debug_ser_gcc_10
# │ │ │ │ ├── debug_par_gcc_10
# │ │ │ │ ├── opt_ser_gcc_10
# │ │ │ │ ├── opt_par_gcc_10
# │ │ │ │ ├── opt_par_gcc_10_sundials
# │ │ │ │ ├── opt_par_gcc_10_petsc
# │ │ │ │ └── opt_par_gcc_10_pumi
# │ │ │ ├── <used jobs>
# │ │ │ │ ├── .reproducer_vars
# │ │ │ │ ├── .on_dane
# │ │ │ │ └── .job_on_dane
# │ │ │ ├── <included and used files>
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ │ └── <defined variables>
# │ │ │ ├── SPEC
# │ │ │ └── THREADS
# │ │ ├── tioga-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_tioga
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ └── tioga.yml
# │ │ ├── <jobs>
# │ │ │ ├── .tioga_reproducer_vars
# │ │ │ ├── setup
# │ │ │ └── opt_mpi_rocm_hypre_rocm
# │ │ ├── <used jobs>
# │ │ │ ├── .reproducer_vars
# │ │ │ ├── .on_tioga
# │ │ │ └── .job_on_tioga
# │ │ ├── <included and used files>
# │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ └── <defined variables>
# │ │ ├── SPEC
# │ │ └── THREADS
# │ ├── scripts
# │ │ ├── baseline
# │ │ │ └── <used variables>
# │ │ │ ├── BASELINE_TEST
# │ │ │ ├── SYS_TYPE
# │ │ │ ├── MACHINE_NAME
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── ARTIFACTS_DIR
# │ │ │ ├── BUILD_ROOT
# │ │ │ └── TPLS_DIR
# │ │ ├── git_try_to_push
# │ │ ├── rebaseline
# │ │ │ └── <used variables>
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── ARTIFACTS_DIR
# │ │ │ ├── SYS_TYPE
# │ │ │ ├── BUILD_ROOT
# │ │ │ ├── MACHINE_NAME
# │ │ │ └── CI_PIPELINE_ID
# │ │ ├── report_build_and_test
# │ │ │ ├── <used files>
# │ │ │ │ ├── .gitlab/scripts/safe_create_rundir
# │ │ │ │ └── .gitlab/scripts/git_try_to_push
# │ │ │ └── <used variables>
# │ │ │ ├── AUTOTEST_ROOT
# │ │ │ ├── CI_COMMIT_REF_SLUG
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── CI_PIPELINE_URL
# │ │ │ ├── AUTOTEST_COMMIT
# │ │ │ └── CI_MACHINE
# │ │ └── safe_create_rundir
# │ ├── custom-jobs-and-variables.yml
# │ │ ├── <jobs>
# │ │ │ ├── .custom_job
# │ │ │ ├── .reproducer_vars
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── <used files>
# │ │ │ ├── tests/gitlab/build_and_test [in JOB_CMD]
# │ │ │ └── .gitlab/scripts/report_build_and_test [by .report_job_*]
# │ │ ├── <defined variables>
# │ │ │ ├── JOB_CMD
# │ │ │ ├── BUILD_ROOT
# │ │ │ ├── ALLOC_NAME
# │ │ │ ├── TPLS_REPO
# │ │ │ ├── TESTS_REPO
# │ │ │ ├── AUTOTEST_REPO
# │ │ │ ├── MFEM_DATA_REPO
# │ │ │ ├── ARTIFACTS_DIR: artifacts
# │ │ │ ├── SLURM_OVERLAP: 1
# │ │ │ ├── DANE_SHARED_ALLOC
# │ │ │ ├── DANE_JOB_ALLOC
# │ │ │ ├── TIOGA_SHARED_ALLOC
# │ │ │ ├── TIOGA_JOB_ALLOC
# │ │ │ └── MATRIX_JOB_ALLOC
# │ │ └── <used variables>
# │ │ ├── SPEC
# │ │ ├── BUILD_ROOT
# │ │ └── ...
# │ ├── dane-baseline.yml
# │ │ ├── <jobs>
# │ │ │ ├── .on_dane
# │ │ │ ├── baselinecheck_mfem_intel_dane
# │ │ │ ├── cleanup
# │ │ │ ├── report_baseline
# │ │ │ └── baselinepublish_mfem_dane
# │ │ ├── <included and used files>
# │ │ │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ │ ├── .gitlab/configs/setup-baseline.yml
# │ │ │ ├── .gitlab/scripts/rebaseline
# │ │ │ ├── .gitlab/scripts/baseline
# │ │ │ └── .gitlab/scripts/git_try_to_push
# │ │ ├── <defined variables>
# │ │ │ ├── BASELINE_TEST: baseline
# │ │ │ ├── MACHINE_NAME: dane
# │ │ │ ├── TPLS_DIR
# │ │ │ └── export MFEM_TEST_NP
# │ │ └── <used variables>
# │ │ ├── ON_DANE
# │ │ ├── AUTOTEST [defined by .gitlab-ci.yml]
# │ │ ├── BUILD_ROOT [defined by custom-jobs-and-variables.yml]
# │ │ ├── TPLS_DIR [defined by this file]
# │ │ ├── ARTIFACTS_DIR [defined by custom-jobs-and-variables.yml]
# │ │ ├── MACHINE_NAME [defined by this file]
# │ │ ├── AUTOTEST_COMMIT [defined by .gitlab-ci.yml]
# │ │ ├── AUTOTEST_ROOT [defined by .gitlab-ci.yml]
# │ │ ├── BASELINE_TEST [defined by this file]
# │ │ └── REBASELINE [defined by .gitlab-ci.yml]
# │ └── subscribed-pipelines.yml
# │ ├── <jobs>
# │ │ ├── .machine-check
# │ │ ├── generate-job-lists
# │ │ ├── dane-up-check
# │ │ ├── dane-build-and-test
# │ │ ├── dane-baseline
# │ │ ├── tioga-up-check
# │ │ ├── tioga-build-and-test
# │ │ ├── matrix-up-check
# │ │ └── matrix-build-and-test
# │ ├── <used jobs>
# │ │ └── .build-and-test [from ".gitlab-ci.yml"]
# │ ├── <included files>
# │ │ └── .gitlab/dane-baseline.yml [by "dane-baseline"]
# │ └── <used variables>
# │ ├── GITHUB_PROJECT_ORG
# │ ├── GITHUB_PROJECT_NAME
# │ ├── AUTOTEST
# │ ├── AUTOTEST_COMMIT
# │ └── REBASELINE
# └── tests
# ├── gitlab
# │ ├── build_and_test
# │ │ ├── <builds and tests a given MFEM spec with uberenv>
# │ │ ├── <used files>
# │ │ │ ├── tests/uberenv/uberenv.py [deps mode, cloned]
# │ │ │ └── tests/gitlab/get_mfem_uberenv [deps mode]
# │ │ └── <used variables>
# │ │ ├── SYS_TYPE
# │ │ ├── THREADS [num. parallel jobs to build MFEM]
# │ │ ├── MODULE_LIST [modules to load]
# │ │ ├── CI_JOB_ID
# │ │ ├── USE_DEV_SHM
# │ │ ├── SPACK_DEBUG
# │ │ ├── DEBUG_MODE
# │ │ ├── REGISTRY_TOKEN
# │ │ ├── CI_REGISTRY_USER (defined by Gitlab)
# │ │ ├── USER
# │ │ ├── CI_REGISTRY_IMAGE (defined by Gitlab)
# │ │ └── CI_JOB_TOKEN (defined by Gitlab)
# │ ├── build_and_test_setup
# │ │ ├── <updates MFEM_DATA_REPO and AUTOTEST_REPO using locks>
# │ │ └── <used variables>
# │ │ ├── MFEM_DATA_REPO
# │ │ ├── SHARED_REPOS_DIR
# │ │ ├── AUTOTEST_REPO
# │ │ └── AUTOTEST_ROOT
# │ └── get_mfem_uberenv
# │ ├── <github.com/mfem/mfem-uberenv.git -> tests/uberenv>
# │ └── <defines the uberenv hash to use>
# └── uberenv [cloned by tests/gitlab/get_mfem_uberenv]
# └── uberenv.py
#
# <root of radiuss/radiuss-shared-ci, ref: 'v2025.09.1'>
# └── pipelines
# ├── matrix.yml
# │ ├── <jobs>
# │ │ ├── .on_matrix
# │ │ ├── .matrix_reproducer_init
# │ │ ├── .matrix_reproducer_vars
# │ │ ├── .matrix_reproducer_job
# │ │ ├── .matrix_job_command
# │ │ └── .job_on_matrix
# │ ├── <used jobs>
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# │ └── <used variables>
# │ ├── ON_MATRIX
# │ ├── ADVANCED_JOB
# │ ├── ALL_TARGETS
# │ ├── SYS_TYPE
# │ ├── LLNL_SERVICE_USER
# │ ├── USER
# │ ├── GITHUB_PROJECT_NAME
# │ ├── GITHUB_PROJECT_ORG
# │ ├── MATRIX_JOB_ALLOC
# │ └── JOB_CMD
# ├── dane.yml
# │ ├── <jobs>
# │ │ ├── .on_dane
# │ │ ├── .dane_reproducer_init
# │ │ ├── .dane_reproducer_vars
# │ │ ├── .dane_reproducer_job
# │ │ ├── .dane_job_command
# │ │ ├── .job_on_dane
# │ │ ├── allocate_resources
# │ │ └── release_resources
# │ ├── <used jobs>
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# │ ├── <defined variables>
# │ │ └── export JOBID
# │ └── <used variables>
# │ ├── ON_DANE
# │ ├── ADVANCED_JOB
# │ ├── ALL_TARGETS
# │ ├── SYS_TYPE
# │ ├── LLNL_SERVICE_USER
# │ ├── USER
# │ ├── GITHUB_PROJECT_NAME
# │ ├── GITHUB_PROJECT_ORG
# │ ├── DANE_JOB_ALLOC
# │ ├── JOB_CMD
# │ ├── JOBID
# │ ├── ALLOC_NAME
# │ └── DANE_SHARED_ALLOC
# └── tioga.yml
# ├── <jobs>
# │ ├── .on_tioga
# │ ├── .tioga_reproducer_init
# │ ├── .tioga_reproducer_vars
# │ ├── .tioga_reproducer_job
# │ ├── .tioga_job_command
# │ ├── .job_on_tioga
# │ ├── allocate_resources
# │ └── release_resources
# ├── <used jobs>
# │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# ├── <defined variables>
# │ └── PROXY
# └── <used variables>
# ├── ON_TIOGA
# ├── ADVANCED_JOB
# ├── ALL_TARGETS
# ├── SYS_TYPE
# ├── LLNL_SERVICE_USER
# ├── USER
# ├── GITHUB_PROJECT_NAME
# ├── GITHUB_PROJECT_ORG
# ├── TIOGA_JOB_ALLOC
# ├── JOB_CMD
# ├── PROXY
# ├── ALLOC_NAME
# └── TIOGA_SHARED_ALLOC
###############################################################################
# We define the following GitLab pipeline variables:
variables:
##### LC GITLAB CONFIGURATION
CUSTOM_CI_BUILDS_DIR: "/usr/workspace/mfem/gitlab-runner"
##### PROJECT VARIABLES
USER_CI_TOP_DIR: "${CUSTOM_CI_BUILDS_DIR}/${GITLAB_USER_LOGIN}"
SHARED_REPOS_DIR: "${USER_CI_TOP_DIR}/repos"
AUTOTEST_ROOT: "${SHARED_REPOS_DIR}"
# MFEM_DATA_DIR is setup in '.gitlab/configs/setup-build-and-test.yml' and
# used in '.gitlab/configs/<machine>-config.yml':
MFEM_DATA_DIR: "${SHARED_REPOS_DIR}/mfem-data"
# AUTOTEST: enable (ON/YES) or disable (any other value) test reporting. See
# also AUTOTEST_COMMIT.
AUTOTEST: "OFF"
# AUTOTEST_COMMIT: used only when AUTOTEST is set to ON/YES.
# * If AUTOTEST_COMMIT is set to ON/YES, reporting jobs will commit their
# files to the MFEM/autotest repo.
# * If AUTOTEST_COMMIT is NOT set to ON/YES, reporting jobs will NOT commit
# their files to the MFEM/autotest repo. Instead they will just show the
# contents of the report files and remove them.
AUTOTEST_COMMIT: "ON"
# REBASELINE:
# Defines the default choice for updating the saved baseline results. By default
# the baseline can only be updated from the master branch. This variable offers
# the option to manually ask for rebaselining from another branch if necessary.
REBASELINE: "NO"
AUTOTEST: "NO"
# AUTOTEST_COMMIT: used only when AUTOTEST is set to YES.
# * If AUTOTEST_COMMIT is NOT set to NO, reporting jobs will commit their
# files to the MFEM/autotest repo.
# * If AUTOTEST_COMMIT is set to NO, reporting jobs will NOT commit their
# files to the MFEM/autotest repo. Instead they will just show the contents
# of the report files and remove them.
AUTOTEST_COMMIT: "YES"
REBASELINE: "OFF"
# Trigger subpipelines:
dane-build-and-test:
stage: sub-pipelines
##### SHARED_CI CONFIGURATION
# Required information about GitHub repository
GITHUB_PROJECT_NAME: "mfem"
GITHUB_PROJECT_ORG: "MFEM"
# Override the pattern describing branches that will skip the "draft PR filter
# test". Add protected branches here. See default value in
# preliminary-ignore-draft-pr.yml.
# ALWAYS_RUN_PATTERN: ""
###############################################################################
##### High level stages
# We organize the test-pipelines stage with sub-pipelines. Each sub-pipeline
# corresponds to a test batch on a given machine.
stages:
- prerequisites
- test-pipelines
###############################################################################
# Template for jobs triggering a build-and-test sub-pipeline:
.build-and-test:
stage: test-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
# Explicitly pass down values that are not always propagated to child
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
# interface (project variables).
# Note: in some cases, this does not work as expected, e.g. when the
# variable is not re-defined in the web interface; in such cases, the child
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
# AUTOTEST is undefined, even though there is a default value in
# .gitlab-ci.yml.
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/dane-build-and-test.yml
include:
- local: '.gitlab/custom-jobs-and-variables.yml'
- project: 'radiuss/radiuss-shared-ci'
ref: 'v2025.09.1'
file: 'pipelines/${CI_MACHINE}.yml'
- artifact: '${CI_MACHINE}-jobs.yml'
job: 'generate-job-lists'
strategy: depend
forward:
pipeline_variables: true
dane-baseline:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
REBASELINE: "${REBASELINE}"
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/dane-baseline.yml
strategy: depend
lassen-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/lassen-build-and-test.yml
strategy: depend
corona-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/corona-build-and-test.yml
strategy: depend
###############################################################################
include:
# Sets ID tokens for every job using `default:`
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# [Optional] checks preliminary to running the actual CI test
#- project: 'radiuss/radiuss-shared-ci'
# ref: 'v2025.09.1'
# file: 'preliminary-ignore-draft-pr.yml'
# pipelines subscribed by the project
- local: '.gitlab/subscribed-pipelines.yml'
+36 -12
View File
@@ -8,6 +8,8 @@
https://mfem.org
FIXME: this file needs to be updated
This directory contains most of the GitLab CI configuration. MFEM runs both PR
and nightly testing on GitLab.
@@ -15,9 +17,9 @@ and nightly testing on GitLab.
## Top level
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo.
This file only defines one stage, in which we trigger several
sub-pipelines.
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo. This
file only defines three stages, a prerequisites one, and two main stages in
which we trigger several sub-pipelines.
We use sub-pipelines to isolate the test for one combination of `machine`
and `test type`.
@@ -25,8 +27,8 @@ and `test type`.
Machines typically include:
* Dane: Intel Sapphire Rapids
* Lassen: Power9 + Nvidia GPU
* Corona: AMD GPU
* Matrix: Intel Sapphire Rapids + Nvidia H100 GPU
* Tioga: AMD MI250X GPU
Test types include:
@@ -39,9 +41,31 @@ altering the scheduling, execution and displaying of the others.
## Sub-pipelines
Each file is this directory is the root configuration file for one
sub-pipeline. The naming reflects the corresponding couple (`machine`,
`test_type`).
### build-and-test
The build-and-test sub-pipelines leverage RADIUSS Shared CI to share most of
the CI implementation. RADIUSS Shared CI provides a shared CI infrastructure
vetted on most LC systems of interest and efficiently leveraging each machine
scheduler to increase CI throughput. The maintenance of RADIUSS Shared CI is
shared among several RADIUSS projects.
Jobs for the build-and-test sub-pipelines are defined in the jobs directory.
Because build-and-test jobs leverage Uberenv and Spack to build the
dependencies automatically, the jobs essentially consists in a `spack spec`
defined in the jobs files, and some scheduling parameters defined in the
`.gitlab/custom-jobs-and-variables.yml` file.
Build-and-test jobs all run the `tests/gitlab/build_and_test` script.
The build-and-test pipelines are controlled by the
`.gitlab/subscribed-pipelines.yml` which defines which machines to run on and
implements additional features like machine availability check, and job list
generation.
### baseline
Baseline sub-pipelines are described by files with names reflecting the
machine it runs on, e.g. `dane-baseline`.
Those files define the *stages* and the *jobs* for the sub-pipeline. They
also contain any configuration that cannot be shared. For the most part
@@ -63,11 +87,11 @@ usage function. This should be improved.
# More testing
## Adding a new target to a build_and_test pipeline
## Adding a new target to a build-and-test pipeline
`build_and_test` pipelines rely on Spack to install dependencies. Spack is
`build-and-test` pipelines rely on Spack to install dependencies. Spack is
driven by Uberenv which helps freezing Spack configuration: the goal being to
point to specific commit in Spack and isolate its configuration so that it is
point to a specific commit in Spack and isolate its configuration so that it is
not influenced by the user environment. More documentation about this can be
found in `tests/gitlab`.
@@ -82,7 +106,7 @@ spack spec to use. Adding a job on Dane for example resumes to:
<job_name>:
variables:
SPEC: "<spack_spec>"
extends: .build_and_test_on_dane
extends: .job_on_dane
```
The remaining and non trivial work is to make sure this spec is working. To
-40
View File
@@ -1,40 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# We define the following GitLab pipeline variables:
variables:
# The path to the shared resource between all jobs. For example, external
# repositories like 'tests' and 'tpls' are cloned here. Also, 'tpls' is built
# once for all targets, so that build happen here. The BUILD_ROOT is unique to
# the pipeline, preventing any form of concurrency with other pipelines. This
# also means that the BUILD_ROOT directory will never be cleaned.
# TODO: add a clean-up mechanism
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${MACHINE_NAME}-pipeline-${CI_PIPELINE_ID}
# On LLNL's Dane, there is only one allocation shared among jobs in order to
# save time and resource. This allocation has to be uniquely named so that we
# are sure to retrieve it.
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
# Git repositories used in the pipeline
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
MFEM_DATA_REPO: https://github.com/mfem/data.git
# Directory used to place artifacts.
ARTIFACTS_DIR: artifacts
SLURM_OVERLAP: 1
-59
View File
@@ -1,59 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipeline configuration for the Corona machine at LLNL
variables:
MACHINE_NAME: corona
.on_corona:
tags:
- shell
- corona
rules:
# Don't run corona jobs if...
# Note: This makes corona an "opt-in" machine. To activate builds on corona
# for a given GitLab clone of MFEM, go to Setting/CI-CD/variables, and set
# "ON_CORONA" to "ON". An LC account on for corona is required to trigger a
# pipeline there.
- if: '$CI_COMMIT_BRANCH =~ /_cnone/ || $ON_CORONA != "ON"'
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always release resource
- if: '$CI_JOB_NAME =~ /release_resource/'
when: always
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
# Spack helped builds
# Generic corona build job, extending build script
.build_and_test_on_corona:
extends: [.on_corona]
stage: build_and_test
script:
# THREADS is used by 'tests/gitlab/build_and_test', run below
- export THREADS=12
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) -t 15 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
-56
View File
@@ -1,56 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Dane machine at LLNL
variables:
MACHINE_NAME: dane
.on_dane:
tags:
- shell
- dane
rules:
# Don't run dane jobs if...
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_DANE == "OFF"'
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always release resource
- if: '$CI_JOB_NAME =~ /release_resource/'
when: always
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
# Spack helped builds
# Generic dane build job, extending build script
.build_and_test_on_dane:
extends: [.on_dane]
stage: build_and_test
script:
# THREADS is used by 'tests/gitlab/build_and_test', run below
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
- export THREADS=28
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) --reservation=ci -t 60 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
-48
View File
@@ -1,48 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Lassen machine at LLNL
variables:
MACHINE_NAME: lassen
.on_lassen:
tags:
- shell
- lassen
rules:
- if: '$CI_COMMIT_BRANCH =~ /_lnone/ || $ON_LASSEN == "OFF"' #run except if ...
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
- when: on_success
# Lassen uses a different job scheduler (spectrum lsf) that does not allow
# pre-allocation the same way slurm does. We use the pci queue on lassen
# to speed-up the allocation.
.build_and_test_on_lassen:
extends: [.on_lassen]
stage: build_and_test
script:
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
# Next script uses 'THREADS': leaving it empty --> it uses 'make all -j'
- lalloc 1 -W 45 -q pci --atsdisable tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
needs: [setup]
-77
View File
@@ -1,77 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
.report_job_success:
script:
- echo ${MACHINE_NAME}
- echo ${AUTOTEST}
- echo ${AUTOTEST_COMMIT}
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# Report SUCCESS while holding the file lock on 'autotest.lock'.
# The next script uses the following environment variables:
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_success
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
.report_job_failure:
script:
- echo ${MACHINE_NAME}
- echo ${AUTOTEST}
- echo ${AUTOTEST_COMMIT}
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# Report FAILURE while holding the file lock on 'autotest.lock'.
# The next script uses the following environment variables:
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_failure
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
-90
View File
@@ -1,90 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
tags:
- shell
- dane
stage: setup
variables:
GIT_STRATEGY: none
script:
#
# Setup MFEM_DATA_DIR=${SHARED_REPOS_DIR}/mfem-data, see '.gitlab-ci.yml'
# and '.gitlab/configs/<machine>-config.yml'
#
- echo "MACHINE_NAME = ${MACHINE_NAME}"
- echo "AUTOTEST = ${AUTOTEST}"
- echo "AUTOTEST_COMMIT = ${AUTOTEST_COMMIT}"
- echo "SHARED_REPOS_DIR ${SHARED_REPOS_DIR}"
- mkdir -p ${SHARED_REPOS_DIR} && cd ${SHARED_REPOS_DIR}
- command -v flock || echo "Required command 'flock' not found"
- |
(
date
echo "Waiting to acquire lock on '$PWD/mfem-data.lock' ..."
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
# try every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/mfem-data.lock'"
date
# clone/update the mfem/data repo while holding the file lock on
# 'mfem-data.lock'
err=0
if [[ ! -d "mfem-data" ]]; then
git clone ${MFEM_DATA_REPO} "mfem-data"
else
cd "mfem-data" && git pull && cd ..
fi || err=1
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> mfem-data.lock
#
# Setup ${AUTOTEST_ROOT}/autotest:
#
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- mkdir -p ${AUTOTEST_ROOT} && cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# clone/update the autotest repo while holding the file lock on
# 'autotest.lock'
err=0
if [[ ! -d "autotest" ]]; then
git clone ${AUTOTEST_REPO}
else
cd autotest && git pull && cd ..
fi || err=1
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
-67
View File
@@ -1,67 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- allocate_resource
- build_and_test
- release_resource_and_report
# Slurm shared allocation
allocate_resource:
variables:
GIT_STRATEGY: none
extends: .on_corona
stage: allocate_resource
script:
- echo ${ALLOC_NAME}
- salloc --exclusive --nodes=1 --partition=mi60 --time=45 --no-shell --job-name=${ALLOC_NAME}
timeout: 6h
needs: [setup]
# Build and test jobs, simply provide a spec
rocm_gcc_8.3.1:
variables:
SPEC: "@develop%gcc@8.3.1+rocm amdgpu_target=gfx906"
extends: .build_and_test_on_corona
needs: [allocate_resource]
# Release slurm allocation
release_resource:
variables:
GIT_STRATEGY: none
extends: .on_corona
stage: release_resource_and_report
script:
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
needs: [rocm_gcc_8.3.1]
# Jobs report
report_job_success:
stage: release_resource_and_report
extends:
- .on_corona
- .report_job_success
report_job_failure:
stage: release_resource_and_report
extends:
- .on_corona
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/corona-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
+132
View File
@@ -0,0 +1,132 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# We define the following GitLab pipeline variables:
variables:
# Set the build-and-test command.
# Nested variables are allowed and useful to customize the job command. We
# protect variables with quotes so that their value may remain a string even if
# they contain whitespaces.
JOB_CMD:
value: tests/gitlab/build_and_test --spec \"${SPEC}\" --data-dir ${MFEM_DATA_DIR} --data
# The path to the shared resource between all jobs in the 'dane-baseline'
# pipeline. For example, external repositories like 'tests' and 'tpls' are
# cloned here. Also, 'tpls' is built once for all targets, so that build happens
# here. The BUILD_ROOT is unique to the pipeline, preventing any form of
# concurrency with other pipelines. This directory is removed by the 'cleanup'
# stage in the 'dane-baseline' pipeline.
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${CI_MACHINE}-pipeline-${CI_PIPELINE_ID}
# On LLNL's dane and tioga, the 'build-and-test' pipelines creates only one
# allocation shared among jobs in the pipeline in order to save time and
# resources. This allocation has to be uniquely named so that we are sure to
# retrieve it and avoid collisions.
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
# Git repositories used in the pipelines:
# - TPLS_REPO and TESTS_REPO are used only by the 'dane-baseline' pipeline
# - AUTOTEST_REPO is used by all pipelines
# - MFEM_DATA_REPO is used only by the 'build-and-test' pipelines
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
MFEM_DATA_REPO: https://github.com/mfem/data.git
# Directory used to place artifacts:
# - ARTIFACTS_DIR is only used by the 'dane-baseline' pipeline
ARTIFACTS_DIR: artifacts
SLURM_OVERLAP: 1
# Dane
# Arguments for top level allocation
DANE_SHARED_ALLOC: "--exclusive --reservation=ci --time=60 --nodes=1"
# Arguments for job level allocation
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
DANE_JOB_ALLOC: "--reservation=ci --overlap --nodes=1"
# Tioga
# Arguments for top level allocation
TIOGA_SHARED_ALLOC: "--queue=pci --exclusive --time-limit=45m --nodes=1"
# Arguments for job level allocation
TIOGA_JOB_ALLOC: "--nodes=1 --begin-time=+5s"
# Matrix
# Arguments for top level allocation
MATRIX_SHARED_ALLOC: "-p pdebug --exclusive --time=45 --nodes=1 -G 4"
# Arguments for job level allocation
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
MATRIX_JOB_ALLOC: "--overlap --nodes=1"
# Configuration shared by build and test jobs specific to this project.
# Not all configuration can be shared. Here projects can fine tune the
# CI behavior.
# See Umpire for an example (export junit test reports).
.custom_job:
artifacts:
reports:
# Note: this part is not used by the 'dane-baseline' pipeline.
# FIXME: BUILD_ROOT, TPLS_REPO, TESTS_REPO are not needed here.
# Also, the definition of SHARED_REPOS_DIR is wrong.
.reproducer_vars:
script:
- |
echo -e "
# Variables \n
export SPEC=\"${SPEC//\"/\\\"}\" \n
# Directories \n
export BUILD_ROOT=\"\${working_dir}\" \n
export SHARED_REPOS_DIR=\"\${BUILD_ROOT}/..\" \n
export MFEM_DATA_DIR=\"\${SHARED_REPOS_DIR}/mfem-data\" \n
# Repositories \n
export TPLS_REPO=\"${TPLS_REPO//\"/\\\"}\" \n
export TESTS_REPO=\"${TESTS_REPO//\"/\\\"}\" \n
export AUTOTEST_REPO=\"${AUTOTEST_REPO//\"/\\\"}\" \n
export MFEM_DATA_REPO=\"${MFEM_DATA_REPO//\"/\\\"}\" \n
# Setup directories \n
./tests/gitlab/build_and_test_setup \n
# Using the CI build cache is optional and requires a token. Set it like so: \n
# export REGISTRY_TOKEN=\"<your token here>\" \n"
#
# Jobs report
.report_job_success:
script:
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test SUCCESS
rules:
- when: on_success
.report_job_failure:
script:
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test FAILURE
rules:
- when: on_failure
# Keep the following for debugging purposes: renaming this job from
# '.show_variables' to 'show_variables' will insert this debug job at the
# beginning of all child pipelines.
.show_variables:
tags: [shell, oslic]
variables:
GIT_STRATEGY: none
stage: .pre
script:
- |
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
echo "AUTOTEST=${AUTOTEST}"
echo "AUTOTEST_COMMIT=${AUTOTEST_COMMIT}"
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
# Fail the job on purpose to prevent the rest of the pipeline from running
false
+33 -5
View File
@@ -11,6 +11,7 @@
variables:
BASELINE_TEST: baseline
MACHINE_NAME: dane
stages:
- setup
@@ -19,6 +20,25 @@ stages:
- cleanup
- baseline_publish
.on_dane:
tags:
- shell
- dane
rules:
# Don't run dane jobs if...
- if: '$ON_DANE == "OFF"'
when: never
# Don't run autotest update if...
# Note: in some cases, the content of AUTOTEST can be '${AUTOTEST}', so we
# need to treat that value as the default value of 'OFF'.
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "ON" && $AUTOTEST != "YES"'
when: never
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
baselinecheck_mfem_intel_dane:
extends: [.on_dane]
stage: baseline_check
@@ -29,6 +49,9 @@ baselinecheck_mfem_intel_dane:
# .gitlab/configs/setup-baseline.yml.
TPLS_DIR: ${BUILD_ROOT}/tpls
script:
- echo "AUTOTEST=$AUTOTEST"
- echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
- echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
- echo ${BUILD_ROOT}
- echo ${TPLS_DIR}
# Used by the tests in MFEM/tests, dane has 224 threads/node:
@@ -89,7 +112,13 @@ report_baseline:
cp ${rundir}/pipeline.txt ${rundir}/autotest-email.html
fi
msg="GitLab CI log for ${BASELINE_TEST} on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
# Note: in some cases, the content of AUTOTEST_COMMIT can be
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
# value of 'ON'.
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
AUTOTEST_COMMIT="ON"
fi
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
@@ -117,8 +146,8 @@ baselinepublish_mfem_dane:
extends: [.on_dane]
stage: baseline_publish
rules:
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "YES"'
- if: '$REBASELINE == "YES"'
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "ON"'
- if: '$REBASELINE == "ON"'
when: manual
script:
- echo ${BUILD_ROOT}
@@ -128,6 +157,5 @@ baselinepublish_mfem_dane:
- .gitlab/scripts/rebaseline
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/dane-config.yml
- local: .gitlab/custom-jobs-and-variables.yml
- local: .gitlab/configs/setup-baseline.yml
-94
View File
@@ -1,94 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- allocate_resource
- build_and_test
- release_resource_and_report
# Allocate
allocate_resource:
variables:
GIT_STRATEGY: none
extends: .on_dane
stage: allocate_resource
script:
- echo ${ALLOC_NAME}
- salloc --exclusive --nodes=1 --reservation=ci --time=60 --no-shell --job-name=${ALLOC_NAME}
timeout: 6h
# GitLab jobs for the Dane machine at LLNL
debug_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug~mpi"
extends: .build_and_test_on_dane
debug_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug+mpi"
extends: .build_and_test_on_dane
opt_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 ~mpi"
extends: .build_and_test_on_dane
opt_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1"
extends: .build_and_test_on_dane
opt_par_gcc_10_sundials:
variables:
SPEC: "%gcc@10.3.1 +sundials"
extends: .build_and_test_on_dane
opt_par_gcc_10_petsc:
variables:
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
extends: .build_and_test_on_dane
opt_par_gcc_10_pumi:
variables:
SPEC: "%gcc@10.3.1 +pumi"
extends: .build_and_test_on_dane
# Release
release_resource:
variables:
GIT_STRATEGY: none
extends: .on_dane
stage: release_resource_and_report
script:
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
# Jobs report
report_job_success:
stage: release_resource_and_report
extends:
- .on_dane
- .report_job_success
report_job_failure:
stage: release_resource_and_report
extends:
- .on_dane
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/dane-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
+19
View File
@@ -0,0 +1,19 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_dane, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_dane, .report_job_failure]
stage: jobs-stage-3
+87
View File
@@ -0,0 +1,87 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define MFEM specific variables.
.dane_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
# TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY
# cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_dane
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs, we can duplicate them here to override description and
# add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
.mfem_job_on_dane:
extends: .job_on_dane
stage: jobs-stage-2
variables:
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
THREADS: 28
debug_ser_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +debug~mpi"
debug_par_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +debug+mpi"
opt_ser_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 ~mpi"
opt_par_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1"
opt_par_gcc_10_sundials:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +sundials"
opt_par_gcc_10_petsc:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
opt_par_gcc_10_pumi:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +pumi"
+19
View File
@@ -0,0 +1,19 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_matrix, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_matrix, .report_job_failure]
stage: jobs-stage-3
+65
View File
@@ -0,0 +1,65 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define UMPIRE specific variables.
.matrix_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_matrix
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
.mfem_job_on_matrix:
extends: .job_on_matrix
stage: jobs-stage-2
variables:
# We run 2 jobs on 1 node that has 112 threads
THREADS: 48
# These modules need to be consistent with the uberenv configurations:
MODULE_LIST: "gcc/10.3.1-magic cuda/12.9.1"
allocate_resources:
timeout: 4h
opt_mpi_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90"
opt_mpi_cuda_hypre_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
+20
View File
@@ -0,0 +1,20 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_tioga, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_tioga, .report_job_failure]
stage: jobs-stage-3
+70
View File
@@ -0,0 +1,70 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define UMPIRE specific variables.
.tioga_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_tioga
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
# Build and test jobs, simply provide a spec
#.tioga_job_command:
# script:
# - echo PROXY="${PROXY}"
# - echo TIOGA_JOB_ALLOC="${TIOGA_JOB_ALLOC}"
# - "printf '#!/bin/bash\n%s\n' \"${JOB_CMD}\" > flux_script.sh"
# - cat flux_script.sh
# - ${PROXY} flux watch $( ${PROXY} flux batch -o output.stdout.type=kvs ${TIOGA_JOB_ALLOC} flux_script.sh )
# - rm -f flux_script.sh
.mfem_job_on_tioga:
extends: .job_on_tioga
stage: jobs-stage-2
variables:
# We run 1 job on 1 node that has 64 threads
THREADS: 64
opt_mpi_rocm_hypre_rocm:
extends: .mfem_job_on_tioga
variables:
SPEC: "%rocmcc@=6.3.1 +rocm amdgpu_target=gfx90a ^hypre+rocm"
# cce_16_0_1:
# extends: .mfem_job_on_tioga
# variables:
# SPEC: "%cce@=16.0.1"
-44
View File
@@ -1,44 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- build_and_test
- report
opt_mpi_cuda_gcc:
variables:
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70"
extends: .build_and_test_on_lassen
opt_mpi_cuda_hypre_cuda_gcc:
variables:
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70 ^hypre+cuda~shared cuda_arch=70"
extends: .build_and_test_on_lassen
# Jobs report
report_job_success:
stage: report
extends:
- .on_lassen
- .report_job_success
report_job_failure:
stage: report
extends:
- .on_lassen
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/lassen-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
+1 -3
View File
@@ -32,11 +32,9 @@ mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
# run
if [[ "${MACHINE_NAME}" == "dane" ]]; then
salloc --nodes=1 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
salloc --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "lassen" ]]; then
lalloc 1 -q pci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
else
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
exit 1
+118
View File
@@ -0,0 +1,118 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
function info_msg ()
{
echo "[Information:] ${1}"
}
function error_msg ()
{
echo "[Error:] ${1}"
}
# Perform a report while holding a lock file to prevent concurrency on
# the destination.
# Usage:
# locked_clone <report_function> <lock_name>
function locked_report ()
{
if ! command -v flock
then
error_msg "Required command 'flock' not found"
exit 1
fi
info_msg "Will report ${1} while holding a lock in ${2}"
( date; info_msg "Waiting to acquire lock on '${PWD}/${2}.lock' ..."
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
# try every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do sleep 5; done
date; info_msg "Acquired lock on '${PWD}/${2}.lock'"
report ${1}
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> ${2}.lock
}
function report ()
{
if [[ "${1}" == "SUCCESS" ]]
then
info_msg "All the ${MACHINE_NAME} jobs passed"
status_msg="The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL."
elif [[ "${1}" == "FAILURE" ]]
then
info_msg "At least one failure on ${MACHINE_NAME}"
status_msg="Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED."
else
error_msg "Unknown status: ${1} ... aborting"
exit 1
fi
cd ${AUTOTEST_ROOT}/autotest || \
{ error_msg "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "${status_msg}" \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
if [[ "${1}" == "FAILURE" ]]
then
# Create 'autotest-email.html' to indicate failure:
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
fi
# Note: in some cases, the content of AUTOTEST_COMMIT can be
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
# value of 'ON'.
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
AUTOTEST_COMMIT="ON"
fi
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
}
export MACHINE_NAME=${CI_MACHINE}
info_msg "MACHINE_NAME is ${MACHINE_NAME}"
info_msg "AUTOTEST_ROOT is ${AUTOTEST_ROOT}"
info_msg "AUTOTEST=$AUTOTEST"
info_msg "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
cd ${AUTOTEST_ROOT} && locked_report ${1} autotest
@@ -1,45 +0,0 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
echo "Runs if there was at least one failure on ${MACHINE_NAME}"
cd ${AUTOTEST_ROOT}/autotest || \
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED." \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
# Create 'autotest-email.html' to indicate failure:
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
@@ -1,42 +0,0 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
echo "Can only run if all the ${MACHINE_NAME} jobs passed"
cd ${AUTOTEST_ROOT}/autotest || \
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL." \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.out
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
+130
View File
@@ -0,0 +1,130 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# The template job to test whether a machine is up.
# Expects CI_MACHINE defined to machine name.
.machine-check:
stage: prerequisites
tags: [shell, oslic]
variables:
GIT_STRATEGY: none
script:
- |
if [[ $(jq '.[env.CI_MACHINE].total_nodes_up' /usr/global/tools/lorenz/data/loginnodeStatus) == 0 ]]
then
echo -e "\e[31mNo node available on ${CI_MACHINE}\e[0m"
false && \
curl --url "https://api.github.com/repos/${GITHUB_PROJECT_ORG}/${GITHUB_PROJECT_NAME}/statuses/${CI_COMMIT_SHA}" \
--header 'Content-Type: application/json' \
--header "authorization: Bearer ${GITHUB_TOKEN}" \
--data "{ \"state\": \"failure\", \"target_url\": \"${CI_PIPELINE_URL}\", \"description\": \"GitLab ${CI_MACHINE} down\", \"context\": \"ci/gitlab/${CI_MACHINE}\" }"
exit 1
fi
###
# Trigger a build-and-test pipeline for a machine.
# Comment the jobs for machines you dont need.
###
# One job to generate the job list for all the subpipelines
generate-job-lists:
stage: prerequisites
tags: [shell, oslic]
variables:
LOCAL_JOBS_PATH: ".gitlab/jobs"
script:
- |
echo "AUTOTEST=$AUTOTEST"
echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
- |
cat ${LOCAL_JOBS_PATH}/dane.yml > dane-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/dane-reports.yml >> dane-jobs.yml
fi
- |
cat ${LOCAL_JOBS_PATH}/matrix.yml > matrix-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/matrix-reports.yml >> matrix-jobs.yml
fi
- |
cat ${LOCAL_JOBS_PATH}/tioga.yml > tioga-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/tioga-reports.yml >> tioga-jobs.yml
fi
artifacts:
paths:
- dane-jobs.yml
- matrix-jobs.yml
- tioga-jobs.yml
# DANE
dane-up-check:
variables:
CI_MACHINE: "dane"
extends: [.machine-check]
dane-build-and-test:
variables:
CI_MACHINE: "dane"
needs: [dane-up-check, generate-job-lists]
extends: [.build-and-test]
# DANE, MFEM Specific
dane-baseline:
stage: test-pipelines
variables:
# Explicitly pass down values that are not always propagated to child
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
# interface (project variables).
# Note: in some cases, this does not work as expected, e.g. when the
# variable is not re-defined in the web interface; in such cases, the child
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
# AUTOTEST is undefined, even though there is a default value in
# .gitlab-ci.yml.
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/dane-baseline.yml
strategy: depend
forward:
pipeline_variables: true
needs: [dane-up-check]
# TIOGA
tioga-up-check:
variables:
CI_MACHINE: "tioga"
extends: [.machine-check]
tioga-build-and-test:
variables:
CI_MACHINE: "tioga"
needs: [tioga-up-check, generate-job-lists]
extends: [.build-and-test]
# Matrix
matrix-up-check:
variables:
CI_MACHINE: "matrix"
extends: [.machine-check]
matrix-build-and-test:
variables:
CI_MACHINE: "matrix"
needs: [matrix-up-check, generate-job-lists]
extends: [.build-and-test]
+30
View File
@@ -43,6 +43,13 @@ Discretization improvements
Meshing improvements
--------------------
- The TMOP kernel hierarchy has been restructured to reduce compilation time.
Most large kernels have been split into smaller, specific ones, with kernels
for each metric. The directory structure has been updated with assemble,
metrics, mult and tools subdirectories. The new kernel dispatch and
specialization system has also been integrated.
Unit tests have been revised to ensure --all tests pass.
- Introduced NC-patch NURBS meshes, which are conforming element-wise but allow
for nonconforming patch topology. This new mesh format supports element
spacing formulas for refinement, as well as local refinement factors for a
@@ -76,6 +83,15 @@ GPU computing
- Added GPU support in GradientGridFunctionCoefficient and
InnerProductCoefficient by implementing their Project methods.
Linear and nonlinear solvers
----------------------------
- Added `FilteredSolver`: a base class for solvers with filtering. It handles cases
where a solver performs well except in small subspaces, by adding a filtering step
formulated as a subspace correction.
- Added `AMGFSolver`: a derived class of `FilteredSolver`, specialized for
AMG with Filtering (AMGF), providing robust preconditioning for linear systems
arising in constrained optimization problems such as frictionless contact.
New and updated examples and miniapps
-------------------------------------
- Added miniapps to demonstrate an implementation of the absolute-value
@@ -95,6 +111,16 @@ New and updated examples and miniapps
of a charged particle, subject to Lorentz forces, in electrostatic and/or
magnetostatic fields as computed by the volta or tesla miniapps.
- Added a new miniapp for optimization-based contact mechanics. This miniapp solves
large-scale frictionless contact problems using a self-contained Interior Point (IP) solver,
with Tribol integration for mortar-based contact constraints. Linear systems arising in the IP
iterations are solved with CG preconditioned by the recently introduced AMGF solver.
Benchmark examples include the two-block, ironing, and beam-sphere problems.
The miniapp is available in `miniapps/contact`
- Updated the mtop miniapp with a GPU enabled forward and adjoint solver for
isotropic linear elasticity.
API changes:
-----------
- mfem::internal::tensor and mfem::internal::dual have been moved to
@@ -126,6 +152,10 @@ Miscellaneous
not need to manually free-up the memory if the destructor is called before
MPI_Finalize().
- Introduced ParticleVector, a Vector-derived container that stores vector
data for an arbitrary number of particles contiguously based on user specified
vector-dimension and ordering (byNODES/byVDIM).
Version 4.8, released on Apr 9, 2025
====================================
+57 -24
View File
@@ -133,33 +133,49 @@ if (MFEM_USE_CUDA)
if (NOT CMAKE_CUDA_HOST_COMPILER)
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
endif()
if (CMAKE_VERSION VERSION_LESS 3.18.0)
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
string(REGEX REPLACE "^sm_" "" ARCH_NUMBER "${CUDA_ARCH}")
if ("${CUDA_ARCH}" STREQUAL "sm_${ARCH_NUMBER}")
set(CMAKE_CUDA_ARCHITECTURES "${ARCH_NUMBER}")
else()
message(FATAL_ERROR "Unknown CUDA_ARCH: ${CUDA_ARCH}")
endif()
if (NOT CMAKE_CUDA_ARCHITECTURES)
# make CUDA_ARCH resemble the same form as CMAKE_CUDA_ARCHITECTURES
string(REPLACE "sm_" "" CUDA_ARCH_TMP "${CUDA_ARCH}")
string(REPLACE "," ";" CUDA_ARCH "${CUDA_ARCH_TMP}")
set(CMAKE_CUDA_ARCHITECTURES "${CUDA_ARCH}")
else()
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
endif()
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
enable_language(CUDA)
if (CMAKE_VERSION VERSION_LESS 3.18.0)
# backup try to detect if this is clang or nvcc
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
# backup try to detect if this is clang or nvcc
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
if ("all" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "native" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "all-major" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}")
set(CUDA_FLAGS "-arch=${CMAKE_CUDA_ARCHITECTURES} ${CUDA_FLAGS}")
else()
# build -gencode sequence for multiple architectures
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(CUDA_FLAGS
"-gencode arch=compute_${ENTRY},code=sm_${ENTRY} ${CUDA_FLAGS}")
endforeach()
endif()
else()
# build cuda-gpu-arch sequence for multiple architectures
# does not support all/all-major/native
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(CUDA_FLAGS "-cuda-gpu-arch=sm_${ENTRY} ${CUDA_FLAGS}")
endforeach()
endif()
else()
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
# TODO: all, native, all-major require CMake 3.24+
# backport support for CMake 3.18 to 3.24
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS
"${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
endif()
set(CMAKE_CUDA_STANDARD ${CMAKE_CXX_STANDARD} CACHE STRING
"CUDA standard to use.")
@@ -242,10 +258,16 @@ endif()
# AMD HIP
if (MFEM_USE_HIP)
if (HIP_ARCH)
message(STATUS "Using HIP architecture: ${HIP_ARCH}")
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for")
if (NOT CMAKE_HIP_ARCHITECTURES)
if (HIP_ARCH)
set(CMAKE_HIP_ARCHITECTURES CACHE STRING "HIP targets to compile for" "${HIP_ARCH}")
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for" FORCE)
endif()
else()
set(HIP_ARCH CACHE STRING "HIP targets to compile for" "${CMAKE_HIP_ARCHITECTURES}")
set(GPU_TARGETS "${CMAKE_HIP_ARCHITECTURES}" CACHE STRING "HIP targets to compile for" FORCE)
endif()
message(STATUS "Using HIP architecture: ${CMAKE_HIP_ARCHITECTURES}")
if (ROCM_PATH)
list(INSERT CMAKE_PREFIX_PATH 0 ${ROCM_PATH})
endif()
@@ -278,8 +300,19 @@ if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
endif()
endif()
# Umpire (must be included before hypre, so hypre can use it if needed)
# Warn user if deprecated FETCH_TPLS is provided
if (DEFINED FETCH_TPLS)
message(STATUS "Setting MFEM_FETCH_TPLS to user-provided value of FETCH_TPLS (i.e., MFEM_FETCH_TPLS=${FETCH_TPLS})")
set (MFEM_FETCH_TPLS FETCH_TPLS)
message(DEPRECATION "The use of FETCH_TPLS is deprecated and will be removed in future verison. Please use MFEM_FETCH_TPLS instead.")
endif()
# Umpire (must be included before hypre, so hypre can use it if needed)
if (MFEM_USE_UMPIRE)
# umpire uses FindCUDA, which needs CMP0146=OLD in CMake >= 3.27
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.27.0)
cmake_policy(SET CMP0146 OLD)
endif()
find_package(UMPIRE REQUIRED)
endif()
+6 -4
View File
@@ -123,7 +123,7 @@ Parallel build:
Parallel build with fetching of hypre and METIS:
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DFETCH_TPLS=YES
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
make -j 4
CUDA build:
@@ -1081,9 +1081,10 @@ The following options are CMake specific:
MFEM_ENABLE_TESTING - Enable the ctest framework for testing.
MFEM_ENABLE_EXAMPLES - Build all of the examples by default.
MFEM_ENABLE_MINIAPPS - Build all of the miniapps by default.
FETCH_TPLS - Enable fetching of all supported third-party libraries.
HYPRE_FETCH - Enable fetching of hypre.
METIS_FETCH - Enable fetching of metis.
MFEM_FETCH_TPLS - Enable fetching of all supported third-party libraries.
MFEM_FETCH_GSLIB - Enable fetching of gslib.
MFEM_FETCH_HYPRE - Enable fetching of hypre.
MFEM_FETCH_METIS - Enable fetching of metis.
External libraries (CMake):
---------------------------
@@ -1149,6 +1150,7 @@ The MFEM CMake build system also provides fetching (automated building) for the
packages/libraries listed below. Note that when fetching is enabled, any related
auto-detection functionality is disabled.
- GSLIB
- HYPRE
- METIS
+38 -1
View File
@@ -9,10 +9,47 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables:
# Defines the following variables if fetching of TPLs is disabled (default):
# - GSLIB_FOUND
# - GSLIB_LIBRARIES
# - GSLIB_INCLUDE_DIRS
# otherwise, the following are defined:
# - GSLIB (imported library target)
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
enable_language(C)
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(GSLIB_FETCH_VERSION 1.0.9)
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
if (CMAKE_C_FLAGS)
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
endif()
if (BUILD_SHARED_LIBS)
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
endif()
add_library(GSLIB STATIC IMPORTED)
# define external project and create future include directory so it is present
# to pass CMake checks at end of MFEM configuration step
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
include(ExternalProject)
ExternalProject_Add(gslib
GIT_REPOSITORY https://github.com/Nek5000/gslib
GIT_TAG v${GSLIB_FETCH_VERSION}
GIT_SHALLOW TRUE
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND ""
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
INSTALL_COMMAND "")
file(MAKE_DIRECTORY ${PREFIX}/include)
# set imported library target properties
add_dependencies(GSLIB gslib)
set_target_properties(GSLIB PROPERTIES
IMPORTED_LOCATION ${PREFIX}/lib/libgs.a
INTERFACE_INCLUDE_DIRECTORIES ${PREFIX}/include)
return()
endif()
include(MfemCmakeUtilities)
mfem_find_package(GSLIB GSLIB GSLIB_DIR "include" gslib.h "lib" gs
+8 -8
View File
@@ -37,21 +37,21 @@ if (HYPRE_FOUND OR TARGET HYPRE)
endif()
endif()
if (HYPRE_FETCH OR FETCH_TPLS)
# Collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
set(HYPRE_FETCH_VERSION 2.33.0)
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
add_library(HYPRE STATIC IMPORTED)
# set options and associated dependencies
set(HYPRE_CMAKE_OPTIONS "")
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
get_cmake_property(all_vars VARIABLES)
foreach(var ${all_vars})
if(var MATCHES "^HYPRE_ENABLE")
list(APPEND HYPRE_CMAKE_OPTIONS "-D${var}:BOOL=${${var}}")
endif()
endforeach()
set(HYPRE_FETCH_VERSION 2.33.0)
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
add_library(HYPRE STATIC IMPORTED)
# set options and associated dependencies
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
# process all MFEM_USE variables that impact hypre
if (MFEM_USE_CUDA)
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_CUDA:BOOL=ON -DCMAKE_CUDA_ARCHITECTURES:STRING=${CMAKE_CUDA_ARCHITECTURES})
find_package(CUDAToolkit REQUIRED)
+1 -1
View File
@@ -18,7 +18,7 @@
# - METIS (imported library target)
# - METIS_VERSION_5 (cache variable)
if (METIS_FETCH OR FETCH_TPLS)
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
set(METIS_FETCH_VERSION 4.0.3)
add_library(METIS STATIC IMPORTED)
# define external project
+4 -3
View File
@@ -91,9 +91,10 @@ option(MFEM_ENABLE_BENCHMARKS "Build all of the benchmarks" OFF)
# Allow a user to specify fetching of certain third-party libraries instead of
# searching for existing installations.
option(FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
option(HYPRE_FETCH "Enable fetching of hypre" OFF)
option(METIS_FETCH "Enable fetching of METIS" OFF)
option(MFEM_FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
option(MFEM_FETCH_GSLIB "Enable fetching of GSLIB" OFF)
option(MFEM_FETCH_HYPRE "Enable fetching of hypre" OFF)
option(MFEM_FETCH_METIS "Enable fetching of METIS" OFF)
# Setting CXX/MPICXX on the command line or in user.cmake will overwrite the
# autodetected C++ compiler.
+2 -1
View File
@@ -996,7 +996,8 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
@MFEM_SOURCE_DIR@/miniapps/spde \
@MFEM_SOURCE_DIR@/miniapps/tools \
@MFEM_SOURCE_DIR@/miniapps/toys \
@MFEM_SOURCE_DIR@/miniapps/tribol
@MFEM_SOURCE_DIR@/miniapps/tribol \
@MFEM_SOURCE_DIR@/miniapps/contact
# This tag can be used to specify the character encoding of the source files
# that doxygen parses. Internally doxygen uses the UTF-8 encoding. Doxygen uses
+3
View File
@@ -471,10 +471,13 @@ int main(int argc, char *argv[])
ofstream sol_r_ofs("sol_r.gf");
ofstream sol_i_ofs("sol_i.gf");
ofstream sol_z_ofs("sol_z.gf");
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
sol_z_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
u.Save(sol_z_ofs);
}
// 14. Send the solution by socket to a GLVis server.
+5 -1
View File
@@ -507,10 +507,11 @@ int main(int argc, char *argv[])
// 15. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_r_name, sol_i_name;
ostringstream mesh_name, sol_r_name, sol_i_name, sol_z_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
sol_z_name << "sol_z." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
@@ -518,10 +519,13 @@ int main(int argc, char *argv[])
ofstream sol_r_ofs(sol_r_name.str().c_str());
ofstream sol_i_ofs(sol_i_name.str().c_str());
ofstream sol_z_ofs(sol_z_name.str().c_str());
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
sol_z_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
u.Save(sol_z_ofs);
}
// 16. Send the solution by socket to a GLVis server.
+2 -2
View File
@@ -195,8 +195,8 @@ clean-build:
clean-exec:
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh ex6p-checkpoint.*
@rm -rf Example5* Example9* Example15* Example16* Example23* ParaView
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* order.*
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* sol_z.*
@rm -f order.* ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f deformed.* velocity.* elastic_energy.* mode_* mode_deriv_* flux.*
@rm -f ex5-p-*.bp ex9-p-*.bp ex12-p-*.bp ex16-p-*.bp
@rm -f ex16.mesh ex16-mesh.* ex16-init.* ex16-final.*
+48 -27
View File
@@ -128,32 +128,46 @@ set(SRCS
normal_deriv_restriction.cpp
staticcond.cpp
tmop.cpp
tmop/tmop_pa.cpp
tmop/tmop_pa_da3.cpp
tmop/tmop_pa_h2d.cpp
tmop/tmop_pa_h2d_c0.cpp
tmop/tmop_pa_h2m.cpp
tmop/tmop_pa_h2m_c0.cpp
tmop/tmop_pa_h2s.cpp
tmop/tmop_pa_h2s_c0.cpp
tmop/tmop_pa_h3d.cpp
tmop/tmop_pa_h3d_c0.cpp
tmop/tmop_pa_h3m.cpp
tmop/tmop_pa_h3m_c0.cpp
tmop/tmop_pa_h3s.cpp
tmop/tmop_pa_h3s_c0.cpp
tmop/tmop_pa_jp2.cpp
tmop/tmop_pa_jp3.cpp
tmop/tmop_pa_p2.cpp
tmop/tmop_pa_p2_c0.cpp
tmop/tmop_pa_p3.cpp
tmop/tmop_pa_p3_c0.cpp
tmop/tmop_pa_tc2.cpp
tmop/tmop_pa_tc3.cpp
tmop/tmop_pa_w2.cpp
tmop/tmop_pa_w2_c0.cpp
tmop/tmop_pa_w3.cpp
tmop/tmop_pa_w3_c0.cpp
tmop/pa.cpp
tmop/assemble/diag2_limit.cpp
tmop/assemble/diag2.cpp
tmop/assemble/grad2_limit.cpp
tmop/assemble/grad2.cpp
tmop/assemble/diag3_limit.cpp
tmop/assemble/diag3.cpp
tmop/assemble/grad3_limit.cpp
tmop/assemble/grad3.cpp
tmop/metrics/001.cpp
tmop/metrics/002.cpp
tmop/metrics/007.cpp
tmop/metrics/056.cpp
tmop/metrics/077.cpp
tmop/metrics/080.cpp
tmop/metrics/094.cpp
tmop/metrics/302.cpp
tmop/metrics/303.cpp
tmop/metrics/315.cpp
tmop/metrics/318.cpp
tmop/metrics/321.cpp
tmop/metrics/332.cpp
tmop/metrics/338.cpp
tmop/mult/grad2_limit.cpp
tmop/mult/grad2.cpp
tmop/mult/mult2_limit.cpp
tmop/mult/mult2.cpp
tmop/mult/grad3_limit.cpp
tmop/mult/grad3.cpp
tmop/mult/mult3_limit.cpp
tmop/mult/mult3.cpp
tmop/tools/det2_jpr.cpp
tmop/tools/det3_jpr.cpp
tmop/tools/discrete.cpp
tmop/tools/energy2_limit.cpp
tmop/tools/energy2.cpp
tmop/tools/energy3_limit.cpp
tmop/tools/energy3.cpp
tmop/tools/target2.cpp
tmop/tools/target3.cpp
tmop_tools.cpp
tmop_amr.cpp
gslib.cpp
@@ -182,6 +196,8 @@ set(HDRS
integ/bilininteg_hdiv_kernels.hpp
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
integ/bilininteg_vecdiffusion_pa.hpp
integ/bilininteg_vecmass_pa.hpp
coefficient.hpp
complex_fem.hpp
convergence.hpp
@@ -279,7 +295,12 @@ set(HDRS
tfespace.hpp
tintrules.hpp
tmop.hpp
tmop/tmop_pa.hpp
tmop/pa.hpp
tmop/assemble/grad2.hpp
tmop/assemble/grad2.hpp
tmop/mult/mult2.hpp
tmop/mult/mult3.hpp
tmop/tools/energy2.hpp
tmop_tools.hpp
tmop_amr.hpp
gslib.hpp
+432 -4
View File
@@ -1820,7 +1820,7 @@ void VectorFEDivergenceIntegrator::AssembleElementMatrix2(
trial_fe.CalcDivShape(ip, divshape);
Trans.SetIntPoint(&ip);
test_fe.CalcPhysShape(Trans, shape);
real_t w = ip.weight;
real_t w = alpha*ip.weight;
if (Q)
{
Trans.SetIntPoint(&ip);
@@ -2869,6 +2869,162 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
}
}
void VectorFEDiffusionIntegrator::AssembleElementMatrix(
const FiniteElement &el,
ElementTransformation &Trans,
DenseMatrix &elmat)
{
const int dof = el.GetDof();
const int spaceDim = Trans.GetSpaceDim();
const int vdim = std::max(spaceDim, el.GetRangeDim());
real_t w;
#ifdef MFEM_THREAD_SAFE
DenseTensor trial_dvshape(dof, vdim, spaceDim);
Vector D(DQ ? DQ->GetVDim() : 0);
DenseMatrix K(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#else
trial_dvshape.SetSize(dof, vdim, spaceDim);
D.SetSize(DQ ? DQ->GetVDim() : 0);
K.SetSize(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#endif
DenseMatrix flat_dvs;
trial_dvshape.GetDenseMatrix(flat_dvs);
DenseMatrix tmp(flat_dvs.Height(), flat_dvs .Width());
elmat.SetSize(dof);
elmat = 0.0;
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
// int order = 2 * el.GetOrder();
int order = Trans.OrderW() + 2 * el.GetOrder() - 2;
ir = &IntRules.Get(el.GetGeomType(), order);
}
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
Trans.SetIntPoint (&ip);
el.CalcPhysDVShape(Trans, trial_dvshape);
w = ip.weight * Trans.Weight();
if (MQ)
{
// flat_dvs has interlaced vector component and derivative direction in 2nd index
// Matrix needs to be (dim*dim) X (dim*dim)
// indexing: 2nd index = dif_dir + comp_dir*dim
// u_x, v_x, u_y, v_y
// u_x, v_x, w_x, u_y, v_y, w_y, u_z, v_z, w_z
MQ->Eval(K, Trans, ip);
K *= w;
Mult(flat_dvs,K,tmp);
AddMultABt(tmp,flat_dvs,elmat);
}
else if (DQ)
{
// Vector needs to be (dim*dim), see comment for MQ case
DQ->Eval(D, Trans, ip);
D *= w;
AddMultADAt(flat_dvs, D, elmat);
}
else
{
if (Q)
{
w *= Q -> Eval (Trans, ip);
}
AddMult_a_AAt (w, flat_dvs, elmat);
}
}
}
void VectorFEDiffusionIntegrator::AssembleElementMatrix2(
const FiniteElement &trial_fe, const FiniteElement &test_fe,
ElementTransformation &Trans, DenseMatrix &elmat)
{
// assume both test_fe and trial_fe are vector FE
const int spaceDim = Trans.GetSpaceDim();
const int trial_vdim = std::max(spaceDim, trial_fe.GetRangeDim());
const int test_vdim = std::max(spaceDim, test_fe.GetRangeDim());
const int trial_dof = trial_fe.GetDof();
const int test_dof = test_fe.GetDof();
real_t w;
#ifdef MFEM_THREAD_SAFE
DenseTensor trial_dvshape(trial_dof, trial_vdim, spaceDim);
DenseTensor test_dvshape(test_dof, test_vdim, spaceDim);
Vector D(DQ ? DQ->GetVDim() : 0);
DenseMatrix K(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#else
trial_dvshape.SetSize(trial_dof, trial_vdim, spaceDim);
test_dvshape.SetSize(test_dof, test_vdim, spaceDim);
D.SetSize(DQ ? DQ->GetVDim() : 0);
K.SetSize(MQ ? MQ->GetVDim() : 0, MQ ? MQ->GetVDim() : 0);
#endif
DenseMatrix trial_fdvs,test_fdvs;
trial_dvshape.GetDenseMatrix(trial_fdvs);
test_dvshape.GetDenseMatrix(test_fdvs);
DenseMatrix tmp(test_fdvs.Height(), K.Width());
elmat.SetSize (test_dof, trial_dof);
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
int order = (Trans.OrderW() + test_fe.GetOrder() + trial_fe.GetOrder() - 2);
ir = &IntRules.Get(test_fe.GetGeomType(), order);
}
elmat = 0.0;
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
Trans.SetIntPoint (&ip);
trial_fe.CalcPhysDVShape(Trans, trial_dvshape);
test_fe.CalcPhysDVShape(Trans, test_dvshape);
w = ip.weight * Trans.Weight();
if (MQ)
{
// flat_dvs has interlaced vector component and derivative direction in 2nd index
// Matrix needs to be (dim*dim) X (dim*dim)
// indexing: 2nd index = dif_dir + comp_dir*dim
// u_x, v_x, u_y, v_y
// u_x, v_x, w_x, u_y, v_y, w_y, u_z, v_z, w_z
MQ->Eval(K, Trans, ip);
K *= w;
Mult(test_fdvs,K,tmp);
AddMultABt(tmp,trial_fdvs,elmat);
}
else if (DQ)
{
// Vector needs to be (dim*dim), see comment for MQ case
DQ->Eval(D, Trans, ip);
D *= w;
AddMultADBt(test_fdvs,D,trial_fdvs,elmat);
}
else
{
if (Q)
{
w *= Q -> Eval (Trans, ip);
}
AddMult_a_ABt(w,test_fdvs,trial_fdvs,elmat);
}
}
}
void VectorDivergenceIntegrator::AssembleElementMatrix2(
const FiniteElement &trial_fe,
const FiniteElement &test_fe,
@@ -3066,7 +3222,6 @@ void VectorDiffusionIntegrator::AssembleElementMatrix(
for (int i = 0; i < ir -> GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
el.CalcDShape(ip, dshape);
@@ -3991,6 +4146,279 @@ const IntegrationRule &DGDiffusionIntegrator::GetRule(
return GetRule(order, T.GetGeometryType());
}
void VectorFEDGDiffusionIntegrator::AssembleFaceMatrix(
const FiniteElement &el1, const FiniteElement &el2,
FaceElementTransformations &Trans, DenseMatrix &elmat)
{
int ndof1, ndof2, ndofs;
bool kappa_is_nonzero = (kappa != 0.);
real_t w, wq = 0.0;
dim = el1.GetDim();
ndof1 = el1.GetDof();
nor.SetSize(dim);
nh.SetSize(dim);
ni.SetSize(dim);
if (MQ)
{
mq.SetSize(dim);
}
vshape1.SetSize(ndof1, dim);
dvshape1.SetSize(ndof1, dim, dim);
dvshape1dn.SetSize(ndof1, dim);
DenseMatrix dvshape1_flat;
dvshape1.GetDenseMatrix2(dvshape1_flat);
Vector dvshape1dn_flat(dvshape1dn.GetData(),ndof1*dim);
Vector vshape1_flat(vshape1.GetData(),ndof1*dim);
if (Trans.Elem2No >= 0)
{
ndof2 = el2.GetDof();
vshape2.SetSize(ndof2, dim);
dvshape2.SetSize(ndof2, dim, dim);
dvshape2dn.SetSize(ndof2, dim);
}
else
{
ndof2 = 0;
}
ndofs = ndof1 + ndof2;
elmat.SetSize(ndofs);
elmat = 0.0;
if (kappa_is_nonzero)
{
jmat.SetSize(ndofs);
jmat = 0.;
}
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
const int order = (ndof2) ? max(el1.GetOrder(),
el2.GetOrder()) : el1.GetOrder();
ir = &GetRule(order, Trans);
}
// assemble: < {(Q \nabla u).n},[v] > --> elmat
// kappa < {h^{-1} Q} [u],[v] > --> jmat
for (int p = 0; p < ir->GetNPoints(); p++)
{
const IntegrationPoint &ip = ir->IntPoint(p);
// Set the integration point in the face and the neighboring elements
Trans.SetAllIntPoints(&ip);
// Access the neighboring elements' integration points
// Note: eip2 will only contain valid data if Elem2 exists
const IntegrationPoint &eip1 = Trans.GetElement1IntPoint();
if (dim == 1)
{
nor(0) = 2*eip1.x - 1.0;
}
else
{
CalcOrtho(Trans.Jacobian(), nor);
}
el1.CalcVShape(*Trans.Elem1, vshape1);
el1.CalcPhysDVShape(*Trans.Elem1, dvshape1);
w = ip.weight;
if (ndof2)
{
w /= 2;
}
if (!MQ)
{
if (Q)
{
w *= Q->Eval(*Trans.Elem1, eip1);
}
ni.Set(w, nor);
}
else
{
nh.Set(w, nor);
MQ->Eval(mq, *Trans.Elem1, eip1);
mq.MultTranspose(nh, ni);
}
if (kappa_is_nonzero)
{
real_t hn = Trans.Elem1->Weight()/Trans.Weight();
wq = kappa*(ni*nor)/(Trans.Weight()*hn);
}
// Note: in the jump term, we use 1/h1 = |nor|/det(J1) which is
// independent of Loc1 and always gives the size of element 1 in
// direction perpendicular to the face. Indeed, for linear transformation
// |nor|=measure(face)/measure(ref. face),
// det(J1)=measure(element)/measure(ref. element),
// and the ratios measure(ref. element)/measure(ref. face) are
// compatible for all element/face pairs.
// For example: meas(ref. tetrahedron)/meas(ref. triangle) = 1/3, and
// for any tetrahedron vol(tet)=(1/3)*height*area(base).
// For interior faces: q_e/h_e=(q1/h1+q2/h2)/2.
dvshape1_flat.Mult(ni, dvshape1dn_flat);
for (int i = 0; i < ndof1; i++)
for (int j = 0; j < ndof1; j++)
for (int d = 0; d < dim; d++)
{
elmat(i, j) += vshape1(i,d) * dvshape1dn(j,d);
}
/* if (ndof2) Modification of this code is necessary for the DG case
{
el2.CalcVShape(eip2, vshape2);
el2.CalcDVShape(eip2, dvshape2);
w = ip.weight/2/Trans.Elem2->Weight();
if (!MQ)
{
if (Q)
{
w *= Q->Eval(*Trans.Elem2, eip2);
}
ni.Set(w, nor);
}
else
{
nh.Set(w, nor);
MQ->Eval(mq, *Trans.Elem2, eip2);
mq.MultTranspose(nh, ni);
}
CalcAdjugate(Trans.Elem2->Jacobian(), adjJ);
adjJ.Mult(ni, nh);
if (kappa_is_nonzero)
{
wq += ni * nor;
}
dvshape2_flat.Mult(nh, dvshape2dn_flat); // dshape2.Mult(nh, dshape2dn);
for (int i = 0; i < ndof1; i++)
for (int j = 0; j < ndof2; j++)
for (int d = 0; d < dim; d++)
{
elmat(i, ndof1 + j) += vshape1(i,d) * dvshape2dn(j,d);
}
for (int i = 0; i < ndof2; i++)
for (int j = 0; j < ndof1; j++)
for (int d = 0; d < dim; d++)
{
elmat(ndof1 + i, j) -= vshape2(i,d) * dvshape1dn(j,d);
}
for (int i = 0; i < ndof2; i++)
for (int j = 0; j < ndof2; j++)
for (int d = 0; d < dim; d++)
{
elmat(ndof1 + i, ndof1 + j) -= vshape2(i,d) * dvshape2dn(j,d);
}
}*/
if (kappa_is_nonzero)
{
// only assemble the lower triangular part of jmat
for (int i = 0; i < ndof1; i++)
{
//const real_t wsi = wq*vshape1(i);
for (int j = 0; j <= i; j++)
{
for (int d = 0; d < dim; d++)
{
jmat(i, j) += wq*vshape1(i,d) * vshape1(j,d);
}
}
}
/* if (ndof2) Modification of this code is necessary for the DG case
{
for (int i = 0; i < ndof2; i++)
{
const int i2 = ndof1 + i;
const real_t wsi = wq*shape2(i);
for (int j = 0; j < ndof1; j++)
{
jmat(i2, j) -= wsi * shape1(j);
}
for (int j = 0; j <= i; j++)
{
jmat(i2, ndof1 + j) += wsi * shape2(j);
}
}
}*/
}
}
if (kappa_is_nonzero)
{
for (int i = 0; i < ndofs; i++)
{
for (int j = 0; j < i; j++)
{
real_t aij = elmat(i,j), aji = elmat(j,i), mij = jmat(i,j);
elmat(i,j) = sigma*aji - aij + mij;
elmat(j,i) = sigma*aij - aji + mij;
}
elmat(i,i) = (sigma - 1.)*elmat(i,i) + jmat(i,i);
}
}
else
{
for (int i = 0; i < ndofs; i++)
{
for (int j = 0; j < i; j++)
{
real_t aij = elmat(i,j), aji = elmat(j,i);
elmat(i,j) = sigma*aji - aij;
elmat(j,i) = sigma*aij - aji;
}
elmat(i,i) *= (sigma - 1.);
}
}
}
VectorFEDGDiffusionIntegrator::VectorFEDGDiffusionIntegrator(const real_t s,
const real_t k)
: sigma(s), kappa(k)
{
}
VectorFEDGDiffusionIntegrator::VectorFEDGDiffusionIntegrator(Coefficient &q,
const real_t s,
const real_t k)
: VectorFEDGDiffusionIntegrator(s, k)
{
Q = &q;
}
VectorFEDGDiffusionIntegrator::VectorFEDGDiffusionIntegrator(
MatrixCoefficient &q,
const real_t s, const real_t k)
: VectorFEDGDiffusionIntegrator(s, k)
{
MQ = &q;
}
const IntegrationRule &VectorFEDGDiffusionIntegrator::GetRule(
int order, Geometry::Type geom)
{
// order is typically the maximum of the order of the left and right elements
// neighboring the given face.
return IntRules.Get(geom, 2*order);
}
const IntegrationRule &VectorFEDGDiffusionIntegrator::GetRule(
int order, FaceElementTransformations &T)
{
return GetRule(order, T.GetGeometryType());
}
// static method
void DGElasticityIntegrator::AssembleBlock(
const int dim, const int row_ndofs, const int col_ndofs,
@@ -4492,8 +4920,8 @@ void NormalTraceIntegrator::AssembleTraceFaceMatrix(int elem,
MFEM_VERIFY(elem == Trans.Elem2->ElementNo, "Elem != Trans.Elem2->ElementNo");
}
real_t scale = 1.0;
if (iel != elem) { scale = -1.; }
real_t scale = alpha;
if (iel != elem) { scale *= -1.0; }
for (int p = 0; p < ir->GetNPoints(); p++)
{
+159 -46
View File
@@ -2596,41 +2596,40 @@ public:
by scalar FE through standard transformation. */
class VectorMassIntegrator: public BilinearFormIntegrator
{
private:
int vdim;
int vdim = -1, Q_order = 0;
Vector shape, te_shape, vec;
DenseMatrix partelmat;
DenseMatrix mcoeff;
int Q_order;
protected:
Coefficient *Q;
VectorCoefficient *VQ;
MatrixCoefficient *MQ;
Coefficient *Q = nullptr;
VectorCoefficient *VQ = nullptr;
MatrixCoefficient *MQ = nullptr;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, ne, nq, dofs1D, quad1D;
int ne, dim, dofs1D, quad1D, coeff_vdim;
Vector pa_data;
public:
/// Construct an integrator with coefficient 1.0
VectorMassIntegrator()
: vdim(-1), Q_order(0), Q(NULL), VQ(NULL), MQ(NULL) { }
VectorMassIntegrator() = default;
/** Construct an integrator with scalar coefficient q. If possible, save
memory by using a scalar integrator since the resulting matrix is block
diagonal with the same diagonal block repeated. */
VectorMassIntegrator(Coefficient &q, int qo = 0)
: vdim(-1), Q_order(qo), Q(&q), VQ(NULL), MQ(NULL) { }
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir)
: BilinearFormIntegrator(ir), vdim(-1), Q_order(0), Q(&q), VQ(NULL),
MQ(NULL) { }
VectorMassIntegrator(Coefficient &q, int qo = 0): Q_order(qo), Q(&q) { }
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir):
BilinearFormIntegrator(ir), Q(&q) { }
/// Construct an integrator with diagonal coefficient q
VectorMassIntegrator(VectorCoefficient &q, int qo = 0)
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(&q), MQ(NULL) { }
VectorMassIntegrator(VectorCoefficient &q, int qo = 0):
vdim(q.GetVDim()), Q_order(qo), VQ(&q) { }
/// Construct an integrator with matrix coefficient q
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0)
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(NULL), MQ(&q) { }
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0):
vdim(q.GetVDim()), Q_order(qo), MQ(&q) { }
int GetVDim() const { return vdim; }
void SetVDim(int vdim_) { vdim = vdim_; }
@@ -2642,6 +2641,7 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat) override;
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleMF(const FiniteElementSpace &fes) override;
@@ -2650,6 +2650,15 @@ public:
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultMF(const Vector &x, Vector &y) const override;
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
using VectorMassAddMultPAType =
void(*)(const int, const int,
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
MFEM_REGISTER_KERNELS(VectorMassAddMultPA,
VectorMassAddMultPAType,
(int, int, int));
};
@@ -2665,6 +2674,7 @@ class VectorFEDivergenceIntegrator : public BilinearFormIntegrator
{
protected:
Coefficient *Q;
real_t alpha;
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &trial_fes,
@@ -2686,8 +2696,8 @@ private:
int dim, ne, dofs1D, L2dofs1D, quad1D;
public:
VectorFEDivergenceIntegrator() { Q = NULL; }
VectorFEDivergenceIntegrator(Coefficient &q) { Q = &q; }
VectorFEDivergenceIntegrator(real_t a = 1.0) { alpha = a; Q = NULL; }
VectorFEDivergenceIntegrator(Coefficient &q, real_t a = 1.0) { alpha = a; Q = &q; }
void AssembleElementMatrix(const FiniteElement &el,
ElementTransformation &Trans,
DenseMatrix &elmat) override { }
@@ -3008,6 +3018,49 @@ public:
const Coefficient *GetCoefficient() const { return Q; }
};
/** Integrator for $(Q u, v)$, where $Q$ is an optional coefficient (of type scalar,
vector (diagonal matrix), or matrix), trial function $u$ is in $H(curl$ or
$H(div)$, and test function $v$ is in $H(curl$, $H(div)$, or $v=(v_1,\dots,v_n)$, where
$v_i$ are in $H^1$. */
class VectorFEDiffusionIntegrator: public BilinearFormIntegrator
{
private:
void Init(Coefficient *q, DiagonalMatrixCoefficient *dq, MatrixCoefficient *mq)
{ Q = q; DQ = dq; MQ = mq; }
#ifndef MFEM_THREAD_SAFE
Vector D;
DenseMatrix K;
DenseTensor test_dvshape;
DenseTensor trial_dvshape;
#endif
protected:
Coefficient *Q;
DiagonalMatrixCoefficient *DQ;
MatrixCoefficient *MQ;
public:
VectorFEDiffusionIntegrator() { Init(NULL, NULL, NULL); }
VectorFEDiffusionIntegrator(Coefficient *q_) { Init(q_, NULL, NULL); }
VectorFEDiffusionIntegrator(Coefficient &q) { Init(&q, NULL, NULL); }
VectorFEDiffusionIntegrator(DiagonalMatrixCoefficient *dq_) { Init(NULL, dq_, NULL); }
VectorFEDiffusionIntegrator(DiagonalMatrixCoefficient &dq) { Init(NULL, &dq, NULL); }
VectorFEDiffusionIntegrator(MatrixCoefficient *mq_) { Init(NULL, NULL, mq_); }
VectorFEDiffusionIntegrator(MatrixCoefficient &mq) { Init(NULL, NULL, &mq); }
virtual void AssembleElementMatrix(const FiniteElement &el,
ElementTransformation &Trans,
DenseMatrix &elmat);
virtual void AssembleElementMatrix2(const FiniteElement &trial_fe,
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat);
const Coefficient *GetCoefficient() const { return Q; }
};
/** Integrator for $(Q \nabla \cdot u, v)$ where $u=(u_1,\cdots,u_n)$ and all $u_i$ are in the same
scalar FE space; $v$ is also in a (different) scalar FE space. */
class VectorDivergenceIntegrator : public BilinearFormIntegrator
@@ -3120,23 +3173,21 @@ public:
to be the spatial dimension (i.e. 2-dimension or 3-dimension). */
class VectorDiffusionIntegrator : public BilinearFormIntegrator
{
protected:
Coefficient *Q = NULL;
VectorCoefficient *VQ = NULL;
MatrixCoefficient *MQ = NULL;
int vdim = -1;
DenseMatrix dshape, dshapedxt, pelmat;
DenseMatrix mcoeff;
Vector vcoeff;
protected:
Coefficient *Q = nullptr;
VectorCoefficient *VQ = nullptr;
MatrixCoefficient *MQ = nullptr;
// PA extension
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int dim, sdim, ne, dofs1D, quad1D;
int ne, dim, sdim, dofs1D, quad1D, coeff_vdim;
Vector pa_data;
private:
DenseMatrix dshape, dshapedxt, pelmat;
int vdim = -1;
DenseMatrix mcoeff;
Vector vcoeff;
public:
VectorDiffusionIntegrator(const IntegrationRule *ir = nullptr);
@@ -3189,6 +3240,7 @@ public:
void AssembleElementVector(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun, Vector &elvect) override;
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleMF(const FiniteElementSpace &fes) override;
@@ -3198,13 +3250,11 @@ public:
void AddMultMF(const Vector &x, Vector &y) const override;
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
/// arguments: ne, B, G, Bt, Gt, pa_data, x, y, d1d, q1d, vdim
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
const Array<real_t> &,
const Array<real_t> &,
const Array<real_t> &, const Vector &,
const Vector &, Vector &, const int,
const int, const int);
/// arguments: ne, coeff_vdim, B, G, pa_data, x, y, d1d, q1d, vdim
using ApplyKernelType = void (*)(const int, const int,
const Array<real_t> &, const Array<real_t> &,
const Vector &, const Vector &, Vector &,
const int, const int, const int);
/// arguments: dim, vdim, d1d, q1d
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int, int));
@@ -3215,10 +3265,7 @@ public:
ApplyPAKernels::Specialization<DIM, VDIM, D1D, Q1D>::Add();
}
struct Kernels
{
Kernels();
};
// struct Kernels { Kernels(); };
};
/** Integrator for the linear elasticity form:
@@ -3496,7 +3543,8 @@ public:
- $\sigma = +1$, $\kappa > 0$: non-symmetric interior penalty (NIPG) method,
- $\sigma = +1$, $\kappa = 0$: the method of Baumann and Oden.
\todo Clarify used notation. */
\todo Clarify used notation
\todo Please add clarification also to VectorFEDGDiffusionIntegrator */
class DGDiffusionIntegrator : public BilinearFormIntegrator
{
protected:
@@ -3508,7 +3556,6 @@ protected:
Vector shape1, shape2, dshape1dn, dshape2dn, nor, nh, ni;
DenseMatrix jmat, dshape1, dshape2, mq, adjJ;
// PA extension
Vector pa_data; // (Q, h, dot(n,J)|el0, dot(n,J)|el1)
const DofToQuad *maps; ///< Not owned
@@ -3565,6 +3612,61 @@ private:
void SetupPA(const FiniteElementSpace &fes, FaceType type);
};
/** Integrator for the DG form:
$$
- \langle \{(Q \nabla u) \cdot n\}, [v] \rangle + \sigma \langle [u], \{(Q \nabla v) \cdot n \} \rangle
+ \kappa \langle \{h^{-1} Q\} [u], [v] \rangle
$$
where $Q$ is a scalar or matrix diffusion coefficient and $u$, $v$ are the trial
and test spaces, respectively. These spaces are defined using vector-valued
finite elements. The parameters $\sigma$ and $\kappa$ determine the
DG method to be used (when this integrator is added to the "broken"
DiffusionIntegrator):
- $\sigma = -1$, $\kappa \geq \kappa_0$: symm. interior penalty (IP or SIPG) method,
- $\sigma = +1$, $\kappa > 0$: non-symmetric interior penalty (NIPG) method,
- $\sigma = +1$, $\kappa = 0$: the method of Baumann and Oden.
This class currently only works when used a weak boundary condition.
The DG case, when defining the penalty on the interior faces, requires either:
- Vector-valued finite elements NURBS defined on multiple patches.
- Standard Vector-valued finite elements to also have gradients.
*/
class VectorFEDGDiffusionIntegrator : public BilinearFormIntegrator
{
protected:
Coefficient *Q = nullptr;
MatrixCoefficient *MQ = nullptr;
real_t sigma, kappa;
int dim;
// these are not thread-safe!
Vector nor, nh, ni;
DenseMatrix jmat, mq;
// these are not thread-safe!
DenseMatrix vshape1, vshape2, dvshape1dn, dvshape2dn;
DenseTensor dvshape1, dvshape2;
public:
VectorFEDGDiffusionIntegrator(const real_t s, const real_t k);
VectorFEDGDiffusionIntegrator(Coefficient &q, const real_t s, const real_t k);
VectorFEDGDiffusionIntegrator(MatrixCoefficient &q, const real_t s,
const real_t k);
using BilinearFormIntegrator::AssembleFaceMatrix;
void AssembleFaceMatrix(const FiniteElement &el1, const FiniteElement &el2,
FaceElementTransformations &Trans,
DenseMatrix &elmat) override;
bool RequiresFaceNormalDerivatives() const override { return true; }
const IntegrationRule &GetRule(int order, FaceElementTransformations &T);
const IntegrationRule &GetRule(int order, Geometry::Type geom);
real_t GetPenaltyParameter() const { return kappa; }
};
/** Integrator for the "BR2" diffusion stabilization term
$$
\sum_e \eta (r_e([u]), r_e([v]))
@@ -3803,14 +3905,25 @@ class NormalTraceIntegrator : public BilinearFormIntegrator
private:
Vector face_shape, normal, shape_n;
DenseMatrix shape;
real_t alpha;
public:
NormalTraceIntegrator() { }
NormalTraceIntegrator(real_t a = 1.0) : alpha(a) { }
void AssembleTraceFaceMatrix(int ielem,
const FiniteElement &trial_face_fe,
const FiniteElement &test_fe,
FaceElementTransformations &Trans,
DenseMatrix &elmat) override;
void AssembleFaceMatrix(const FiniteElement &trial_face_fe,
const FiniteElement &test_fe1,
const FiniteElement &test_fe2,
FaceElementTransformations &Trans,
DenseMatrix &elmat) override
{
AssembleTraceFaceMatrix(Trans.Elem1->ElementNo,
trial_face_fe, test_fe1, Trans,elmat);
}
};
+9 -5
View File
@@ -207,7 +207,8 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
}
}
PLBound::PLBound(FiniteElementSpace *fes, int ncp_i, int cp_type_i)
PLBound::PLBound(const FiniteElementSpace *fes, const int ncp_i,
const int cp_type_i)
{
MFEM_VERIFY(!fes->IsVariableOrder(),
"Variable order meshes not yet supported.");
@@ -264,7 +265,8 @@ PLBound::PLBound(FiniteElementSpace *fes, int ncp_i, int cp_type_i)
Setup(nb, ncp, b_type, cp_type, tol);
}
void PLBound::Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get1DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
real_t x,w;
intmin.SetSize(ncp);
@@ -346,7 +348,8 @@ void PLBound::Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get2DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
intmin.SetSize(ncp*ncp);
intmax.SetSize(ncp*ncp);
@@ -482,7 +485,8 @@ void PLBound::Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
void PLBound::Get3DBounds(const Vector &coeff, Vector &intmin,
Vector &intmax) const
{
int nb2 = nb*nb,
ncp2 = ncp*ncp,
@@ -624,7 +628,7 @@ void PLBound::Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const
}
}
void PLBound::GetNDBounds(int rdim, Vector &coeff,
void PLBound::GetNDBounds(const int rdim, const Vector &coeff,
Vector &intmin, Vector &intmax) const
{
if (rdim == 1)
+6 -5
View File
@@ -89,7 +89,8 @@ public:
}
// Constructor
PLBound(FiniteElementSpace *fes, int ncp_i = -1, int cp_type_i = 0);
PLBound(const FiniteElementSpace *fes,
const int ncp_i = -1, const int cp_type_i = 0);
// Get minimum number of control points needed to bound the given bases
int GetMinimumPointsForGivenBases(int nb_i, int b_type_i,
@@ -105,7 +106,7 @@ public:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D/2D/3D.
void GetNDBounds(int rdim, Vector &coeff,
void GetNDBounds(const int rdim, const Vector &coeff,
Vector &intmin, Vector &intmax) const;
/// Get number of control points used to compute the bounds.
@@ -113,15 +114,15 @@ public:
private:
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 1D.
void Get1DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get1DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 2D.
void Get2DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get2DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Compute piecewise linear bounds for the lexicographically-ordered
/// coefficients in @a coeff in 3D.
void Get3DBounds(Vector &coeff, Vector &intmin, Vector &intmax) const;
void Get3DBounds(const Vector &coeff, Vector &intmin, Vector &intmax) const;
/// Setup matrix used to compute values at given 1D locations in [0,1]
/// for Bernstein bases.
+72
View File
@@ -1085,6 +1085,29 @@ void SumCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void SumCoefficient::Project(QuadratureFunction &qf)
{
if (a == nullptr)
{
// qf = alpha*aConst + beta * b
const real_t d_alpha_a = aConst*alpha;
const real_t d_beta = beta;
b->Project(qf);
auto d_qf = qf.ReadWrite();
mfem::forall(qf.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_qf[i] = d_alpha_a + d_beta*d_qf[i];
});
}
else
{
a->Project(qf);
QuadratureFunction qf_b(*qf.GetSpace());
b->Project(qf_b);
add(alpha, qf, beta, qf_b, qf);
}
}
void ProductCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
@@ -1092,6 +1115,23 @@ void ProductCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void ProductCoefficient::Project(QuadratureFunction &qf)
{
if (a == nullptr)
{
// qf = aConst * b
b->Project(qf);
qf *= aConst;
}
else
{
a->Project(qf);
QuadratureFunction qf_b(qf.GetSpace());
b->Project(qf_b);
qf *= qf_b;
}
}
void RatioCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
@@ -1099,6 +1139,38 @@ void RatioCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void RatioCoefficient::Project(QuadratureFunction &qf)
{
if (b == nullptr)
{
if (a == nullptr)
{
qf = aConst / bConst;
}
else
{
a->Project(qf);
qf *= 1.0/bConst;
}
}
else
{
if (a == nullptr)
{
b->Project(qf);
qf.Reciprocal();
qf *= aConst;
}
else
{
a->Project(qf);
QuadratureFunction qf_b(qf.GetSpace());
b->Project(qf_b);
qf /= qf_b;
}
}
}
void PowerCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
+9
View File
@@ -1456,6 +1456,9 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the first term in the linear combination as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the first term in the linear combination
@@ -1637,6 +1640,9 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the first term in the product as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the first term in the product
@@ -1685,6 +1691,9 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the numerator in the ratio as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the numerator of the ratio
+281 -8
View File
@@ -11,14 +11,15 @@
#include "complex_fem.hpp"
#include "../general/forall.hpp"
#include "../general/text.hpp"
using namespace std;
namespace mfem
{
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
: Vector(2*(fes->GetVSize()))
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *f)
: Vector(2*(f->GetVSize())), fes(f), fec_owned(NULL)
{
UseDevice(true);
this->Vector::operator=(0.0);
@@ -28,12 +29,88 @@ ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
gfi = new GridFunction();
gfi->MakeRef(fes, *this, fes->GetVSize());
fes_sequence = fes->GetSequence();
}
ComplexGridFunction::ComplexGridFunction(Mesh *m, std::istream &input)
: Vector(), fes(NULL), fec_owned(NULL)
{
string buff;
// Grid functions are stored on the device
UseDevice(true);
input >> std::ws;
getline(input, buff); // 'ComplexGridFunction'
filter_dos(buff);
if (buff != "ComplexGridFunction")
{
MFEM_ABORT("unrecognized file header: " << buff);
}
fes = new FiniteElementSpace;
fec_owned = fes->Load(m, input);
skip_comment_lines(input, '#');
istream::int_type next_char = input.peek();
if (next_char == 'N') // First letter of "NURBS_patches"
{
getline(input, buff);
filter_dos(buff);
if (buff == "NURBS_patches")
{
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
}
else
{
MFEM_ABORT("unknown section: " << buff);
}
}
else
{
Vector::Load(input, 2*fes->GetVSize());
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
if (fes->Nonconforming() &&
fes->GetMesh()->ncmesh->IsLegacyLoaded())
{
// LegacyNCReorder();
MFEM_ABORT("LegacyNCReorder not supported for "
"ComplexGridFunction objects");
}
}
gfr = new GridFunction();
gfr->MakeRef(fes, *this, 0);
gfi = new GridFunction();
gfi->MakeRef(fes, *this, fes->GetVSize());
fes_sequence = fes->GetSequence();
}
void ComplexGridFunction::Destroy()
{
delete gfr; delete gfi;
if (fec_owned)
{
delete fes;
delete fec_owned;
fec_owned = NULL;
}
}
void
ComplexGridFunction::Update()
{
FiniteElementSpace *fes = gfr->FESpace();
if (fes->GetSequence() == fes_sequence)
{
return; // space and grid function are in sync, no-op
}
fes_sequence = fes->GetSequence();
const int vsize = fes->GetVSize();
const Operator *T = fes->GetUpdateOperator();
@@ -84,6 +161,17 @@ ComplexGridFunction::Update()
}
}
int ComplexGridFunction::VectorDim() const
{
const FiniteElement *fe = fes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return fes->GetVDim();
}
return fes->GetVDim()*std::max(fes->GetMesh()->SpaceDimension(),
fe->GetRangeDim());
}
void
ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
@@ -149,6 +237,35 @@ ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
gfi->SyncAliasMemory(*this);
}
void ComplexGridFunction::Save(std::ostream &os) const
{
os << "ComplexGridFunction\n";
fes->Save(os);
os << '\n';
if (fes->GetOrdering() == Ordering::byNODES)
{
Vector::Print(os, 1);
}
else
{
Vector::Print(os, fes->GetVDim());
}
os.flush();
}
void ComplexGridFunction::Save(const char *fname, int precision) const
{
ofstream ofs(fname);
ofs.precision(precision);
Save(ofs);
}
std::ostream &operator<<(std::ostream &os, const ComplexGridFunction &sol)
{
sol.Save(os);
return os;
}
ComplexLinearForm::ComplexLinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention convention)
@@ -654,8 +771,8 @@ SesquilinearForm::Update(FiniteElementSpace *nfes)
#ifdef MFEM_USE_MPI
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
: Vector(2*(pfes->GetVSize()))
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pf)
: Vector(2*(pf->GetVSize())), pfes(pf), fec_owned(NULL)
{
UseDevice(true);
this->Vector::operator=(0.0);
@@ -665,12 +782,105 @@ ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
pgfi = new ParGridFunction();
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
fes_sequence = pfes->GetSequence();
}
ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
: Vector(), pfes(NULL), fec_owned(NULL)
{
string buff;
// Grid functions are stored on the device
UseDevice(true);
input >> std::ws;
getline(input, buff); // 'ParComplexGridFunction'
filter_dos(buff);
if (buff != "ParComplexGridFunction")
{
MFEM_ABORT("unrecognized file header: " << buff);
}
FiniteElementSpace *fes = new FiniteElementSpace;
fec_owned = fes->Load(m, input);
pfes = new ParFiniteElementSpace(m, fec_owned, fes->GetVDim(),
fes->GetOrdering());
delete fes;
skip_comment_lines(input, '#');
istream::int_type next_char = input.peek();
if (next_char == 'N') // First letter of "NURBS_patches"
{
getline(input, buff);
filter_dos(buff);
if (buff == "NURBS_patches")
{
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
}
else
{
MFEM_ABORT("unknown section: " << buff);
}
}
else
{
int vsize = pfes->GetVSize();
Vector::Load(input, 2*vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
if (pfes->Nonconforming() &&
pfes->GetMesh()->ncmesh->IsLegacyLoaded())
{
// LegacyNCReorder();
MFEM_ABORT("LegacyNCReorder not supported for "
"ComplexGridFunction objects");
}
}
pgfr = new ParGridFunction();
pgfr->MakeRef(pfes, *this, 0);
pgfi = new ParGridFunction();
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
fes_sequence = pfes->GetSequence();
}
void ParComplexGridFunction::Destroy()
{
delete pgfr; delete pgfi;
if (fec_owned)
{
delete pfes;
delete fec_owned;
fec_owned = NULL;
}
}
void
ParComplexGridFunction::Update()
{
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
if (pfes->GetSequence() == fes_sequence)
{
return; // space and grid function are in sync, no-op
}
fes_sequence = pfes->GetSequence();
const int vsize = pfes->GetVSize();
const Operator *T = pfes->GetUpdateOperator();
@@ -719,6 +929,17 @@ ParComplexGridFunction::Update()
}
}
int ParComplexGridFunction::VectorDim() const
{
const FiniteElement *fe = pfes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return pfes->GetVDim();
}
return pfes->GetVDim()*std::max(pfes->GetMesh()->SpaceDimension(),
fe->GetRangeDim());
}
void
ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
@@ -789,7 +1010,6 @@ ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
void
ParComplexGridFunction::Distribute(const Vector *tv)
{
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
const int tvsize = pfes->GetTrueVSize();
tv->Read();
@@ -807,7 +1027,6 @@ ParComplexGridFunction::Distribute(const Vector *tv)
void
ParComplexGridFunction::ParallelProject(Vector &tv) const
{
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
const int tvsize = pfes->GetTrueVSize();
tv.Write();
@@ -825,6 +1044,60 @@ ParComplexGridFunction::ParallelProject(Vector &tv) const
tvi.SyncAliasMemory(tv);
}
void ParComplexGridFunction::Save(std::ostream &os) const
{
os << "ParComplexGridFunction\n";
pfes->Save(os);
os << '\n';
int vsize = pfes->GetVSize();
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
if (pfes->GetOrdering() == Ordering::byNODES)
{
Vector::Print(os, 1);
}
else
{
Vector::Print(os, pfes->GetVDim());
}
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
os.flush();
}
void ParComplexGridFunction::Save(const char *fname, int precision) const
{
int rank = pfes->GetMyRank();
ostringstream fname_with_suffix;
fname_with_suffix << fname << "." << setfill('0') << setw(6) << rank;
ofstream ofs(fname_with_suffix.str().c_str());
ofs.precision(precision);
Save(ofs);
}
std::ostream &operator<<(std::ostream &os, const ParComplexGridFunction &sol)
{
sol.Save(os);
return os;
}
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ComplexOperator::Convention
+159 -16
View File
@@ -35,15 +35,53 @@ private:
GridFunction * gfi;
protected:
void Destroy() { delete gfr; delete gfi; }
/// FE space on which the grid function lives. Owned if #fec_owned
/// is not NULL.
FiniteElementSpace *fes;
/** @brief Used when the grid function is read from a file. It can also be
set explicitly, see MakeOwner().
If not NULL, this pointer is owned by the ComplexGridFunction. */
FiniteElementCollection *fec_owned;
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
void Destroy();
public:
/** @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
ComplexGridFunction(FiniteElementSpace *f);
/** @brief Construct a ComplexGridFunction on the given Mesh, using the data
from @a input.
The content of @a input should be in the format created by the method
Save(). The reconstructed FiniteElementSpace and FiniteElementCollection
are owned by the ComplexGridFunction. */
ComplexGridFunction(Mesh *m, std::istream &input);
void Update();
/** Return update counter, similar to Mesh::GetSequence(). Used to
check if it is up to date with the space. */
long GetSequence() const { return fes_sequence; }
/// Make the ComplexGridFunction the owner of #fec_owned and #fes.
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
of #fec_owned and #fes is taken away. */
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
/// Returns a pointer to the FiniteElementCollection used to
/// construct this ComplexGridFunction if this class owns that
/// object. Otherwise this function will return NULL.
FiniteElementCollection *OwnFEC() { return fec_owned; }
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
/// underlying #fes
int VectorDim() const;
/// Assign constant values to the ComplexGridFunction data.
ComplexGridFunction &operator=(const std::complex<real_t> & value)
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
@@ -63,8 +101,8 @@ public:
VectorCoefficient &imag_coeff,
Array<int> &attr);
FiniteElementSpace *FESpace() { return gfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return gfr->FESpace(); }
FiniteElementSpace *FESpace() { return fes; }
const FiniteElementSpace *FESpace() const { return fes; }
GridFunction & real() { return *gfr; }
GridFunction & imag() { return *gfi; }
@@ -79,11 +117,52 @@ public:
/// @a gfr and @a gfi to match the ComplexGridFunction.
void SyncAlias() { gfr->SyncAliasMemory(*this); gfi->SyncAliasMemory(*this); }
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued scalar fields
///
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
real_t err_r = gfr->ComputeL2Error(exsolr, irs);
real_t err_i = gfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
}
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued vector fields
///
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
{
real_t err_r = gfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = gfi->ComputeL2Error(exsoli, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
/// Save the ComplexGridFunction to an output stream.
virtual void Save(std::ostream &out) const;
/// Save the ComplexGridFunction to a file
/** The given @a precision will be used for ASCII output. */
virtual void Save(const char *fname, int precision=16) const;
/// Destroys the grid function.
virtual ~ComplexGridFunction() { Destroy(); }
};
/** Overload operator<< for std::ostream and ComplexGridFunction; not valid
for the class ParComplexGridFunction */
std::ostream &operator<<(std::ostream &out, const ComplexGridFunction &sol);
/** Class for a complex-valued linear form
The @a convention argument in the class's constructor is documented in the
@@ -345,12 +424,23 @@ public:
class ParComplexGridFunction : public Vector
{
private:
ParGridFunction * pgfr;
ParGridFunction * pgfi;
protected:
void Destroy() { delete pgfr; delete pgfi; }
/// FE space on which the grid function lives. Owned if #fec_owned
/// is not NULL.
ParFiniteElementSpace *pfes;
/** @brief Used when the grid function is read from a file. It can also be
set explicitly, see MakeOwner().
If not NULL, this pointer is owned by the ParComplexGridFunction. */
FiniteElementCollection *fec_owned;
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
void Destroy();
public:
@@ -358,8 +448,33 @@ public:
ParFiniteElementSpace @a *pf. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
/** @brief Construct a ParComplexGridFunction on a given ParMesh,
@a pmesh, reading from an std::istream.
In the process, a ParFiniteElementSpace and a FiniteElementCollection are
constructed. The new ParComplexGridFunction assumes ownership of both. */
ParComplexGridFunction(ParMesh *pmesh, std::istream &input);
void Update();
/** Return update counter, similar to Mesh::GetSequence(). Used to
check if it is up to date with the space. */
long GetSequence() const { return fes_sequence; }
/// Make the ParComplexGridFunction the owner of #fec_owned and #pfes.
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
of #fec_owned and #pfes is taken away. */
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
/// Returns a pointer to the FiniteElementCollection used to
/// construct this ParComplexGridFunction if this class owns that
/// object. Otherwise this function will return NULL.
FiniteElementCollection *OwnFEC() { return fec_owned; }
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
/// underlying #pfes
int VectorDim() const;
/// Assign constant values to the ParComplexGridFunction data.
ParComplexGridFunction &operator=(const std::complex<real_t> & value)
{ *pgfr = value.real(); *pgfi = value.imag(); return *this; }
@@ -385,11 +500,11 @@ public:
/// Returns the vector restricted to the true dofs.
void ParallelProject(Vector &tv) const;
FiniteElementSpace *FESpace() { return pgfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return pgfr->FESpace(); }
FiniteElementSpace *FESpace() { return pfes; }
const FiniteElementSpace *FESpace() const { return pfes; }
ParFiniteElementSpace *ParFESpace() { return pgfr->ParFESpace(); }
const ParFiniteElementSpace *ParFESpace() const { return pgfr->ParFESpace(); }
ParFiniteElementSpace *ParFESpace() { return pfes; }
const ParFiniteElementSpace *ParFESpace() const { return pfes; }
ParGridFunction & real() { return *pgfr; }
ParGridFunction & imag() { return *pgfi; }
@@ -402,17 +517,32 @@ public:
/// Update the alias memory location of the real and imaginary
/// ParGridFunction @a pgfr and @a pgfi to match the ParComplexGridFunction.
void SyncAlias() { pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
void SyncAlias()
{ pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
/// scalar fields
///
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
{
real_t err_r = pgfr->ComputeL2Error(exsolr, irs);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return hypot(err_r, err_i);
}
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
/// vector fields
///
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
@@ -420,15 +550,28 @@ public:
{
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
return hypot(err_r, err_i);
}
/// Save the local portion of the ParComplexGridFunction
/** This differs from the serial ComplexGridFunction::Save in that it
takes into account the signs of the local dofs. */
void Save(std::ostream &out) const;
/// Save the ParComplexGridFunction to files
/** Saves one file for each MPI rank. The files will be given suffixes
according to the MPI rank. The given @a precision will be used for ASCII
output. */
void Save(const char *fname, int precision=16) const;
/// Destroys grid function.
virtual ~ParComplexGridFunction() { Destroy(); }
};
/** Overload operator<< for std::ostream and ParComplexGridFunction */
std::ostream &operator<<(std::ostream &out, const ParComplexGridFunction &sol);
/** Class for a complex-valued, parallel linear form
The @a convention argument in the class's constructor is documented in the
+169 -11
View File
@@ -310,9 +310,9 @@ void DataCollection::SaveField(const std::string &field_name)
}
}
void DataCollection::SaveQField(const std::string &q_field_name)
void DataCollection::SaveQField(const std::string &field_name)
{
QFieldMapIterator it = q_field_map.find(q_field_name);
QFieldMapIterator it = q_field_map.find(field_name);
if (it != q_field_map.end())
{
SaveOneQField(it);
@@ -780,6 +780,11 @@ void ParaViewDataCollectionBase::SetHighOrderOutput(bool high_order_output_)
high_order_output = high_order_output_;
}
void ParaViewDataCollectionBase::SetBoundaryOutput(bool bdr_output_)
{
bdr_output = bdr_output_;
}
void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
{
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
@@ -935,16 +940,19 @@ void ParaViewDataCollection::Save()
std::string vtu_prefix = col_path + "/" + GenerateVTUPath() + "/";
// Save the local part of the mesh and grid functions fields to the local
// VTU file
// VTU file. Also save coefficient fields.
{
std::ofstream os(vtu_prefix + GenerateVTUFileName("proc", myid));
os.precision(precision);
SaveDataVTU(os, levels_of_detail);
}
// Save the local part of the quadrature function fields
// Save the local part of the quadrature function fields.
for (const auto &qfield : q_field_map)
{
MFEM_VERIFY(!bdr_output,
"QuadratureFunction output is not supported for "
"ParaViewDataCollection on domain boundary!");
const std::string &field_name = qfield.first;
std::ofstream os(vtu_prefix + GenerateVTUFileName(field_name, myid));
qfield.second->SaveVTU(os, pv_data_format, GetCompressionLevel(), field_name);
@@ -960,7 +968,7 @@ void ParaViewDataCollection::Save()
std::ofstream pvtu_out(vtu_prefix + GeneratePVTUFileName("data"));
WritePVTUHeader(pvtu_out);
// Grid function fields
// Grid function fields and coefficient fields
pvtu_out << "<PPointData>\n";
for (auto &field_it : field_map)
{
@@ -971,7 +979,24 @@ void ParaViewDataCollection::Save()
<< VTKComponentLabels(vec_dim) << " "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
for (auto &field_it : coeff_field_map)
{
int vec_dim = 1;
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << field_it.first
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
for (auto &field_it : vcoeff_field_map)
{
int vec_dim = field_it.second->GetVDim();
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << field_it.first
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
pvtu_out << "</PPointData>\n";
// Element attributes
pvtu_out << "<PCellData>\n";
pvtu_out << "\t<PDataArray type=\"Int32\" Name=\"" << "attribute"
@@ -1069,7 +1094,8 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
}
os << " version=\"2.2\" byte_order=\"" << VTKByteOrder() << "\">\n";
os << "<UnstructuredGrid>\n";
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel());
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel(),
bdr_output);
// dump out the grid functions as point data
os << "<PointData >\n";
@@ -1077,8 +1103,21 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
// iterate over all grid functions
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
MFEM_VERIFY(!bdr_output,
"GridFunction output is not supported for "
"ParaViewDataCollection on domain boundary!");
SaveGFieldVTU(os,ref,it);
}
// save the coefficient functions
// iterate over all Coefficient and VectorCoefficient functions
for (const auto &kv : coeff_field_map)
{
SaveCoeffFieldVTU(os, ref, kv.first, *kv.second);
}
for (const auto &kv : vcoeff_field_map)
{
SaveVCoeffFieldVTU(os, ref, kv.first, *kv.second);
}
os << "</PointData>\n";
// close the mesh
os << "</Piece>\n"; // close the piece open in the PrintVTU method
@@ -1101,7 +1140,6 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
<< "format=\"" << GetDataFormatString() << "\" >" << '\n';
if (vec_dim == 1)
{
// scalar data
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
@@ -1131,11 +1169,131 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
}
}
}
if (IsBinaryFormat())
if (pv_data_format != VTKFormat::ASCII)
{
WriteVTKEncodedCompressed(os,buf.data(),buf.size(),GetCompressionLevel());
os << '\n';
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
}
os << "</DataArray>" << std::endl;
}
void ParaViewDataCollection::SaveCoeffFieldVTU(std::ostream &os, int ref_,
const std::string &name, Coefficient &coeff)
{
RefinedGeometry *RefG;
real_t val;
std::vector<char> buf;
int vec_dim = 1;
os << "<DataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << name
<< "\" NumberOfComponents=\"" << vec_dim << "\""
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
{
// scalar data
if (!bdr_output)
{
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
val = coeff.Eval(*eltrans, ip);
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
}
}
}
else
{
for (int i = 0; i < mesh->GetNBE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
val = coeff.Eval(*eltrans, ip);
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
}
}
}
}
if (pv_data_format != VTKFormat::ASCII)
{
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
}
os << "</DataArray>" << std::endl;
}
void ParaViewDataCollection::SaveVCoeffFieldVTU(std::ostream &os, int ref_,
const std::string &name, VectorCoefficient &coeff)
{
RefinedGeometry *RefG;
Vector val;
std::vector<char> buf;
int vec_dim = coeff.GetVDim();
os << "<DataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << name
<< "\" NumberOfComponents=\"" << vec_dim << "\""
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
{
// vector data
if (!bdr_output)
{
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
coeff.Eval(val, *eltrans, ip);
for (int jj = 0; jj < val.Size(); jj++)
{
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
}
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
}
}
}
else
{
for (int i = 0; i < mesh->GetNBE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
coeff.Eval(val, *eltrans, ip);
for (int jj = 0; jj < val.Size(); jj++)
{
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
}
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
}
}
}
}
if (pv_data_format != VTKFormat::ASCII)
{
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
}
os << "</DataArray>" << std::endl;
}
+47 -11
View File
@@ -133,6 +133,7 @@ private:
/// A collection of named QuadratureFunctions
typedef NamedFieldsMap<QuadratureFunction> QFieldMap;
public:
typedef GFieldMap::MapType FieldMapType;
typedef GFieldMap::iterator FieldMapIterator;
@@ -249,10 +250,9 @@ public:
{ field_map.Deregister(field_name, own_data); }
/// Add a QuadratureFunction to the collection.
virtual void RegisterQField(const std::string& q_field_name,
virtual void RegisterQField(const std::string& field_name,
QuadratureFunction *qf)
{ q_field_map.Register(q_field_name, qf, own_data); }
{ q_field_map.Register(field_name, qf, own_data); }
/// Remove a QuadratureFunction from the collection
virtual void DeregisterQField(const std::string& field_name)
@@ -280,13 +280,13 @@ public:
#endif
/// Check if a QuadratureFunction with the given name is in the collection.
bool HasQField(const std::string& q_field_name) const
{ return q_field_map.Has(q_field_name); }
bool HasQField(const std::string& field_name) const
{ return q_field_map.Has(field_name); }
/// Get a pointer to a QuadratureFunction in the collection.
/** Returns NULL if @a field_name is not in the collection. */
QuadratureFunction *GetQField(const std::string& q_field_name)
{ return q_field_map.Get(q_field_name); }
QuadratureFunction *GetQField(const std::string& field_name)
{ return q_field_map.Get(field_name); }
/// Get a const reference to the internal field map.
/** The keys in the map are the field names and the values are pointers to
@@ -302,11 +302,13 @@ public:
/// Get a pointer to the mesh in the collection
Mesh *GetMesh() { return mesh; }
/// Set/change the mesh associated with the collection
/** When passed a Mesh, assumes the serial case: MPI rank id is set to 0 and
MPI num_procs is set to 1. When passed a ParMesh, MPI info from the
ParMesh is used to set the DataCollection's MPI rank and num_procs. */
virtual void SetMesh(Mesh *new_mesh);
#ifdef MFEM_USE_MPI
/// Set/change the mesh associated with the collection.
/** For this case, @a comm is used to set the DataCollection's MPI rank id
@@ -369,8 +371,7 @@ public:
/// Save one field, assuming the collection directory already exists.
virtual void SaveField(const std::string &field_name);
/// Save one q-field, assuming the collection directory already exists.
virtual void SaveQField(const std::string &q_field_name);
virtual void SaveQField(const std::string &field_name);
/// Load the collection. Not implemented in the base class DataCollection.
virtual void Load(int cycle_ = 0);
@@ -510,7 +511,9 @@ protected:
int compression_level = -1;
bool high_order_output = false;
bool restart_mode = false;
bool bdr_output = false;
VTKFormat pv_data_format = VTKFormat::BINARY;
public:
ParaViewDataCollectionBase(const std::string &name, Mesh *mesh);
@@ -543,6 +546,10 @@ public:
/// Reading high-order data requires ParaView 5.5 or later.
void SetHighOrderOutput(bool high_order_output_);
/// @brief Configures collection to save only fields evaluated on boundaries of
/// the mesh.
void SetBoundaryOutput(bool bdr_output_);
/// If compression is enabled, return the compression level, else return 0.
int GetCompressionLevel() const;
@@ -564,8 +571,6 @@ public:
///
/// If restart is enabled, new writes will preserve timestep metadata for any
/// solutions prior to the currently defined time.
///
/// Initially, restart mode is disabled.
void UseRestartMode(bool restart_mode_);
};
@@ -575,11 +580,23 @@ class ParaViewDataCollection : public ParaViewDataCollectionBase
private:
std::fstream pvd_stream;
/// A collection of named Coefficients and VectorCoefficients
using CoeffFieldMap = NamedFieldsMap<Coefficient>;
using VCoeffFieldMap = NamedFieldsMap<VectorCoefficient>;
/** A FieldMap mapping registered names to Coefficient and VectorCoefficient
pointers. */
CoeffFieldMap coeff_field_map;
VCoeffFieldMap vcoeff_field_map;
protected:
void WritePVTUHeader(std::ostream &out);
void WritePVTUFooter(std::ostream &out, const std::string &vtu_prefix);
void SaveDataVTU(std::ostream &out, int ref);
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
void SaveCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
Coefficient &coeff);
void SaveVCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
VectorCoefficient& coeff);
const char *GetDataFormatString() const;
const char *GetDataTypeString() const;
@@ -598,6 +615,25 @@ public:
ParaViewDataCollection(const std::string& collection_name,
Mesh *mesh_ = nullptr);
/// Get a const reference to the internal coefficient-field map.
const typename CoeffFieldMap::MapType &GetCoeffFieldMap() const
{ return coeff_field_map.GetMap(); }
const typename VCoeffFieldMap::MapType &GetVCoeffFieldMap() const
{ return vcoeff_field_map.GetMap(); }
/// Add a Coefficient or VectorCoefficient to the collection.
void RegisterCoeffField(const std::string& field_name, Coefficient *coeff)
{ coeff_field_map.Register(field_name, coeff, own_data); }
void RegisterVCoeffField(const std::string& field_name,
VectorCoefficient *vcoeff)
{ vcoeff_field_map.Register(field_name, vcoeff, own_data); }
/// Remove a Coefficient or VectorCoefficient from the collection
void DeregisterCoeffField(const std::string& field_name)
{ coeff_field_map.Deregister(field_name, own_data); }
void DeregisterVCoeffField(const std::string& field_name)
{ vcoeff_field_map.Deregister(field_name, own_data); }
/// Save the collection - the directory name is constructed based on the
/// cycle value
void Save() override;
+403
View File
@@ -0,0 +1,403 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "util.hpp"
namespace mfem::future
{
/// @brief Assemble element matrix for three dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_t3d(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, real_t>& qpdc,
const DeviceTensor<1, real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& q1d,
const int& td1d)
{
constexpr int dimension = 3;
// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, num_qp]
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
// [num_test_dof, ...]
const auto num_test_dof = A.GetShape()[0];
for (int Jx = 0; Jx < td1d; Jx++)
{
for (int Jy = 0; Jy < td1d; Jy++)
{
for (int Jz = 0; Jz < td1d; Jz++)
{
const int J = Jx + td1d * (Jy + td1d * Jz);
for (int j = 0; j < trial_vdim; j++)
{
for (int tv = 0; tv < test_vdim; tv++)
{
for (int tod = 0; tod < test_op_dim; tod++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
fhat(tv, tod, q) = 0.0;
}
}
}
}
}
// MSVC lambda capture workaround
[[maybe_unused]] const auto& inputs_ref = inputs;
int m_offset = 0;
for_constexpr<num_inputs>([&](auto s)
{
using fop_t = std::decay_t<decltype(get<s>(inputs_ref))>;
const int trial_op_dim = static_cast<int>(itod(static_cast<int>(s)));
if (trial_op_dim == 0)
{
// This is inside a lambda so we have to return
// instead of idiomatic 'continue'.
return;
}
auto& B = input_dtqmaps[s].B;
auto& G = input_dtqmaps[s].G;
if constexpr (is_value_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy) * B(qz, 0, Jz);
}
}
}
}
}
}
}
else if constexpr (is_gradient_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const int q = qx + q1d * (qy + q1d * qz);
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
if (m == 0)
{
fhat(i, k, q) += f * G(qx, 0, Jx) * B(qy, 0, Jy) * B(qz, 0, Jz);
}
else if (m == 1)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * G(qy, 0, Jy) * B(qz, 0, Jz);
}
else if (m == 2)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy) * G(qz, 0, Jz);
}
}
}
}
}
}
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("sum factorized sparse matrix assemble routine "
"not implemented for field operator");
#endif
}
MFEM_SYNC_THREAD;
m_offset += trial_op_dim;
});
auto bvtfhat = Reshape(&A(0, 0, J, j), num_test_dof, test_vdim);
map_quadrature_data_to_fields(bvtfhat, fhat, output, output_dtqmap,
scratch_shmem, dimension, true);
}
}
}
}
}
/// @brief Assemble element matrix for two dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_t2d(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, real_t>& qpdc,
const DeviceTensor<1, real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& q1d,
const int& td1d)
{
constexpr int dimension = 2;
// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, num_qp]
const int test_vdim = qpdc.GetShape()[0];
const int test_op_dim = qpdc.GetShape()[1];
const int trial_vdim = qpdc.GetShape()[2];
// [num_test_dof, ...]
const auto num_test_dof = A.GetShape()[0];
for (int Jx = 0; Jx < td1d; Jx++)
{
for (int Jy = 0; Jy < td1d; Jy++)
{
const int J = Jy + Jx * td1d;
for (int j = 0; j < trial_vdim; j++)
{
for (int tv = 0; tv < test_vdim; tv++)
{
for (int tod = 0; tod < test_op_dim; tod++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
fhat(tv, tod, q) = 0.0;
}
}
}
}
// MSVC lambda capture workaround
[[maybe_unused]] const auto& inputs_ref = inputs;
int m_offset = 0;
for_constexpr<num_inputs>([&](auto s)
{
using fop_t = std::decay_t<decltype(get<s>(inputs_ref))>;
const int trial_op_dim = static_cast<int>(itod(static_cast<int>(s)));
if (trial_op_dim == 0)
{
// This is inside a lambda so we have to return
// instead of idiomatic 'continue'.
return;
}
auto& B = input_dtqmaps[s].B;
auto& G = input_dtqmaps[s].G;
if constexpr (is_value_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
fhat(i, k, q) += f * B(qx, 0, Jx) * B(qy, 0, Jy);
}
}
}
}
}
}
else if constexpr (is_gradient_fop<fop_t>::value)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const int q = qy + qx * q1d;
for (int m = 0; m < trial_op_dim; m++)
{
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
const real_t f = qpdc(i, k, j, m + m_offset, q);
if (m == 0)
{
fhat(i, k, q) += f * B(qx, 0, Jx) * G(qy, 0, Jy);
}
else
{
fhat(i, k, q) += f * G(qx, 0, Jx) * B(qy, 0, Jy);
}
}
}
}
}
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("sum factorized sparse matrix assemble routine "
"not implemented for field operator");
#endif
}
MFEM_SYNC_THREAD;
m_offset += trial_op_dim;
});
auto bvtfhat = Reshape(&A(0, 0, J, j), num_test_dof, test_vdim);
map_quadrature_data_to_fields(bvtfhat, fhat, output, output_dtqmap,
scratch_shmem, dimension, true);
}
}
}
}
/// @brief Assemble element matrix for two or three dimensional data.
///
/// Note: In the below layouts, total_trial_op_dim is > 1 if
/// there are more than one inputs dependent on the derivative variable.
///
/// @param A Memory for one element matrix with layout
/// [test_ndof, test_vdim, trial_ndof, trial_vdim].
/// @param fhat Memory to hold the residual computation with layout
/// [test_vdim, test_op_dim, nqp].
/// @param qpdc The quadrature point data cache with data layout
/// [test_vdim, test_op_dim, trial_vdim, total_trial_op_dim, nqp].
/// @param itod Input Trial Operator Dimension array. If the trial
/// operator is not dependent, the dimension is 0 to indicate that.
/// @param inputs The input field operator types.
/// @param output The output field operator types.
/// @param input_dtqmaps The input DofToQuad maps.
/// @param output_dtqmap The output DofToQuad maps.
/// @param scratch_shmem Scratch shared memory for computations.
/// @param dimension The spatial dimension.
/// @param q1d The number of quadrature points in one dimension.
/// @param td1d The number of trial dofs in one dimension.
/// @param use_sum_factorization Indicator if sum factorization is used.
template <typename input_fop_ts, size_t num_inputs, typename output_fop_t>
MFEM_HOST_DEVICE void assemble_element_mat_naive(
const DeviceTensor<4, real_t>& A,
const DeviceTensor<3, real_t>& fhat,
const DeviceTensor<5, real_t>& qpdc,
const DeviceTensor<1, real_t>& itod,
const input_fop_ts& inputs,
const output_fop_t& output,
const std::array<DofToQuadMap, num_inputs>& input_dtqmaps,
const DofToQuadMap& output_dtqmap,
std::array<DeviceTensor<1>, 6>& scratch_shmem,
const int& dimension,
const int& q1d,
const int& td1d,
const bool& use_sum_factorization)
{
if (use_sum_factorization)
{
if (dimension == 2)
{
assemble_element_mat_t2d(A, fhat, qpdc, itod, inputs, output,
input_dtqmaps, output_dtqmap, scratch_shmem, q1d, td1d);
}
else if (dimension == 3)
{
assemble_element_mat_t3d(A, fhat, qpdc, itod, inputs, output,
input_dtqmaps, output_dtqmap, scratch_shmem, q1d, td1d);
}
}
else
{
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
MFEM_ABORT("element matrix assemble not implemented for non tensor "
"product basis");
#endif
}
}
} // namespace mfem::future
+540 -31
View File
@@ -22,6 +22,7 @@
#include "interpolate.hpp"
#include "integrate.hpp"
#include "qfunction_apply.hpp"
#include "assemble.hpp"
namespace mfem::future
{
@@ -34,10 +35,15 @@ using action_t =
using derivative_action_t =
std::function<void(std::vector<Vector> &, const Vector &, Vector &)>;
/// @brief Type alias for a function that assembles the sparse matrix of a
/// @brief Type alias for a function that assembles the SparseMatrix of a
/// derivative operator
using assemble_derivative_sparsematrix_callback_t =
std::function<void(std::vector<Vector> &, SparseMatrix *&)>;
/// @brief Type alias for a function that assembles the HypreParMatrix of a
/// derivative operator
using assemble_derivative_hypreparmatrix_callback_t =
std::function<void(std::vector<Vector> &, HypreParMatrix &)>;
std::function<void(std::vector<Vector> &, HypreParMatrix *&)>;
/// @brief Type alias for a function that applies the appropriate restriction to
/// the solution and parameters
@@ -81,6 +87,8 @@ public:
const std::vector<Vector *> &parameters_l,
const restriction_callback_t &restriction_callback,
const std::function<void(Vector &, Vector &)> &prolongation_transpose,
const std::vector<assemble_derivative_sparsematrix_callback_t>
&assemble_derivative_sparsematrix_callbacks,
const std::vector<assemble_derivative_hypreparmatrix_callback_t>
&assemble_derivative_hypreparmatrix_callbacks) :
Operator(height, width),
@@ -91,6 +99,8 @@ public:
derivative_actions_transpose(derivative_actions_transpose),
transpose_direction(transpose_direction),
prolongation_transpose(prolongation_transpose),
assemble_derivative_sparsematrix_callbacks(
assemble_derivative_sparsematrix_callbacks),
assemble_derivative_hypreparmatrix_callbacks(
assemble_derivative_hypreparmatrix_callbacks)
{
@@ -156,14 +166,29 @@ public:
prolongation_transpose(daction_l, result_t);
};
/// @brief Assemble the derivative operator into a SparseMatrix.
///
/// @param A The SparseMatrix to assemble the derivative operator into. Can
/// be an uninitialized object.
void Assemble(SparseMatrix *&A)
{
MFEM_ASSERT(!assemble_derivative_sparsematrix_callbacks.empty(),
"derivative can't be assembled into a SparseMatrix");
for (const auto &f : assemble_derivative_sparsematrix_callbacks)
{
f(fields_e, A);
}
}
/// @brief Assemble the derivative operator into a HypreParMatrix.
///
/// @param A The HypreParMatrix to assemble the derivative operator into. Can
/// be an uninitialized object.
void Assemble(HypreParMatrix &A)
void Assemble(HypreParMatrix *&A)
{
MFEM_ASSERT(!assemble_derivative_hypreparmatrix_callbacks.empty(),
"derivative can't be assembled into a matrix");
"derivative can't be assembled into a HypreParMatrix");
for (const auto &f : assemble_derivative_hypreparmatrix_callbacks)
{
@@ -196,6 +221,10 @@ private:
std::function<void(Vector &, Vector &)> prolongation_transpose;
/// Callbacks that assemble derivatives into a SparseMatrix.
std::vector<assemble_derivative_sparsematrix_callback_t>
assemble_derivative_sparsematrix_callbacks;
/// Callbacks that assemble derivatives into a HypreParMatrix.
std::vector<assemble_derivative_hypreparmatrix_callback_t>
assemble_derivative_hypreparmatrix_callbacks;
@@ -211,8 +240,8 @@ private:
///
/// The operator is constructed with solution fields that it will act on and
/// parameter fields that define coefficients. Quadrature functions are added by
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates f
/// those functionas and parameters at quadrature points.
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates
/// those functions and parameters at quadrature points.
///
/// Derivatives can be computed by obtaining a DerivativeOperator using
/// GetDerivative().
@@ -280,6 +309,22 @@ public:
}
}
/// @brief Add an integrator to the operator.
/// Called only from AddDomainIntegrator() and AddBoundaryIntegrator().
template <
typename entity_t,
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void AddIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &attributes,
derivative_ids_t derivative_ids);
/// @brief Add a domain integrator to the operator.
///
/// @param qfunc The quadrature function to be added.
@@ -305,6 +350,31 @@ public:
const Array<int> &domain_attributes,
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
/// @brief Add a boundary integrator to the operator.
///
/// @param qfunc The quadrature function to be added.
/// @param inputs Tuple of FieldOperators for the inputs of the quadrature
/// function.
/// @param outputs Tuple of FieldOperators for the outputs of the quadrature
/// function.
/// @param integration_rule IntegrationRule to use with this integrator.
/// @param boundary_attributes Boundary attributes marker array indicating over
/// which attributes this integrator will integrate over.
/// @param derivative_ids Derivatives to be made available for this
/// integrator.
template <
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t = decltype(std::make_index_sequence<0> {})>
void AddBoundaryIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &boundary_attributes,
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
/// @brief Set the parameters for the operator.
///
/// This has to be called before using Mult() or MultTranspose().
@@ -370,6 +440,7 @@ public:
par_l,
restriction_callback,
prolongation_transpose,
assemble_derivative_sparsematrix_callbacks[derivative_id],
assemble_derivative_hypreparmatrix_callbacks[derivative_id]);
}
@@ -383,6 +454,9 @@ private:
std::vector<derivative_action_t>> derivative_action_callbacks;
std::map<size_t,
std::vector<derivative_action_t>> daction_transpose_callbacks;
std::map<size_t,
std::vector<assemble_derivative_sparsematrix_callback_t>>
assemble_derivative_sparsematrix_callbacks;
std::map<size_t,
std::vector<assemble_derivative_hypreparmatrix_callback_t>>
assemble_derivative_hypreparmatrix_callbacks;
@@ -403,6 +477,8 @@ private:
std::function<void(Vector &, Vector &)> output_restriction_transpose;
restriction_callback_t restriction_callback;
std::map<size_t, Vector> derivative_qp_caches;
std::map<size_t, size_t> assembled_vector_sizes;
bool use_tensor_product_structure = true;
@@ -423,7 +499,52 @@ void DifferentiableOperator::AddDomainIntegrator(
const Array<int> &domain_attributes,
derivative_ids_t derivative_ids)
{
using entity_t = Entity::Element;
AddIntegrator<Entity::Element>(
qfunc, inputs, outputs, integration_rule, domain_attributes, derivative_ids);
}
template <
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void DifferentiableOperator::AddBoundaryIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &boundary_attributes,
derivative_ids_t derivative_ids)
{
if (mesh.GetNFbyType(FaceType::Boundary) != mesh.GetNBE())
{
MFEM_ABORT("AddBoundaryIntegrator on meshes with interior boundaries is not supported.");
}
AddIntegrator<Entity::BoundaryElement>(
qfunc, inputs, outputs, integration_rule, boundary_attributes, derivative_ids);
}
template <
typename entity_t,
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void DifferentiableOperator::AddIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &attributes,
derivative_ids_t derivative_ids)
{
if constexpr (!(std::is_same_v<entity_t, Entity::Element> ||
std::is_same_v<entity_t, Entity::BoundaryElement>))
{
static_assert(dfem::always_false<entity_t>,
"entity type not supported in AddIntegrator");
}
static constexpr size_t num_inputs =
tuple_size<decltype(inputs)>::value;
@@ -484,25 +605,44 @@ void DifferentiableOperator::AddDomainIntegrator(
inputs_vdim[i] = get<i>(inputs).vdim;
});
Array<int> elem_attributes;
elem_attributes.SetSize(mesh.GetNE());
for (int i = 0; i < mesh.GetNE(); ++i)
const Array<int> *elem_attributes = nullptr;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
elem_attributes[i] = mesh.GetAttribute(i);
elem_attributes = &mesh.GetElementAttributes();
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
elem_attributes = &mesh.GetBdrFaceAttributes();
}
const auto output_fop = get<0>(outputs);
test_space_field_idx = FindIdx(output_fop.GetFieldId(), fields);
bool use_sum_factorization = false;
auto entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
if ((entity_element_type == Element::QUADRILATERAL ||
entity_element_type == Element::HEXAHEDRON) &&
use_tensor_product_structure == true)
Element::Type entity_element_type;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
use_sum_factorization = true;
entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
if ((entity_element_type == Element::QUADRILATERAL ||
entity_element_type == Element::HEXAHEDRON) &&
use_tensor_product_structure == true)
{
use_sum_factorization = true;
}
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalFaceGeometry());
if ((entity_element_type == Element::SEGMENT ||
entity_element_type == Element::QUADRILATERAL) &&
use_tensor_product_structure == true)
{
use_sum_factorization = true;
}
}
ElementDofOrdering element_dof_ordering = ElementDofOrdering::NATIVE;
@@ -540,8 +680,17 @@ void DifferentiableOperator::AddDomainIntegrator(
prolongation_transpose = get_prolongation_transpose(
fields[test_space_field_idx], output_fop, mesh.GetComm());
const int dimension = mesh.Dimension();
[[maybe_unused]] const int num_elements = GetNumEntities<Entity::Element>(mesh);
int dimension;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
dimension = mesh.Dimension();
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
dimension = mesh.Dimension() - 1;
}
[[maybe_unused]] const int num_elements = GetNumEntities<entity_t>(mesh);
const int num_entities = GetNumEntities<entity_t>(mesh);
const int num_qp = integration_rule.GetNPoints();
@@ -615,6 +764,12 @@ void DifferentiableOperator::AddDomainIntegrator(
thread_blocks.z = 1;
}
}
else if (dimension == 1)
{
thread_blocks.x = q1d;
thread_blocks.y = 1;
thread_blocks.z = 1;
}
action_callbacks.push_back(
// Explicitly capture everything we need, so we can make explicit choice
@@ -630,7 +785,7 @@ void DifferentiableOperator::AddDomainIntegrator(
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
domain_attributes, // Array<int>
attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
@@ -663,13 +818,13 @@ void DifferentiableOperator::AddDomainIntegrator(
action_shmem_info.field_sizes,
num_entities);
const bool has_attr = domain_attributes.Size() > 0;
const auto d_domain_attr = domain_attributes.Read();
const auto d_elem_attr = elem_attributes.Read();
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, input_shmem,
residual_shmem, scratch_shmem] =
@@ -716,7 +871,8 @@ void DifferentiableOperator::AddDomainIntegrator(
// print_shared_memory_info(shmem_info);
Vector direction_e;
Vector direction_e(get_restriction<entity_t>(fields[d_field_idx],
element_dof_ordering)->Height());
Vector derivative_action_e(output_e_size);
derivative_action_e = 0.0;
@@ -728,6 +884,30 @@ void DifferentiableOperator::AddDomainIntegrator(
}
const auto input_is_dependent = it->second;
// Trial operator dimension for each input.
// The trial operator dimension is set for each input that is
// dependent and if it is independent the dimension is 0.
Vector inputs_trial_op_dim(num_inputs);
int total_trial_op_dim = 0;
{
auto itod = Reshape(inputs_trial_op_dim.HostReadWrite(), num_inputs);
int idx = 0;
for_constexpr<num_inputs>([&](auto s)
{
if (!input_is_dependent[s])
{
itod(idx) = 0;
}
else
{
// TODO: BUG! Make this a general function that works for all kinds of inputs.
itod(idx) = input_size_on_qp[s] / get<s>(inputs).vdim;
}
total_trial_op_dim += static_cast<int>(itod(idx));
idx++;
});
}
derivative_action_callbacks[derivative_id].push_back(
[
// capture by copy:
@@ -739,7 +919,7 @@ void DifferentiableOperator::AddDomainIntegrator(
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
domain_attributes, // Array<int>
attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
@@ -777,14 +957,14 @@ void DifferentiableOperator::AddDomainIntegrator(
shmem_info.direction_size,
num_entities);
const auto d_elem_attr = elem_attributes.Read();
const bool has_attr = domain_attributes.Size() > 0;
const auto d_domain_attr = domain_attributes.Read();
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
derivative_action_e = 0.0;
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
{
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
direction_shmem, input_shmem,
@@ -821,6 +1001,335 @@ void DifferentiableOperator::AddDomainIntegrator(
shmem_cache.ReadWrite());
or_transpose(derivative_action_e, der_action_l);
});
// First Input index of the derivative
const size_t d_input_idx = [d_field_idx, &input_to_field]
{
for (size_t i = 0; i < input_to_field.size(); i++)
{
if (input_to_field[i] == d_field_idx)
{
return i;
}
}
return size_t(SIZE_MAX);
}();
const int trial_vdim = GetVDim(fields[d_field_idx]);
const int num_trial_dof =
get_restriction<entity_t>(fields[d_field_idx], element_dof_ordering)->Height() /
inputs_vdim[d_input_idx] / num_entities;
const int num_trial_dof_1d =
input_dtq_maps[d_input_idx].B.GetShape()[DofToQuadMap::Index::DOF];
Vector Ae_mem(num_test_dof * test_vdim * num_trial_dof * trial_vdim *
num_entities);
Ae_mem = 0.0;
// Quadrature point local derivative cache for each element, with data
// layout:
// [test_vdim, test_op_dim, trial_vdim, trial_op_dim, qp, num_entities].
derivative_qp_caches[derivative_id] = Vector(test_vdim * test_op_dim *
trial_vdim *
total_trial_op_dim * num_qp * num_entities);
// Create local references for MSVC lambda capture compatibility
auto& fields_ref = this->fields;
auto& derivative_qp_caches_ref = this->derivative_qp_caches[derivative_id];
assemble_derivative_sparsematrix_callbacks[derivative_id].push_back(
[
// capture by copy:
dimension, // int
num_entities, // int
num_test_dof, // int
num_qp, // int
q1d, // int
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
output_dtq_maps, // std::array<DofToQuadMap, num_fields>
input_to_field, // std::array<int, s>
output_fop, // class derived from FieldOperator
qfunc, // qfunc_t
thread_blocks, // ThreadBlocks
shmem_cache, // Vector (local)
shmem_info, // SharedMemoryInfo
// TODO: make this Array<int> a member of the DifferentiableOperator
// and capture it by ref.
elem_attributes, // Array<int>
input_is_dependent, // std::array<bool, num_inputs>
direction_e, // Vector
da_size_on_qp, // int
total_trial_op_dim,
trial_vdim,
num_trial_dof,
num_trial_dof_1d,
inputs_trial_op_dim,
Ae_mem,
output_to_field,
// capture by ref:
&qpdc_mem = derivative_qp_caches_ref,
&fields = fields_ref
](std::vector<Vector> &f_e, SparseMatrix *&A) mutable
{
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
num_entities);
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(),
shmem_info.direction_size,
num_entities);
auto qpdc = Reshape(qpdc_mem.ReadWrite(), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp, num_entities);
auto itod = Reshape(inputs_trial_op_dim.ReadWrite(), num_inputs);
auto Ae = Reshape(Ae_mem.ReadWrite(), num_test_dof, test_vdim, num_trial_dof,
trial_vdim, num_entities);
const auto d_elem_attr = elem_attributes->Read();
const bool has_attr = attributes.Size() > 0;
const auto d_domain_attr = attributes.Read();
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
{
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
direction_shmem, input_shmem,
shadow_shmem_, residual_shmem,
scratch_shmem] =
unpack_shmem(shmem, shmem_info, input_dtq_maps, output_dtq_maps,
wrapped_fields_e, wrapped_direction_e, num_qp, e);
auto &shadow_shmem = shadow_shmem_;
map_fields_to_quadrature_data(
input_shmem, fields_shmem, input_dtq_shmem, input_to_field,
inputs, ir_weights, scratch_shmem, dimension,
use_sum_factorization);
set_zero(shadow_shmem);
for (int q = 0; q < num_qp; q++)
{
for (int j = 0; j < trial_vdim; j++)
{
int m_offset = 0;
for (size_t s = 0; s < num_inputs; s++)
{
const int trial_op_dim = static_cast<int>(itod(s));
if (trial_op_dim == 0)
{
continue;
}
auto d_qp = Reshape(&(shadow_shmem[s])[0], trial_vdim, trial_op_dim, num_qp);
for (int m = 0; m < trial_op_dim; m++)
{
d_qp(j, m, q) = 1.0;
auto r = Reshape(&residual_shmem(0, q), da_size_on_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
#ifdef MFEM_USE_ENZYME
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
shadow_shmem, q);
#else
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
d_qp(j, m, q) = 0.0;
auto f = Reshape(&r(0), test_vdim, test_op_dim);
for (int i = 0; i < test_vdim; i++)
{
for (int k = 0; k < test_op_dim; k++)
{
qpdc(i, k, j, m + m_offset, q, e) = f(i, k);
}
}
}
m_offset += trial_op_dim;
}
}
}
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
auto Aee = Reshape(&Ae(0, 0, 0, 0, e), num_test_dof, test_vdim, num_trial_dof,
trial_vdim);
auto qpdce = Reshape(&qpdc(0, 0, 0, 0, 0, e), test_vdim, test_op_dim,
trial_vdim, total_trial_op_dim, num_qp);
assemble_element_mat_naive(Aee, fhat, qpdce, itod, inputs, output_fop,
input_dtq_shmem, output_dtq_shmem[0], scratch_shmem, dimension, q1d,
num_trial_dof_1d, use_sum_factorization);
}, num_entities, thread_blocks, shmem_info.total_size,
shmem_cache.ReadWrite());
FieldDescriptor *trial_field = nullptr;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
A = new SparseMatrix(test_fes->GetVSize(), trial_fes->GetVSize());
auto tmp = Reshape(Ae_mem.HostReadWrite(), num_test_dof * test_vdim,
num_trial_dof * trial_vdim, num_entities);
for (int e = 0; e < num_entities; e++)
{
DenseMatrix Aee(&tmp(0, 0, e), num_test_dof * test_vdim,
num_trial_dof * trial_vdim);
Array<int> test_vdofs, trial_vdofs;
test_fes->GetElementVDofs(e, test_vdofs);
trial_fes->GetElementVDofs(e, trial_vdofs);
if (use_sum_factorization)
{
Array<int> test_vdofs_mapped(test_vdofs.Size());
const Array<int> &test_dofmap =
dynamic_cast<const TensorBasisElement&>(*test_fes->GetFE(0)).GetDofMap();
if (test_dofmap.Size() == 0)
{
test_vdofs_mapped = test_vdofs;
}
else
{
MFEM_ASSERT(test_dofmap.Size() == num_test_dof,
"internal error: dof map of the test space does not "
"match previously determined number of test space dofs");
for (int vd = 0; vd < test_vdim; vd++)
{
for (int i = 0; i < num_test_dof; i++)
{
test_vdofs_mapped[i + vd * num_test_dof] =
test_vdofs[test_dofmap[i] + vd * num_test_dof];
}
}
}
Array<int> trial_vdofs_mapped(trial_vdofs.Size());
const Array<int> &trial_dofmap =
dynamic_cast<const TensorBasisElement&>(*trial_fes->GetFE(0)).GetDofMap();
if (trial_dofmap.Size() == 0)
{
trial_vdofs_mapped = trial_vdofs;
}
else
{
MFEM_ASSERT(trial_dofmap.Size() == num_trial_dof,
"internal error: dof map of the test space does not "
"match previously determined number of test space dofs");
for (int vd = 0; vd < trial_vdim; vd++)
{
for (int i = 0; i < num_trial_dof; i++)
{
trial_vdofs_mapped[i + vd * num_trial_dof] =
trial_vdofs[trial_dofmap[i] + vd * num_trial_dof];
}
}
}
A->AddSubMatrix(test_vdofs_mapped, trial_vdofs_mapped, Aee, 1);
}
else
{
A->AddSubMatrix(test_vdofs, trial_vdofs, Aee, 1);
}
}
A->Finalize();
});
// Create local references for MSVC lambda capture compatibility
auto& assemble_derivative_sparsematrix_callbacks_ref =
this->assemble_derivative_sparsematrix_callbacks[derivative_id];
assemble_derivative_hypreparmatrix_callbacks[derivative_id].push_back(
[
input_is_dependent,
input_to_field,
output_to_field,
&spmatcb = assemble_derivative_sparsematrix_callbacks_ref,
&fields = fields_ref
](std::vector<Vector> &f_e, HypreParMatrix *&A) mutable
{
SparseMatrix *spmat = nullptr;
for (const auto &f : spmatcb)
{
f(f_e, spmat);
}
if (spmat == nullptr)
{
MFEM_ABORT("internal error");
}
bool same_test_and_trial = false;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
if (output_to_field[0] == input_to_field[s])
{
same_test_and_trial = true;
break;
}
}
}
FieldDescriptor *trial_field = nullptr;
for (size_t s = 0; s < num_inputs; s++)
{
if (input_is_dependent[s])
{
trial_field = &fields[input_to_field[s]];
}
}
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
(&trial_field->data);
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
(&fields[output_to_field[0]].data);
if (same_test_and_trial)
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
spmat);
A = RAP(&tmp, test_fes->Dof_TrueDof_Matrix());
}
else
{
HypreParMatrix tmp(test_fes->GetComm(),
test_fes->GlobalVSize(),
trial_fes->GlobalVSize(),
test_fes->GetDofOffsets(),
trial_fes->GetDofOffsets(),
spmat);
A = RAP(test_fes->Dof_TrueDof_Matrix(), &tmp,
trial_fes->Dof_TrueDof_Matrix());
}
delete spmat;
});
}, derivative_ids);
}
}
+84 -1
View File
@@ -95,6 +95,85 @@ void map_quadrature_data_to_fields_impl(
}
}
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_tensor_impl_1d(
DeviceTensor<2, real_t> &y,
const DeviceTensor<3, real_t> &f,
const output_t &output,
const DofToQuadMap &dtq,
std::array<DeviceTensor<1>, 6> &scratch_mem)
{
[[maybe_unused]] auto B = dtq.B;
[[maybe_unused]] auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
auto yd = Reshape(&y(0, 0), d1d, vdim);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t acc = 0.0;
for (int qx = 0; qx < q1d; qx++)
{
acc += fqp(vd, 0, qx) * B(qx, 0, dx);
}
yd(dx, vd) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = G.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
auto yd = Reshape(&y(0, 0), d1d, vdim);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t acc = 0.0;
for (int qx = 0; qx < q1d; qx++)
{
acc += fqp(vd, 0, qx) * G(qx, 0, dx);
}
yd(dx, vd) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d);
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d);
for (int sq = 0; sq < output.size_on_qp; sq++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
yqp(sq, qx) = fqp(sq, qx);
}
MFEM_SYNC_THREAD;
}
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
}
}
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_tensor_impl_2d(
@@ -431,7 +510,11 @@ void map_quadrature_data_to_fields(
{
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
map_quadrature_data_to_fields_tensor_impl_1d(y, f, output, dtq, scratch_mem);
}
else if (dimension == 2)
{
map_quadrature_data_to_fields_tensor_impl_2d(y, f, output, dtq, scratch_mem);
}
+113 -8
View File
@@ -338,6 +338,92 @@ void map_field_to_quadrature_data_tensor_product_2d(
}
}
template <typename field_operator_t>
MFEM_HOST_DEVICE inline
void map_field_to_quadrature_data_tensor_product_1d(
DeviceTensor<2> &field_qp,
const DofToQuadMap &dtq,
const DeviceTensor<1> &field_e,
const field_operator_t &input,
const DeviceTensor<1, const real_t> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem)
{
[[maybe_unused]] auto B = dtq.B;
[[maybe_unused]] auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
{
auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e[0], d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t acc = 0.0;
for (int dx = 0; dx < d1d; dx++)
{
acc += B(qx, 0, dx) * field(dx, vd);
}
fqp(vd, qx) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (
is_gradient_fop<std::decay_t<field_operator_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const int dim = input.dim;
const auto field = Reshape(&field_e[0], d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t acc = 0.0;
for (int dx = 0; dx < d1d; dx++)
{
acc += G(qx, 0, dx) * field(dx, vd);
}
fqp(vd, 0, qx) = acc;
}
MFEM_SYNC_THREAD;
}
}
// TODO: Create separate function for clarity
else if constexpr (
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
{
const int num_qp = integration_weights.GetShape()[0];
// TODO: eeek
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
auto w = Reshape(&integration_weights[0], q1d);
auto f = Reshape(&field_qp[0], q1d);
MFEM_FOREACH_THREAD(qx, x, q1d)
{
f(qx) = w(qx);
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_identity_fop<std::decay_t<field_operator_t>>::value)
{
const int q1d = B.GetShape()[0];
auto field = Reshape(&field_e[0], input.size_on_qp, q1d);
field_qp = field;
}
else
{
static_assert(dfem::always_false<std::decay_t<field_operator_t>>,
"can't map field to quadrature data");
}
}
template <typename field_operator_t>
MFEM_HOST_DEVICE
void map_field_to_quadrature_data(
@@ -425,7 +511,7 @@ void map_fields_to_quadrature_data(
std::array<DeviceTensor<2>, num_inputs> &fields_qp,
const std::array<DeviceTensor<1>, num_fields> &fields_e,
const std::array<DofToQuadMap, num_inputs> &dtqmaps,
const std::array<int, num_inputs> &input_to_field,
const std::array<size_t, num_inputs> &input_to_field,
const field_operator_ts &fops,
const DeviceTensor<1, const real_t> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem,
@@ -440,11 +526,18 @@ void map_fields_to_quadrature_data(
for_constexpr<num_inputs>([&](auto i)
{
const DeviceTensor<1> &field_e =
(input_to_field[i] == -1) ? dummy_field_weight : fields_e[input_to_field[i]];
(input_to_field[i] == SIZE_MAX) ? dummy_field_weight :
fields_e[input_to_field[i]];
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
map_field_to_quadrature_data_tensor_product_1d(
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
integration_weights, scratch_mem);
}
else if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
@@ -489,14 +582,20 @@ void map_field_to_quadrature_data_conditional(
{
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
map_field_to_quadrature_data_tensor_product_3d(
map_field_to_quadrature_data_tensor_product_1d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
else if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
else if (dimension == 3)
{
map_field_to_quadrature_data_tensor_product_2d(
map_field_to_quadrature_data_tensor_product_3d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
}
@@ -539,7 +638,7 @@ void map_direction_to_quadrature_data_conditional(
const std::array<DeviceTensor<1>, 6> &scratch_mem,
const std::array<bool, num_inputs> &conditions,
const int &dimension,
const bool &use_sum_factorization = false)
const bool &use_sum_factorization)
{
for_constexpr<num_inputs>([&](auto i)
{
@@ -547,7 +646,13 @@ void map_direction_to_quadrature_data_conditional(
{
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
map_field_to_quadrature_data_tensor_product_1d(
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
integration_weights, scratch_mem);
}
else if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
+31 -4
View File
@@ -44,7 +44,16 @@ void call_qfunction(
{
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
{
auto qf_args = decay_tuple<qf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), rs_qp);
apply_kernel(r, qfunc, qf_args, input_shmem, q);
}
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
@@ -123,7 +132,22 @@ void call_qfunction_derivative_action(
{
if (use_sum_factorization)
{
if (dimension == 2)
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
{
auto r = Reshape(&residual_shmem(0, q), das_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
#ifdef MFEM_USE_ENZYME
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
shadow_shmem, q);
#else
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
}
}
else if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
@@ -164,7 +188,10 @@ void call_qfunction_derivative_action(
}
}
}
MFEM_SYNC_THREAD;
else
{
MFEM_ABORT_KERNEL("unsupported dimension");
}
}
else
{
@@ -180,8 +207,8 @@ void call_qfunction_derivative_action(
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
}
MFEM_SYNC_THREAD;
}
MFEM_SYNC_THREAD;
}
template <typename qfunc_t, typename args_ts, size_t num_args>
+13 -5
View File
@@ -44,7 +44,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * m) + j);
arg(j, i).value = u((i * n) + j);
}
}
}
@@ -94,8 +94,8 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * m) + j);
arg(j, i).gradient = v((i * m) + j);
arg(j, i).value = u((i * n) + j);
arg(j, i).gradient = v((i * n) + j);
}
}
}
@@ -181,6 +181,14 @@ void process_derivative_from_native_dual(
}
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_derivative_from_native_dual(
DeviceTensor<1, T> &r,
const dual<T, T> &x)
{
r(0) = x.gradient;
}
template <typename T0, typename T1>
MFEM_HOST_DEVICE inline
@@ -230,7 +238,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * m) + j);
arg(j, i) = u((i * n) + j);
}
}
}
@@ -330,7 +338,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * m) + j);
arg(j, i) = u((i * n) + j);
}
}
}
+3 -3
View File
@@ -454,7 +454,7 @@ MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
*
* @tparam T the types stored in the tuples x and y
* @tparam i integer sequence used to index the tuples
* @param x tuple of values to be subracted from
* @param x tuple of values to be subtracted from
* @param y tuple of values to subtract from x
*/
template <typename... T, int... i>
@@ -596,7 +596,7 @@ MFEM_HOST_DEVICE constexpr auto div_helper(const real_t a,
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a the constant denomenator
* @param a the constant denominator
* @return the returned tuple ratio
*/
template <typename... T, int... i>
@@ -726,7 +726,7 @@ MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const real_t a)
/**
* @tparam T the types stored in the tuple
* @tparam i a list of indices used to acces each element of the tuple
* @tparam i a list of indices used to access each element of the tuple
* @param out the ostream to write the output to
* @param A the tuple of values
* @brief helper used to implement printing a tuple of values
+106 -35
View File
@@ -20,6 +20,7 @@
#include <vector>
#include <type_traits>
#include <numeric>
#include <iomanip>
#include "../../general/communication.hpp"
#include "../../general/forall.hpp"
@@ -107,25 +108,32 @@ constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg)
indices{});
}
template <typename... input_ts, std::size_t... Is>
auto make_dependency_map_impl(
tuple<input_ts...> inputs,
std::index_sequence<Is...>)
template <std::size_t I, typename Tuple, std::size_t... Is>
std::array<bool, sizeof...(Is)>
make_dependency_array(const Tuple& inputs, std::index_sequence<Is...>)
{
auto make_dependency_array = [&](auto i)
{
return std::array<bool, sizeof...(input_ts)>
{
(get<i>(inputs).GetFieldId() == get<Is>(inputs).GetFieldId())...
};
};
return { (get<I>(inputs).GetFieldId() == get<Is>(inputs).GetFieldId())... };
}
std::unordered_map<int, std::array<bool, sizeof...(input_ts)>> map;
for_constexpr<sizeof...(input_ts)>([&](auto i)
template <typename... input_ts, std::size_t... Is>
auto make_dependency_map_impl(tuple<input_ts...> inputs,
std::index_sequence<Is...>)
{
constexpr std::size_t N = sizeof...(input_ts);
if constexpr (N == 0)
return std::unordered_map<int, std::array<bool, 0>> {};
std::unordered_map<int, std::array<bool, N>> map;
(void)std::initializer_list<int>
{
map[get<i>(inputs).GetFieldId()] =
make_dependency_array(std::integral_constant<std::size_t, i> {});
});
(
map[get<Is>(inputs).GetFieldId()] =
make_dependency_array<Is>(inputs, std::make_index_sequence<N>{}),
0
)...
};
return map;
}
@@ -200,24 +208,45 @@ void print_tuple(const std::tuple<Args...>& t)
/// ..., vmn]]
/// which is compatible with numpy syntax.
///
/// @param m mfem::DenseMatrix to print
/// @param out ostream to print to
/// @param A mfem::DenseMatrix to print
inline
void pretty_print(const mfem::DenseMatrix& m)
void pretty_print(std::ostream &out, const mfem::DenseMatrix &A)
{
out << "[";
for (int i = 0; i < m.NumRows(); i++)
// Determine the max width of any entry in scientific notation
int max_width = 0;
for (int i = 0; i < A.NumRows(); ++i)
{
for (int j = 0; j < m.NumCols(); j++)
for (int j = 0; j < A.NumCols(); ++j)
{
out << m(i, j);
if (j < m.NumCols() - 1)
std::ostringstream oss;
oss << std::scientific << std::setprecision(2) << A(i, j);
max_width = std::max(max_width, static_cast<int>(oss.str().length()));
}
}
out << "[\n";
for (int i = 0; i < A.NumRows(); ++i)
{
out << " [";
for (int j = 0; j < A.NumCols(); ++j)
{
out << std::setw(max_width) << std::scientific << std::setprecision(2) <<
A(i, j);
if (j < A.NumCols() - 1)
{
out << ", ";
}
}
if (i < m.NumRows() - 1)
out << "]";
if (i < A.NumRows() - 1)
{
out << ", ";
out << ",\n";
}
else
{
out << "\n";
}
}
out << "]\n";
@@ -944,7 +973,44 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
}
else
{
static_assert(dfem::always_false<T>, "can't use GetElementRestriction on type");
static_assert(dfem::always_false<T>,
"can't use get_element_restriction on type");
}
return nullptr; // Unreachable, but avoids compiler warning
}, f.data);
}
/// @brief Get the face restriction operator for a field descriptor.
///
/// @param f the field descriptor.
/// @param o the face dof ordering.
/// @param ft the face type
/// @param m indicator if single or double valued
/// @returns the face restriction operator for the field descriptor in
/// specified ordering.
inline
const Operator *get_face_restriction(const FieldDescriptor &f,
ElementDofOrdering o,
FaceType ft,
L2FaceValues m)
{
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
{
using T = std::decay_t<decltype(arg)>;
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
std::is_same_v<T, const ParFiniteElementSpace *>)
{
return arg->GetFaceRestriction(o, ft, m);
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
// ParameterSpace does not support face restrictions
MFEM_ABORT("internal error");
}
else
{
static_assert(dfem::always_false<T>,
"can't use get_face_restriction on type");
}
return nullptr; // Unreachable, but avoids compiler warning
}, f.data);
@@ -965,6 +1031,11 @@ const Operator *get_restriction(const FieldDescriptor &f,
{
return get_element_restriction(f, o);
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
return get_face_restriction(f, o, FaceType::Boundary,
L2FaceValues::SingleValued);
}
MFEM_ABORT("restriction not implemented for Entity");
return nullptr;
}
@@ -974,7 +1045,7 @@ const Operator *get_restriction(const FieldDescriptor &f,
/// @param f the field descriptor.
/// @param o the element dof ordering.
/// @param fop the field operator.
/// @returns a tuple containting a std::function with the transpose
/// @returns a tuple containing a std::function with the transpose
/// restriction callback and it's height.
template <typename entity_t, typename fop_t>
inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
@@ -1362,12 +1433,12 @@ int GetSizeOnQP(const field_operator_t &, const FieldDescriptor &f)
/// @tparam entity_t the entity type (see Entity).
/// @returns an array mapping field operator types to field descriptor indices.
template <typename entity_t, typename field_operator_ts>
std::array<int, tuple_size<field_operator_ts>::value>
std::array<size_t, tuple_size<field_operator_ts>::value>
create_descriptors_to_fields_map(
const std::vector<FieldDescriptor> &fields,
field_operator_ts &fops)
{
std::array<int, tuple_size<field_operator_ts>::value> map;
std::array<size_t, tuple_size<field_operator_ts>::value> map;
auto find_id = [](const std::vector<FieldDescriptor> &fields, std::size_t i)
{
@@ -1379,9 +1450,9 @@ create_descriptors_to_fields_map(
if (it == fields.end())
{
return -1;
return SIZE_MAX;
}
return static_cast<int>(it - fields.begin());
return static_cast<size_t>(it - fields.begin());
};
auto f = [&](auto &fop, auto &map)
@@ -1389,10 +1460,10 @@ create_descriptors_to_fields_map(
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
{
// TODO-bug: stealing dimension from the first field
fop.dim = GetDimension<Entity::Element>(fields[0]);
fop.dim = GetDimension<entity_t>(fields[0]);
fop.vdim = 1;
fop.size_on_qp = 1;
map = -1;
map = SIZE_MAX;
}
else
{
@@ -2178,7 +2249,7 @@ template <
std::array<DofToQuadMap, N> create_dtq_maps_impl(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqs,
const std::array<int, N> &field_map,
const std::array<size_t, N> &field_map,
std::index_sequence<Is...>)
{
auto f = [&](auto fop, std::size_t idx)
@@ -2263,7 +2334,7 @@ template <
std::array<DofToQuadMap, num_fields> create_dtq_maps(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqmaps,
const std::array<int, num_fields> &to_field_map)
const std::array<size_t, num_fields> &to_field_map)
{
return create_dtq_maps_impl<entity_t>(
fops, dtqmaps,
+71 -42
View File
@@ -103,6 +103,18 @@ void FiniteElement::CalcPhysCurlShape(ElementTransformation &Trans,
}
}
void FiniteElement::CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const
{
MFEM_ABORT("method is not implemented for this class");
}
void FiniteElement::CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const
{
MFEM_ABORT("method is not implemented for this class");
}
void FiniteElement::GetFaceDofs(int face, int **dofs, int *ndofs) const
{
MFEM_ABORT("method is not overloaded");
@@ -661,65 +673,78 @@ void ScalarFiniteElement::ScalarLocalL2Restriction(
void NodalFiniteElement::CreateLexicographicFullMap(const IntegrationRule &ir)
const
{
// Get the FULL version of the map.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
for (int d = 0; d < b_dim; d++)
// Get the FULL version of the map.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
{
for (int j = 0; j < dof; j++)
for (int d = 0; d < b_dim; d++)
{
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
for (int j = 0; j < dof; j++)
{
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
}
}
}
}
const int g_dim = [this]()
{
switch (deriv_type)
const int g_dim = [this]()
{
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
switch (deriv_type)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
}
}
}
}
dof2quad_array.Append(d2q_new);
dof2quad_array.Append(d2q_new);
}
}
const DofToQuad &NodalFiniteElement::GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const
{
//Should make this loop a function of FiniteElement
for (int i = 0; i < dof2quad_array.Size(); i++)
DofToQuad *d2q = nullptr;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
{
const DofToQuad &d2q = *dof2quad_array[i];
if (d2q.IntRule == &ir && d2q.mode == mode) { return d2q; }
//Should make this loop a function of FiniteElement
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule == &ir && d2q->mode == mode) { break; }
d2q = nullptr;
}
}
if (d2q) { return *d2q; }
if (mode != DofToQuad::LEXICOGRAPHIC_FULL)
{
return FiniteElement::GetDofToQuad(ir, mode);
@@ -2620,8 +2645,12 @@ const DofToQuad &TensorBasisElement::GetTensorDofToQuad(
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
auto* d2q_ = dof2quad_array[i];
if (d2q_->IntRule == &ir && d2q_->mode == mode)
{
d2q = d2q_;
break;
}
}
if (!d2q)
{
+21
View File
@@ -259,6 +259,7 @@ protected:
IntegrationRule Nodes;
#ifndef MFEM_THREAD_SAFE
mutable DenseMatrix vshape; // Dof x Dim
mutable DenseTensor tshape; // Dof x Dim x Dim
#endif
/// Container for all DofToQuad objects created by the FiniteElement.
/** Multiple DofToQuad objects may be needed when different quadrature rules
@@ -481,6 +482,26 @@ public:
virtual void CalcPhysCurlShape(ElementTransformation &Trans,
DenseMatrix &curl_shape) const;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in reference space at the given point @a ip. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #dim), must be set in advance. */
virtual void CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in physical space at the point described by @a Trans. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #vdim), must be set in advance,
where #vdim >= #dim is the physical space dimension as described
by @a Trans. */
virtual void CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const;
/** @brief Get the dofs associated with the given @a face.
@a *dofs is set to an internal array of the local dofc on the
face, while *ndofs is set to the number of dofs on that face.
+334 -74
View File
@@ -470,24 +470,6 @@ void NURBS_HDiv2DFiniteElement::CalcVShape(const IntegrationPoint &ip,
}
}
void NURBS_HDiv2DFiniteElement::CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const
{
CalcVShape(Trans.GetIntPoint(), shape);
const DenseMatrix & J = Trans.Jacobian();
MFEM_ASSERT(J.Width() == 2 && J.Height() == 2,
"NURBS_HDiv2DFiniteElement cannot be embedded in "
"3 dimensional spaces");
for (int i=0; i<dof; i++)
{
real_t sx = shape(i, 0);
real_t sy = shape(i, 1);
shape(i, 0) = sx * J(0, 0) + sy * J(0, 1);
shape(i, 1) = sx * J(1, 0) + sy * J(1, 1);
}
shape *= (1.0 / Trans.Weight());
}
void NURBS_HDiv2DFiniteElement::CalcDivShape(const IntegrationPoint &ip,
Vector &divshape) const
{
@@ -517,6 +499,72 @@ void NURBS_HDiv2DFiniteElement::CalcDivShape(const IntegrationPoint &ip,
}
}
void NURBS_HDiv2DFiniteElement::CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const
{
dvshape = 0.0;
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv1[0]->CalcShape ( shape1_x, ijk[0], ip.x);
kv1[1]->CalcShape ( shape1_y, ijk[1], ip.y);
kv1[0]->CalcDShape(dshape1_x, ijk[0], ip.x);
kv1[1]->CalcDShape(dshape1_y, ijk[1], ip.y);
int o = 0;
for (int j = 0; j <= orders[1]; j++)
{
const real_t sy = shape_y(j);
const real_t dsy = dshape_y(j);
for (int i = 0; i <= orders[0]+1; i++, o++)
{
dvshape(o,0,0) = dshape1_x(i)*sy;
dvshape(o,0,1) = shape1_x(i)*dsy;
}
}
for (int j = 0; j <= orders[1]+1; j++)
{
const real_t sy1 = shape1_y(j);
const real_t dsy1 = dshape1_y(j);
for (int i = 0; i <= orders[0]; i++, o++)
{
dvshape(o,1,0) = dshape_x(i)*sy1;
dvshape(o,1,1) = shape_x(i)*dsy1;
}
}
}
void NURBS_HDiv2DFiniteElement::CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const
{
#ifdef MFEM_THREAD_SAFE
DenseTensor tshape(dof, dim, dim);
#else
tshape.SetSize(dof, dim, dim);
#endif
CalcDVShape(Trans.GetIntPoint(), tshape);
const DenseMatrix & J = Trans.Jacobian();
DenseMatrix Jinv = Trans.InverseJacobian();
Jinv *= (1.0 / Trans.Weight());
dvshape = 0.0;
for (int i=0; i<2; i++)
for (int j=0; j<2; j++)
for (int k=0; k<2; k++)
for (int l=0; l<2; l++)
{
real_t JijJIlk = J(i,j)*Jinv(l,k);
for (int d=0; d<dof; d++)
{
dvshape(d,i,k) += tshape(d,j,l)*JijJIlk ;
}
}
}
NURBS_HDiv2DFiniteElement::~NURBS_HDiv2DFiniteElement()
{
if (kv1[0]) { delete kv1[0]; }
@@ -624,26 +672,6 @@ void NURBS_HDiv3DFiniteElement::CalcVShape(const IntegrationPoint &ip,
}
}
void NURBS_HDiv3DFiniteElement::CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const
{
CalcVShape(Trans.GetIntPoint(), shape);
const DenseMatrix & J = Trans.Jacobian();
MFEM_ASSERT(J.Width() == 3 && J.Height() == 3,
"RT_R2D_FiniteElement cannot be embedded in "
"3 dimensional spaces");
for (int i=0; i<dof; i++)
{
real_t sx = shape(i, 0);
real_t sy = shape(i, 1);
real_t sz = shape(i, 2);
shape(i, 0) = sx * J(0, 0) + sy * J(0, 1) + sz * J(0, 2);
shape(i, 1) = sx * J(1, 0) + sy * J(1, 1) + sz * J(1, 2);
shape(i, 2) = sx * J(2, 0) + sy * J(2, 1) + sz * J(2, 2);
}
shape *= (1.0 / Trans.Weight());
}
void NURBS_HDiv3DFiniteElement::CalcDivShape(const IntegrationPoint &ip,
Vector &divshape) const
{
@@ -696,6 +724,112 @@ void NURBS_HDiv3DFiniteElement::CalcDivShape(const IntegrationPoint &ip,
}
}
void NURBS_HDiv3DFiniteElement::CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const
{
dvshape = 0.0;
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv[2]->CalcShape ( shape_z, ijk[2], ip.z);
kv1[0]->CalcShape(shape1_x, ijk[0], ip.x);
kv1[1]->CalcShape(shape1_y, ijk[1], ip.y);
kv1[2]->CalcShape(shape1_z, ijk[2], ip.z);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv[2]->CalcDShape(dshape_z, ijk[2], ip.z);
kv1[0]->CalcDShape(dshape1_x, ijk[0], ip.x);
kv1[1]->CalcDShape(dshape1_y, ijk[1], ip.y);
kv1[2]->CalcDShape(dshape1_z, ijk[2], ip.z);
int o = 0;
for (int k = 0; k <= orders[2]; k++)
{
const real_t sz = shape_z(k);
const real_t dsz = dshape_z(k);
for (int j = 0; j <= orders[1]; j++)
{
const real_t sy_sz = shape_y(j)*sz;
const real_t dsy_sz = dshape_y(j)*sz;
const real_t sy_dsz = shape_y(j)*dsz;
for (int i = 0; i <= orders[0]+1; i++, o++)
{
dvshape(o,0,0) = dshape1_x(i)*sy_sz;
dvshape(o,0,1) = shape1_x(i)*dsy_sz;
dvshape(o,0,2) = shape1_x(i)*sy_dsz;
}
}
}
for (int k = 0; k <= orders[2]; k++)
{
const real_t sz = shape_z(k);
const real_t dsz = dshape_z(k);
for (int j = 0; j <= orders[1]+1; j++)
{
const real_t sy1_sz = shape1_y(j)*sz;
const real_t dsy1_sz = dshape1_y(j)*sz;
const real_t sy1_dsz = shape1_y(j)*dsz;
for (int i = 0; i <= orders[0]; i++, o++)
{
dvshape(o,1,0) = dshape_x(i)*sy1_sz;
dvshape(o,1,1) = shape_x(i)*dsy1_sz;
dvshape(o,1,2) = shape_x(i)*sy1_dsz;
}
}
}
for (int k = 0; k <= orders[2]+1; k++)
{
const real_t sz1 = shape1_z(k);
const real_t dsz1 = dshape1_z(k);
for (int j = 0; j <= orders[1]; j++)
{
const real_t sy_sz1 = shape_y(j)*sz1;
const real_t dsy_sz1 = dshape_y(j)*sz1;
const real_t sy_dsz1 = shape_y(j)*dsz1;
for (int i = 0; i <= orders[0]; i++, o++)
{
dvshape(o,2,0) = dshape_x(i)*sy_sz1;
dvshape(o,2,1) = shape_x(i)*dsy_sz1;
dvshape(o,2,2) = shape_x(i)*sy_dsz1;
}
}
}
}
void NURBS_HDiv3DFiniteElement::CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const
{
#ifdef MFEM_THREAD_SAFE
DenseTensor tshape(dof, dim, dim);
#else
tshape.SetSize(dof, dim, dim);
#endif
CalcDVShape(Trans.GetIntPoint(), tshape);
const DenseMatrix & J = Trans.Jacobian();
DenseMatrix Jinv = Trans.InverseJacobian();
Jinv *= (1.0 / Trans.Weight());
dvshape = 0.0;
for (int i=0; i<3; i++)
for (int j=0; j<3; j++)
for (int k=0; k<3; k++)
for (int l=0; l<3; l++)
{
real_t JijJIlk = J(i,j)*Jinv(l,k);
for (int d=0; d<dof; d++)
{
dvshape(d,i,k) += tshape(d,j,l)*JijJIlk ;
}
}
}
NURBS_HDiv3DFiniteElement::~NURBS_HDiv3DFiniteElement()
{
if (kv1[0]) { delete kv1[0]; }
@@ -771,23 +905,6 @@ void NURBS_HCurl2DFiniteElement::CalcVShape(const IntegrationPoint &ip,
}
}
void NURBS_HCurl2DFiniteElement::CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const
{
CalcVShape(Trans.GetIntPoint(), shape);
const DenseMatrix & JI = Trans.InverseJacobian();
MFEM_ASSERT(JI.Width() == 2 && JI.Height() == 2,
"NURBS_HCurl2DFiniteElement cannot be embedded in "
"3 dimensional spaces");
for (int i=0; i<dof; i++)
{
real_t sx = shape(i, 0);
real_t sy = shape(i, 1);
shape(i, 0) = sx * JI(0, 0) + sy * JI(1, 0);
shape(i, 1) = sx * JI(0, 1) + sy * JI(1, 1);
}
}
void NURBS_HCurl2DFiniteElement::CalcCurlShape(const IntegrationPoint &ip,
DenseMatrix &curl_shape) const
{
@@ -817,6 +934,69 @@ void NURBS_HCurl2DFiniteElement::CalcCurlShape(const IntegrationPoint &ip,
}
}
void NURBS_HCurl2DFiniteElement::CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const
{
dvshape = 0.0;
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv1[0]->CalcShape(shape1_x, ijk[0], ip.x);
kv1[1]->CalcShape(shape1_y, ijk[1], ip.y);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv1[0]->CalcDShape(dshape1_x, ijk[0], ip.x);
kv1[1]->CalcDShape(dshape1_y, ijk[1], ip.y);
int o = 0;
for (int j = 0; j <= orders[1]+1; j++)
{
const real_t sy1 = shape1_y(j);
const real_t dsy1 = dshape1_y(j);
for (int i = 0; i <= orders[0]; i++, o++)
{
dvshape(o,0,0) = dshape_x(i)*sy1;
dvshape(o,0,1) = shape_x(i)*dsy1;
}
}
for (int j = 0; j <= orders[1]; j++)
{
const real_t sy = shape_y(j);
const real_t dsy = dshape_y(j);
for (int i = 0; i <= orders[0]+1; i++, o++)
{
dvshape(o,1,0) = dshape1_x(i)*sy;
dvshape(o,1,1) = shape1_x(i)*dsy;
}
}
}
void NURBS_HCurl2DFiniteElement::CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const
{
MFEM_ASSERT(map_type == H_CURL, "");
DenseTensor tshape(dof, dim, dim);
CalcDVShape(Trans.GetIntPoint(), tshape);
const DenseMatrix &JI = Trans.InverseJacobian();
dvshape = 0.0;
for (int i=0; i<2; i++)
for (int j=0; j<2; j++)
for (int k=0; k<2; k++)
for (int l=0; l<2; l++)
{
real_t JijJIlk = JI(j,i)*JI(l,k);
for (int d=0; d<dof; d++)
{
dvshape(d,i,k) += tshape(d,j,l)*JijJIlk ;
}
}
}
NURBS_HCurl2DFiniteElement::~NURBS_HCurl2DFiniteElement()
{
if (kv1[0]) { delete kv1[0]; }
@@ -924,25 +1104,6 @@ void NURBS_HCurl3DFiniteElement::CalcVShape(const IntegrationPoint &ip,
}
}
void NURBS_HCurl3DFiniteElement::CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const
{
CalcVShape(Trans.GetIntPoint(), shape);
const DenseMatrix & JI = Trans.InverseJacobian();
MFEM_ASSERT(JI.Width() == 3 && JI.Height() == 3,
"NURBS_HCurl3DFiniteElement must be in a"
"3 dimensional spaces");
for (int i=0; i<dof; i++)
{
real_t sx = shape(i, 0);
real_t sy = shape(i, 1);
real_t sz = shape(i, 2);
shape(i, 0) = sx * JI(0, 0) + sy * JI(1, 0) + sz * JI(2, 0);
shape(i, 1) = sx * JI(0, 1) + sy * JI(1, 1) + sz * JI(2, 1);
shape(i, 2) = sx * JI(0, 2) + sy * JI(1, 2) + sz * JI(2, 2);
}
}
void NURBS_HCurl3DFiniteElement::CalcCurlShape(const IntegrationPoint &ip,
DenseMatrix &curl_shape) const
{
@@ -1008,6 +1169,105 @@ void NURBS_HCurl3DFiniteElement::CalcCurlShape(const IntegrationPoint &ip,
}
}
void NURBS_HCurl3DFiniteElement::CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const
{
dvshape = 0.0;
kv[0]->CalcShape ( shape_x, ijk[0], ip.x);
kv[1]->CalcShape ( shape_y, ijk[1], ip.y);
kv[2]->CalcShape ( shape_z, ijk[2], ip.z);
kv1[0]->CalcShape(shape1_x, ijk[0], ip.x);
kv1[1]->CalcShape(shape1_y, ijk[1], ip.y);
kv1[2]->CalcShape(shape1_z, ijk[2], ip.z);
kv[0]->CalcDShape(dshape_x, ijk[0], ip.x);
kv[1]->CalcDShape(dshape_y, ijk[1], ip.y);
kv[2]->CalcDShape(dshape_z, ijk[2], ip.z);
kv1[0]->CalcDShape(dshape1_x, ijk[0], ip.x);
kv1[1]->CalcDShape(dshape1_y, ijk[1], ip.y);
kv1[2]->CalcDShape(dshape1_z, ijk[2], ip.z);
int o = 0;
for (int k = 0; k <= orders[2]+1; k++)
{
const real_t sz1 = shape1_z(k);
const real_t dsz1 = dshape1_z(k);
for (int j = 0; j <= orders[1]+1; j++)
{
const real_t sy1_sz1 = shape1_y(j)*sz1;
const real_t dsy1_sz1 = dshape1_y(j)*sz1;
const real_t sy1_dsz1 = shape1_y(j)*dsz1;
for (int i = 0; i <= orders[0]; i++, o++)
{
dvshape(o,0,0) = dshape_x(i)*sy1_sz1;
dvshape(o,0,1) = shape_x(i)*dsy1_sz1;
dvshape(o,0,2) = shape_x(i)*sy1_dsz1;
}
}
}
for (int k = 0; k <= orders[2]+1; k++)
{
const real_t sz1 = shape1_z(k);
const real_t dsz1 = dshape1_z(k);
for (int j = 0; j <= orders[1]; j++)
{
const real_t sy_sz1 = shape_y(j)*sz1;
const real_t dsy_sz1 = dshape_y(j)*sz1;
const real_t sy_dsz1 = shape_y(j)*dsz1;
for (int i = 0; i <= orders[0]+1; i++, o++)
{
dvshape(o,1,0) = dshape1_x(i)*sy_sz1;
dvshape(o,1,1) = shape1_x(i)*dsy_sz1;
dvshape(o,1,2) = shape1_x(i)*sy_dsz1;
}
}
}
for (int k = 0; k <= orders[2]; k++)
{
const real_t sz = shape_z(k);
const real_t dsz = dshape_z(k);
for (int j = 0; j <= orders[1]+1; j++)
{
const real_t sy1_sz = shape1_y(j)*sz;
const real_t dsy1_sz = dshape1_y(j)*sz;
const real_t sy1_dsz = shape1_y(j)*dsz;
for (int i = 0; i <= orders[0]+1; i++, o++)
{
dvshape(o,2,0) = dshape1_x(i)*sy1_sz;
dvshape(o,2,1) = shape1_x(i)*dsy1_sz;
dvshape(o,2,2) = shape1_x(i)*sy1_dsz;
}
}
}
}
void NURBS_HCurl3DFiniteElement::CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const
{
MFEM_ASSERT(map_type == H_CURL, "");
DenseTensor tshape(dof, dim, dim);
CalcDVShape(Trans.GetIntPoint(), tshape);
const DenseMatrix &JI = Trans.InverseJacobian();
dvshape = 0.0;
for (int i=0; i<3; i++)
for (int j=0; j<3; j++)
for (int k=0; k<3; k++)
for (int l=0; l<3; l++)
{
real_t JijJIlk = JI(j,i)*JI(l,k);
for (int d=0; d<dof; d++)
{
dvshape(d,i,k) += tshape(d,j,l)*JijJIlk ;
}
}
}
NURBS_HCurl3DFiniteElement::~NURBS_HCurl3DFiniteElement()
{
if (kv1[0]) { delete kv1[0]; }
+88 -6
View File
@@ -233,8 +233,8 @@ public:
in advance, where SDim >= #dim is the physical space dimension as
described by @a Trans. */
void CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const override;
DenseMatrix &shape) const override
{ CalcVShape_RT(Trans, shape); }
/** @brief Evaluate the divergence of all shape functions of a *vector*
finite element in reference space at the given point @a ip. */
/** The size (#dof) of the result Vector @a divshape must be set in advance.
@@ -242,6 +242,26 @@ public:
void CalcDivShape(const IntegrationPoint &ip,
Vector &divshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in reference space at the given point @a ip. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #dim), must be set in advance. */
void CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in physical space at the point described by @a Trans. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #vdim), must be set in advance,
where #vdim >= #dim is the physical space dimension as described
by @a Trans. */
void CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const override;
~NURBS_HDiv2DFiniteElement();
};
@@ -327,8 +347,8 @@ public:
in advance, where SDim >= #dim is the physical space dimension as
described by @a Trans. */
void CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const override;
DenseMatrix &shape) const override
{ CalcVShape_RT(Trans, shape); }
/** @brief Evaluate the divergence of all shape functions of a *vector*
finite element in reference space at the given point @a ip. */
/** The size (#dof) of the result Vector @a divshape must be set in advance.
@@ -336,6 +356,26 @@ public:
void CalcDivShape(const IntegrationPoint &ip,
Vector &divshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in reference space at the given point @a ip. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #dim), must be set in advance. */
void CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in physical space at the point described by @a Trans. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #vdim), must be set in advance,
where #vdim >= #dim is the physical space dimension as described
by @a Trans. */
void CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const override;
~NURBS_HDiv3DFiniteElement();
};
@@ -404,7 +444,8 @@ public:
in advance, where SDim >= #dim is the physical space dimension as
described by @a Trans. */
void CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const override;
DenseMatrix &shape) const override
{ CalcVShape_ND(Trans, shape); }
/** @brief Evaluate the curl of all shape functions of a *vector* finite
element in reference space at the given point @a ip. */
@@ -415,6 +456,26 @@ public:
void CalcCurlShape(const IntegrationPoint &ip,
DenseMatrix &curl_shape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in reference space at the given point @a ip. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #dim), must be set in advance. */
void CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in physical space at the point described by @a Trans. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #vdim), must be set in advance,
where #vdim >= #dim is the physical space dimension as described
by @a Trans. */
void CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const override;
~NURBS_HCurl2DFiniteElement();
};
@@ -495,7 +556,8 @@ public:
in advance, where SDim >= #dim is the physical space dimension as
described by @a Trans. */
void CalcVShape(ElementTransformation &Trans,
DenseMatrix &shape) const override;
DenseMatrix &shape) const override
{ CalcVShape_ND(Trans, shape); }
/** @brief Evaluate the curl of all shape functions of a *vector* finite
element in reference space at the given point @a ip. */
@@ -506,6 +568,26 @@ public:
void CalcCurlShape(const IntegrationPoint &ip,
DenseMatrix &curl_shape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in reference space at the given point @a ip. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #dim), must be set in advance. */
void CalcDVShape(const IntegrationPoint &ip,
DenseTensor &dvshape) const override;
/** @brief Evaluate the gradients of all shape functions of a vector finite
element in physical space at the point described by @a Trans. */
/** The 1st index of DenseTensor @a dvshape refers to a vector shapefunction,
the 2nd index selects the scalar component of this vector, the 3rd
index selects the direction of the derivative. The size of @a dvshape,
which should be (#dof x #dim x #vdim), must be set in advance,
where #vdim >= #dim is the physical space dimension as described
by @a Trans. */
void CalcPhysDVShape(ElementTransformation &Trans,
DenseTensor &dvshape) const override;
~NURBS_HCurl3DFiniteElement();
};
+33 -9
View File
@@ -308,13 +308,25 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
FiniteElement::INTEGRAL,
BasisType::GetType(name[12]));
}
else if (!strncmp(name, "RT_R1D",6))
else if (!strncmp(name, "RT_R1D_", 7))
{
fec = new RT_R1D_FECollection(atoi(name+11),atoi(name + 7));
fec = new RT_R1D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "RT_R2D",6))
else if (!strncmp(name, "RT_R1D@", 7))
{
fec = new RT_R2D_FECollection(atoi(name+11),atoi(name + 7));
fec = new RT_R1D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "RT_R2D_", 7))
{
fec = new RT_R2D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "RT_R2D@", 7))
{
fec = new RT_R2D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "RT_", 3))
{
@@ -336,13 +348,25 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
BasisType::GetType(name[9]),
BasisType::GetType(name[10]));
}
else if (!strncmp(name, "ND_R1D",6))
else if (!strncmp(name, "ND_R1D_", 7))
{
fec = new ND_R1D_FECollection(atoi(name+11),atoi(name + 7));
fec = new ND_R1D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "ND_R2D",6))
else if (!strncmp(name, "ND_R1D@", 7))
{
fec = new ND_R2D_FECollection(atoi(name+11),atoi(name + 7));
fec = new ND_R1D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "ND_R2D_", 7))
{
fec = new ND_R2D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "ND_R2D@", 7))
{
fec = new ND_R2D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "ND_", 3))
{
@@ -485,7 +509,7 @@ GetFace(int &nv, v_t &v, int &ne, e_t &e, eo_t &eo,
int v0 = v[f_consts::Edges[i][0]];
int v1 = v[f_consts::Edges[i][1]];
int eor = 0;
if (v0 > v1) { swap(v0, v1); eor = 1; }
if (v0 > v1) { std::swap(v0, v1); eor = 1; }
for (int j = g_consts::VertToVert::I[v0]; true; j++)
{
MFEM_ASSERT(j < g_consts::VertToVert::I[v0+1],
+32
View File
@@ -120,7 +120,9 @@ public:
| ND_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
| ND_Trace@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
| ND_R1D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
| ND_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
| ND_R2D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
| ND_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
| RT_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | Raviart-Thomas vector elements |
| RT@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | Raviart-Thomas vector elements |
| RT_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
@@ -128,7 +130,9 @@ public:
| RT_Trace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
| RT_ValTrace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | VALUE | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
| RT_R1D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
| RT_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
| RT_R2D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
| RT_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
| L2_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
| L2_T[BTYPE]_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
| L2Int_[DIM]_[ORDER] | L2 | * | 0 | INTEGRAL | Discontinuous L2 elements |
@@ -819,6 +823,20 @@ public:
virtual ~NURBS_HDivFECollection();
};
/** @brief Arbitrary order H(div) U H1 NURBS finite elements.
This class is identical to NURBS_HDivFECollection.
However fespace will behave slightly different for this FECollection,
the boundary dofs will also include the tangential components.
This will allow enforcing essential BC for a diffusion problem,
which requires H1 conformity. */
class NURBS_HDivH1FECollection : public NURBS_HDivFECollection
{
public:
explicit NURBS_HDivH1FECollection(int Order = VariableOrder,
const int vdim = -1)
: NURBS_HDivFECollection(Order, vdim) {};
};
/// Arbitrary order H(curl) NURBS finite elements.
class NURBS_HCurlFECollection : public NURBSFECollection
{
@@ -870,6 +888,20 @@ public:
virtual ~NURBS_HCurlFECollection();
};
/// Arbitrary order H(curl) U H1 NURBS finite elements.
/// This class is identical to NURBS_HCurlFECollection.
/// However fespace will behave slightly different for this FECollection,
/// the boundary dofs will also include the normal component.
/// This will allow enforcing essential BC for a diffusion problem,
/// which requires H1 conformity.
class NURBS_HCurlH1FECollection : public NURBS_HCurlFECollection
{
public:
explicit NURBS_HCurlH1FECollection(int Order = VariableOrder,
const int vdim = -1)
: NURBS_HCurlFECollection(Order, vdim) {};
};
/// Piecewise-(bi/tri)linear continuous finite elements.
class LinearFECollection : public FiniteElementCollection
{
+32 -36
View File
@@ -27,37 +27,6 @@ using namespace std;
namespace mfem
{
template <>
void Ordering::DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = 1; vd < vdim; vd++)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byNODES>(ndofs, vdim, dofs[i], vd);
}
}
}
template <>
void Ordering::DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim,
Array<int> &dofs)
{
// static method
int size = dofs.Size();
dofs.SetSize(size*vdim);
for (int vd = vdim-1; vd >= 0; vd--)
{
for (int i = 0; i < size; i++)
{
dofs[i+size*vd] = Map<byVDIM>(ndofs, vdim, dofs[i], vd);
}
}
}
FiniteElementSpace::FiniteElementSpace()
: mesh(NULL), fec(NULL), vdim(0), ordering(Ordering::byNODES),
@@ -1583,6 +1552,11 @@ const FaceRestriction *FiniteElementSpace::GetFaceRestriction(
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const IntegrationRule &ir) const
{
if (!QuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
for (int i = 0; i < E2Q_array.Size(); i++)
{
const QuadratureInterpolator *qi = E2Q_array[i];
@@ -1597,6 +1571,11 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
const QuadratureSpace &qs) const
{
if (!QuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
for (int i = 0; i < E2Q_array.Size(); i++)
{
const QuadratureInterpolator *qi = E2Q_array[i];
@@ -1612,6 +1591,11 @@ const FaceQuadratureInterpolator
*FiniteElementSpace::GetFaceQuadratureInterpolator(
const IntegrationRule &ir, FaceType type) const
{
if (!FaceQuadratureInterpolator::SupportsFESpace(*this))
{
return nullptr;
}
if (type==FaceType::Interior)
{
for (int i = 0; i < E2IFQ_array.Size(); i++)
@@ -2642,19 +2626,21 @@ void FiniteElementSpace::UpdateNURBS()
if (dynamic_cast<const NURBS_HDivFECollection *>(fec))
{
bool H1 = dynamic_cast<const NURBS_HDivH1FECollection *>(fec);
VNURBSext.SetSize(mesh->Dimension());
for (int d = 0; d < mesh->Dimension(); d++)
{
VNURBSext[d] = NURBSext->GetDivExtension(d);
VNURBSext[d] = NURBSext->GetDivExtension(d, H1);
}
}
if (dynamic_cast<const NURBS_HCurlFECollection *>(fec))
{
bool H1 = dynamic_cast<const NURBS_HCurlH1FECollection *>(fec);
VNURBSext.SetSize(mesh->Dimension());
for (int d = 0; d < mesh->Dimension(); d++)
{
VNURBSext[d] = NURBSext->GetCurlExtension(d);
VNURBSext[d] = NURBSext->GetCurlExtension(d,H1);
}
}
@@ -2720,8 +2706,9 @@ void FiniteElementSpace::BuildNURBSFaceToDofTable() const
{
int b = face_to_be[f];
if (b == -1) { continue; }
// FIXME: this assumes that the boundary element and the face element have
// the same orientation.
// Check if the boundary element and the face element have
// the same vertices.
if (dim > 1)
{
const Element *fe = mesh->GetFace(f);
@@ -2731,7 +2718,16 @@ void FiniteElementSpace::BuildNURBSFaceToDofTable() const
const int *bv = be->GetVertices();
for (int i = 0; i < nv; i++)
{
MFEM_VERIFY(fv[i] == bv[i],
bool found = false;
for (int j = 0; j < nv; j++)
{
if (fv[i] == bv[j])
{
found = true;
break;
}
}
MFEM_VERIFY(found,
"non-matching face and boundary elements detected!");
}
}
+13 -40
View File
@@ -13,6 +13,7 @@
#define MFEM_FESPACE
#include "../config/config.hpp"
#include "../linalg/ordering.hpp"
#include "../linalg/sparsemat.hpp"
#include "../mesh/mesh.hpp"
#include "fe_coll.hpp"
@@ -24,29 +25,6 @@
namespace mfem
{
/** @brief The ordering method used when the number of unknowns per mesh node
(vector dimension) is bigger than 1. */
class Ordering
{
public:
/// %Ordering methods:
enum Type
{
byNODES, /**< loop first over the nodes (inner loop) then over the vector
dimension (outer loop); symbolically it can be represented
as: XXX...,YYY...,ZZZ... */
byVDIM /**< loop first over the vector dimension (inner loop) then over
the nodes (outer loop); symbolically it can be represented
as: XYZ,XYZ,XYZ,... */
};
template <Type Ord>
static inline int Map(int ndofs, int vdim, int dof, int vd);
template <Type Ord>
static void DofsToVDofs(int ndofs, int vdim, Array<int> &dofs);
};
/// @brief Type describing possible layouts for Q-vectors.
/// @sa QuadratureInterpolator and FaceQuadratureInterpolator.
enum class QVectorLayout
@@ -64,20 +42,6 @@ enum class QVectorLayout
byVDIM
};
template <> inline int
Ordering::Map<Ordering::byNODES>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? dof+ndofs*vd : dof-ndofs*vd;
}
template <> inline int
Ordering::Map<Ordering::byVDIM>(int ndofs, int vdim, int dof, int vd)
{
MFEM_ASSERT(dof < ndofs && -1-dof < ndofs && 0 <= vd && vd < vdim, "");
return (dof >= 0) ? vd+vdim*dof : -1-(vd+vdim*(-1-dof));
}
/// Constants describing the possible orderings of the DOFs in one element.
enum class ElementDofOrdering
{
@@ -799,7 +763,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by QuadratureInterpolator, nullptr is
returned. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const IntegrationRule &ir) const;
@@ -815,7 +782,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by QuadratureInterpolator, nullptr is
returned. */
const QuadratureInterpolator *GetQuadratureInterpolator(
const QuadratureSpace &qs) const;
@@ -825,7 +795,10 @@ public:
@note The returned pointer is shared. A good practice, before using it,
is to set all its properties to their expected values, as other parts of
the code may also change them. That is, it's good to call
SetOutputLayout() and DisableTensorProducts() before interpolating. */
SetOutputLayout() and DisableTensorProducts() before interpolating.
@note If the space is not supported by FaceQuadratureInterpolator,
nullptr is returned. */
const FaceQuadratureInterpolator *GetFaceQuadratureInterpolator(
const IntegrationRule &ir, FaceType type) const;
+5 -5
View File
@@ -4610,7 +4610,7 @@ GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
void GridFunction::GetElementBoundsAtControlPoints(const int elem,
const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
const FiniteElement *fe = fes->GetFE(elem);
int fes_dim = fes->GetVDim();
@@ -4658,7 +4658,7 @@ void GridFunction::GetElementBoundsAtControlPoints(const int elem,
void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
Vector lowerC, upperC;
GetElementBoundsAtControlPoints(elem, plb, lowerC, upperC, vdim);
@@ -4681,7 +4681,7 @@ void GridFunction::GetElementBounds(const int elem, const PLBound &plb,
void GridFunction::GetElementBounds(const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim)
const int vdim) const
{
int nel = fes->GetNE();
int fes_dim = fes->GetVDim();
@@ -4704,7 +4704,7 @@ void GridFunction::GetElementBounds(const PLBound &plb,
PLBound GridFunction::GetElementBounds(Vector &lower,
Vector &upper,
const int ref_factor,
const int vdim)
const int vdim) const
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
@@ -4713,7 +4713,7 @@ PLBound GridFunction::GetElementBounds(Vector &lower,
}
PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
const int ref_factor, const int vdim) const
{
int max_order = fes->GetMaxElementOrder();
PLBound plb(fes, ref_factor*(max_order+1));
+5 -5
View File
@@ -1604,7 +1604,7 @@ public:
/// We compute the bounds for each vdim if @a vdim < 1.
/// Note: For most cases, this method/interface will be sufficient.
virtual PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
const int ref_factor=1, const int vdim=-1) const;
/// Computes the \ref PLBound for the gridfunction with number of control
/// points based on @a ref_factor, and returns the bounds for each element
@@ -1614,27 +1614,27 @@ public:
/// PLBound object used to compute the bounds.
/// We compute the bounds for each vdim if @a vdim < 1.
PLBound GetElementBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1);
const int ref_factor=1, const int vdim=-1) const;
/// Compute piecewise linear bounds on the given element at the grid of
/// [plb.ncp x plb.ncp x plb.ncp] control points for each of the vdim
/// components of the gridfunction.
void GetElementBoundsAtControlPoints(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
const int vdim = -1) const;
/// Compute bounds on the grid function for the given element.
/// The bounds are stored in @b lower and @b upper.
void GetElementBounds(const int elem, const PLBound &plb,
Vector &lower, Vector &upper,
const int vdim = -1);
const int vdim = -1) const;
/// Compute bounds on the grid function for all the elements. The bounds
/// are returned in @b lower and @b upper, ordered byVDim:
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
const int vdim=-1);
const int vdim=-1) const;
///@}
/// Destroys grid function.
+5 -5
View File
@@ -201,7 +201,7 @@ inline void SmemPADiffusionDiagonal2D(const int NE,
MFEM_SHARED real_t BG[2][MQ1*MD1];
real_t (*B)[MD1] = (real_t (*)[MD1]) (BG+0);
real_t (*G)[MD1] = (real_t (*)[MD1]) (BG+1);
MFEM_SHARED real_t QD[3][NBZ][MD1][MQ1];
MFEM_SHARED real_t QD[3][NBZ][MQ1][MD1];
real_t (*QD0)[MD1] = (real_t (*)[MD1])(QD[0] + tidz);
real_t (*QD1)[MD1] = (real_t (*)[MD1])(QD[1] + tidz);
real_t (*QD2)[MD1] = (real_t (*)[MD1])(QD[2] + tidz);
@@ -769,8 +769,8 @@ inline void SmemPADiffusionApply2D(const int NE,
u += Gt[dx][qx] * QQ0[qy][qx];
v += Bt[dx][qx] * QQ1[qy][qx];
}
DQ0[qy][dx] = u;
DQ1[qy][dx] = v;
DQ0[dx][qy] = u;
DQ1[dx][qy] = v;
}
}
MFEM_SYNC_THREAD;
@@ -782,8 +782,8 @@ inline void SmemPADiffusionApply2D(const int NE,
real_t v = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += DQ0[qy][dx] * Bt[dy][qy];
v += DQ1[qy][dx] * Gt[dy][qy];
u += DQ0[dx][qy] * Bt[dy][qy];
v += DQ1[dx][qy] * Gt[dy][qy];
}
Y(dx,dy,e) += (u + v);
}
+5 -22
View File
@@ -17,11 +17,9 @@
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
/// \cond DO_NOT_DOCUMENT
namespace mfem
{
namespace internal
/// \cond DO_NOT_DOCUMENT
namespace mfem::internal
{
// PA Diffusion Apply 2D kernel
@@ -333,23 +331,8 @@ PAVectorDiffusionApply3D(const int NE, const Array<real_t> &b,
}
});
}
} // namespace internal
} // namespace mfem::internal
template <int DIM, int VDIM, int T_D1D, int T_Q1D>
VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 2)
{
return internal::PAVectorDiffusionApply2D<T_D1D, T_Q1D, VDIM>;
}
else if constexpr (DIM == 3)
{
return internal::PAVectorDiffusionApply3D;
}
MFEM_ABORT("");
}
} // namespace mfem
/// \endcond DO_NOT_DOCUMENT
#endif
#endif // MFEM_BILININTEG_VECDIFFUSION_KERNELS_HPP
+248 -186
View File
@@ -9,13 +9,14 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../../general/forall.hpp"
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
#include "../../general/forall.hpp"
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "bilininteg_vecdiffusion_kernels.hpp"
#include "./bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
// #include "bilininteg_vecdiffusion_kernels.hpp"
// #include "bilininteg_vecdiffusion_pa.hpp"
namespace mfem
{
@@ -23,7 +24,7 @@ namespace mfem
VectorDiffusionIntegrator::VectorDiffusionIntegrator(const IntegrationRule *ir)
: BilinearFormIntegrator(ir)
{
static Kernels kernels;
// static Kernels kernels;
}
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q)
@@ -67,210 +68,263 @@ VectorDiffusionIntegrator::VectorDiffusionIntegrator(MatrixCoefficient &mq)
vdim = mq.GetVDim();
}
// PA Diffusion Assemble 2D kernel
static void PAVectorDiffusionSetup2D(const int Q1D,
const int NE,
const Array<real_t> &w,
const Vector &j,
const Vector &c,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 3, NE);
const bool const_c = c.Size() == 1;
const auto C = const_c ? Reshape(c.Read(), 1,1) :
Reshape(c.Read(), NQ, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < NQ; ++q)
{
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t c_detJ = W[q] * C1 / ((J11*J22)-(J21*J12));
y(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
y(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
y(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
});
}
// PA Diffusion Assemble 3D kernel
static void PAVectorDiffusionSetup3D(const int Q1D,
const int NE,
const Array<real_t> &w,
const Vector &j,
const Vector &c,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 6, NE);
const bool const_c = c.Size() == 1;
const auto C = const_c ? Reshape(c.Read(), 1,1) :
Reshape(c.Read(), NQ,NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < NQ; ++q)
{
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J31 = J(q,2,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t J32 = J(q,2,1,e);
const real_t J13 = J(q,0,2,e);
const real_t J23 = J(q,1,2,e);
const real_t J33 = J(q,2,2,e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t c_detJ = W[q] * C1 / detJ;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
y(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
y(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
y(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
y(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
y(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
y(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
static void PAVectorDiffusionSetup(const int dim,
const int Q1D,
const int NE,
const Array<real_t> &W,
const Vector &J,
const Vector &C,
Vector &op)
{
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
PAVectorDiffusionSetup2D(Q1D, NE, W, J, C, op);
}
if (dim == 3)
{
PAVectorDiffusionSetup3D(Q1D, NE, W, J, C, op);
}
}
void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
const IntegrationRule *ir
= IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
const auto *ir = IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
if (DeviceCanUseCeed())
{
delete ceedOp;
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
fes.IsVariableOrder();
if (mixed)
{
ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q);
}
else
{
ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q);
}
const bool mixed =
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
if (mixed) { ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q); }
else { ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q); }
return;
}
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
// If vdim is not set, set it to the space dimension
vdim = (vdim == -1) ? fes.GetVDim() : vdim;
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
const MemoryType mt = pa_mt == MemoryType::DEFAULT
? Device::GetDeviceMemoryType()
: pa_mt;
ne = fes.GetNE();
dim = mesh->Dimension();
sdim = mesh->SpaceDimension();
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
const int nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
const int q1d = quad1D;
MFEM_VERIFY(!VQ && !MQ,
"Only scalar coefficient supported for partial assembly for VectorDiffusionIntegrator");
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
CoefficientVector coeff(qs, CoefficientStorage::FULL);
if (Q)
{
coeff.Project(*Q);
}
else if (VQ)
{
coeff.Project(*VQ);
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
}
else if (MQ)
{
coeff.ProjectTranspose(*MQ);
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
}
else { coeff.SetConstant(1.0); }
coeff_vdim = coeff.GetVDim();
const bool scalar_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == vdim;
const bool matrix_coeff = coeff_vdim == vdim * vdim;
MFEM_VERIFY(scalar_coeff + vector_coeff + matrix_coeff == 1, "");
const int pa_size = dim * dim;
pa_data.SetSize(nq * pa_size * vdim * (matrix_coeff ? dim : 1) * ne, mt);
const Array<real_t> &w = ir->GetWeights();
const Vector &j = geom->J;
Vector &d = pa_data;
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PAVectorDiffusionSetup"); }
if (dim == 2 && sdim == 3)
{
constexpr int DIM = 2;
constexpr int SDIM = 3;
const int NQ = quad1D*quad1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, SDIM, DIM, ne);
auto D = Reshape(d.Write(), NQ, SDIM, ne);
MFEM_VERIFY(scalar_coeff, "");
const int nc = vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
const bool const_c = coeff.Size() == 1;
const auto C = const_c ? Reshape(coeff.Read(), 1,1) :
Reshape(coeff.Read(), NQ,ne);
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
for (int q = 0; q < NQ; ++q)
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const real_t wq = W[q];
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J31 = J(q,2,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t J32 = J(q,2,1,e);
const real_t E = J11*J11 + J21*J21 + J31*J31;
const real_t G = J12*J12 + J22*J22 + J32*J32;
const real_t F = J11*J12 + J21*J22 + J31*J32;
const real_t iw = 1.0 / sqrt(E*G - F*F);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t alpha = wq * C1 * iw;
D(q,0,e) = alpha * G; // 1,1
D(q,1,e) = -alpha * F; // 1,2
D(q,2,e) = alpha * E; // 2,2
MFEM_FOREACH_THREAD(qx, x, q1d)
{
for (int i = 0; i < nc; ++i)
{
const real_t wq = W(qx, qy);
const real_t J11 = J(qx, qy, 0, 0, e);
const real_t J21 = J(qx, qy, 1, 0, e);
const real_t J31 = J(qx, qy, 2, 0, e);
const real_t J12 = J(qx, qy, 0, 1, e);
const real_t J22 = J(qx, qy, 1, 1, e);
const real_t J32 = J(qx, qy, 2, 1, e);
const real_t E = J11*J11 + J21*J21 + J31*J31;
const real_t G = J12*J12 + J22*J22 + J32*J32;
const real_t F = J11*J12 + J21*J22 + J31*J32;
const real_t iw = 1.0 / sqrt(E*G - F*F);
const auto C0 = C(0, qx, qy, e);
const real_t alpha = wq * C0 * iw;
D(qx, qy, 0, i, e) = alpha * G; // 1,1
D(qx, qy, 1, i, e) = -alpha * F; // 1,2
D(qx, qy, 2, i, e) = -alpha * F; // 2,1 == 1,2
D(qx, qy, 3, i, e) = alpha * E; // 2,2
}
}
}
});
}
else if (dim == 2 && sdim == 2)
{
const int nc = vdim, cvdim = coeff_vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
auto DE = Reshape(pa_data.Write(), q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, 0, 0, e);
const real_t J21 = J(qx, qy, 1, 0, e);
const real_t J12 = J(qx, qy, 0, 1, e);
const real_t J22 = J(qx, qy, 1, 1, e);
const real_t w_detJ = W(qx, qy) / ((J11*J22)-(J21*J12));
const real_t D0 = w_detJ * (J12*J12 + J22*J22);
const real_t D1 = -w_detJ * (J12*J11 + J22*J21);
const real_t D2 = w_detJ * (J11*J11 + J21*J21);
const int map[4] = {0, 2, 1, 3};
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
{
const auto k = matrix_coeff ? map[i] : (vector_coeff ? i : 0);
const auto Cc = C(k, qx, qy, e);
DE(qx, qy, 0, i, e) = D0 * Cc;
DE(qx, qy, 1, i, e) = D1 * Cc;
DE(qx, qy, 2, i, e) = D1 * Cc;
DE(qx, qy, 3, i, e) = D2 * Cc;
}
}
}
});
}
else if (dim == 3 && sdim == 3)
{
const int nc = vdim, cvdim = coeff_vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
auto DE = Reshape(pa_data.Write(), q1d, q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, qz, 0, 0, e);
const real_t J21 = J(qx, qy, qz, 1, 0, e);
const real_t J31 = J(qx, qy, qz, 2, 0, e);
const real_t J12 = J(qx, qy, qz, 0, 1, e);
const real_t J22 = J(qx, qy, qz, 1, 1, e);
const real_t J32 = J(qx, qy, qz, 2, 1, e);
const real_t J13 = J(qx, qy, qz, 0, 2, e);
const real_t J23 = J(qx, qy, qz, 1, 2, e);
const real_t J33 = J(qx, qy, qz, 2, 2, e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t c_detJ = W(qx, qy, qz) / detJ;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
const real_t D11 = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
const real_t D21 = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
const real_t D31 = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
const real_t D22 = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
const real_t D32 = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
const real_t D33 = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
const int map[9] = {0, 3, 6, 1, 4, 7, 2, 5, 8};
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
{
const auto k = matrix_coeff ? map[i] : vector_coeff ? i : 0;
const auto Ck = C(k, qx, qy, qz, e);
DE(qx, qy, qz, 0, i, e) = D11 * Ck;
DE(qx, qy, qz, 1, i, e) = D21 * Ck;
DE(qx, qy, qz, 2, i, e) = D31 * Ck;
DE(qx, qy, qz, 3, i, e) = D22 * Ck;
DE(qx, qy, qz, 4, i, e) = D32 * Ck;
DE(qx, qy, qz, 5, i, e) = D33 * Ck;
}
}
}
}
});
}
else
{
PAVectorDiffusionSetup(dim, quad1D, ne, w, j, coeff, d);
MFEM_ABORT("Unknown VectorDiffusionIntegrator::AssemblePA kernel for"
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
}
}
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
// Use CEED backend if available
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
// Add the VectorDiffusionAddMultPA specializations
static const auto vector_diffusion_kernel_specializations =
(
// 2D, SDIM = 2
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 3,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 4,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 5,5>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 6,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 7,7>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 8,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 9,9>::Add(),
// 2D, SDIM = 3
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 3,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 4,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 5,5>::Add(),
// 3D, SDIM = 3
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 3,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,5>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 6,7>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 7,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 8,9>::Add(),
true);
MFEM_CONTRACT_VAR(vector_diffusion_kernel_specializations);
ApplyPAKernels::Run(dim, sdim, dofs1D, quad1D,
ne, coeff_vdim, maps->B, maps->G, pa_data, x, y,
sdim, dofs1D, quad1D);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAVectorDiffusionDiagonal2D(const int NE,
const Array<real_t> &b,
@@ -284,12 +338,15 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto G = Reshape(g.Read(), Q1D, D1D);
// note the different shape for D, this is a (symmetric) matrix so we only
// store necessary entries
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
MFEM_VERIFY(d.Size() == Q1D*Q1D*4*2*NE, "");
const auto D = Reshape(d.Read(), Q1D*Q1D, /*3*/4, 2, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, 2, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -310,9 +367,9 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const real_t D0 = D(q,0,e);
const real_t D1 = D(q,1,e);
const real_t D2 = D(q,2,e);
const real_t D0 = D(q,0,0,e);
const real_t D1 = D(q,1,0,e);
const real_t D2 = D(q,3/*2*/,0,e); // size from 3 (symmetric) to 4 (dims x dims)
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
@@ -356,7 +413,8 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
MFEM_VERIFY(Q1D <= max_q1d, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
MFEM_VERIFY(d.Size() == Q1D*Q1D*Q1D*9*3*NE, "");
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 9/*PA_SIZE:dims*dims*/, 3/*VDIM*/, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, 3, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
@@ -384,7 +442,8 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
const real_t O = Q(q,k,e);
// using 6 symmetric values
const real_t O = Q(q,k,0,e);
const real_t Bz = B(qz,dz);
const real_t Gz = G(qz,dz);
const real_t L = i==2 ? Gz : Bz;
@@ -468,12 +527,14 @@ void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
}
else
{
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported.");
PAVectorDiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G,
pa_data, diag);
}
}
/*
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
@@ -514,5 +575,6 @@ VectorDiffusionIntegrator::Kernels::Kernels()
}
/// \endcond DO_NOT_DOCUMENT
*/
} // namespace mfem
+202
View File
@@ -0,0 +1,202 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "../kernels.hpp"
using mfem::kernels::internal::SetMaxOf;
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorDiffusionApply2D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Array<real_t> &g,
const Vector &d,
const Vector &x,
Vector &y,
const int sdim = 0,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 2;
const int SDIM = T_SDIM ? T_SDIM : sdim;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int PA_SIZE = DIM*DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read(), G = g.Read();
const auto DE = Reshape(d.Read(), Q1D, Q1D, PA_SIZE,
SDIM * (matrix_coeff ? SDIM : 1), NE);
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::vd_regs2d_t<3, DIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
for (int i = 0; i < SDIM; i++)
{
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
{
kernels::internal::LoadDofs2d(e, D1D, i, XE, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t gradX = r1[i][0][qy][qx];
const real_t gradY = r1[i][1][qy][qx];
const int k = matrix_coeff ? (j + i * SDIM) : i;
const real_t O11 = DE(qx,qy,0,k,e), O12 = DE(qx,qy,1,k,e);
const real_t O21 = DE(qx,qy,2,k,e), O22 = DE(qx,qy,3,k,e);
r0[i][0][qy][qx] = (O11 * gradX) + (O12 * gradY);
r0[i][1][qy][qx] = (O21 * gradX) + (O22 * gradY);
} // qx
} // qy
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
const int ij = matrix_coeff ? j : i;
kernels::internal::WriteDofs2d(e, D1D, i, ij, r1, YE);
} // j
} // i
});
}
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorDiffusionApply3D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Array<real_t> &g,
const Vector &d,
const Vector &x,
Vector &y,
const int sdim = 0,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 3;
const int SDIM = T_SDIM ? T_SDIM : sdim;
MFEM_VERIFY(SDIM == 3, "SDIM must be 3");
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int PA_SIZE = DIM*DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read(), G = g.Read();
const auto DE = Reshape(d.Read(), Q1D, Q1D, Q1D, PA_SIZE,
SDIM * (matrix_coeff ? SDIM : 1), NE);
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::vd_regs3d_t<3, DIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
for (int i = 0; i < SDIM; i++)
{
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
{
kernels::internal::LoadDofs3d(e, D1D, i, XE, r0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t gradX = r1[i][0][qz][qy][qx];
const real_t gradY = r1[i][1][qz][qy][qx];
const real_t gradZ = r1[i][2][qz][qy][qx];
const int k = matrix_coeff ? (j + i * SDIM) : i;
const real_t O11 = DE(qx,qy,qz,0,k,e), O12 = DE(qx,qy,qz,1,k,e),
O13 = DE(qx,qy,qz,2,k,e);
const real_t O22 = DE(qx,qy,qz,3,k,e), O23 = DE(qx,qy,qz,4,k,e);
const real_t O33 = DE(qx,qy,qz,5,k,e);
r0[i][0][qz][qy][qx] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
r0[i][1][qz][qy][qx] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
r0[i][2][qz][qy][qx] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
} // qx
} // qy
} // qz
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
const int ij = matrix_coeff ? j : i;
kernels::internal::WriteDofs3d(e, D1D, i, ij, r1, YE);
} // j
} // i
});
}
} // namespace internal
template<int DIM, int T_SDIM, int T_D1D, int T_Q1D>
VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
{
if (DIM == 2)
{
return internal::SmemPAVectorDiffusionApply2D<T_SDIM, T_D1D, T_Q1D>;
}
else if (DIM == 3)
{
return internal::SmemPAVectorDiffusionApply3D<T_SDIM, T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int sdim,
int d1d, int q1d)
{
if (dim == 2)
{
return internal::SmemPAVectorDiffusionApply2D;
}
else if (dim == 3)
{
return internal::SmemPAVectorDiffusionApply3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+199 -382
View File
@@ -9,120 +9,218 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../../general/forall.hpp"
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../../general/forall.hpp"
#include "../ceed/integrators/mass/mass.hpp"
#include "./bilininteg_vecmass_pa.hpp" // IWYU pragma: keep
namespace mfem
{
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation *T = mesh->GetTypicalElementTransformation();
const IntegrationRule *ir
= IntRule ? IntRule : &MassIntegrator::GetRule(el, el, *T);
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
const auto *ir = IntRule ? IntRule : &MassIntegrator::GetRule(el, el, Trans);
if (DeviceCanUseCeed())
{
delete ceedOp;
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
fes.IsVariableOrder();
if (mixed)
{
ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q);
}
else
{
ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q);
}
const bool mixed =
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
if (mixed) { ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q); }
else { ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q); }
return;
}
// If vdim is not set, set it to the space dimension
vdim = (vdim == -1) ? Trans.GetSpaceDim() : vdim;
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
MFEM_VERIFY(vdim == mesh->Dimension(), "vdim != dim");
const MemoryType mt = pa_mt == MemoryType::DEFAULT
? Device::GetDeviceMemoryType()
: pa_mt;
ne = mesh->GetNE();
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
const int nq = ir->GetNPoints();
const int sdim = mesh->SpaceDimension();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
real_t coeff = 1.0;
const int q1d = quad1D;
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs);
if (Q)
{
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "Only ConstantCoefficient is supported.");
coeff = cQ->constant;
coeff.Project(*Q);
}
if (!(dim == 2 || dim == 3))
else if (VQ)
{
MFEM_ABORT("Dimension not supported.");
coeff.Project(*VQ);
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
}
else if (MQ)
{
coeff.ProjectTranspose(*MQ);
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
}
else { coeff.SetConstant(1.0); }
coeff_vdim = coeff.GetVDim();
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == vdim;
const bool matrix_coeff = coeff_vdim == vdim * vdim;
MFEM_VERIFY(const_coeff + vector_coeff + matrix_coeff == 1, "");
pa_data.SetSize(coeff_vdim * nq * ne, mt);
const auto w_r = ir->GetWeights().Read();
if (dim == 2)
{
const real_t constant = coeff;
const int NE = ne;
const int NQ = nq;
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const auto W = Reshape(w_r, q1d, q1d);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, coeff_vdim, ne);
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
for (int q = 0; q < NQ; ++q)
MFEM_FOREACH_THREAD(qy, y, q1d)
{
const real_t J11 = J(q,0,0,e);
const real_t J12 = J(q,1,0,e);
const real_t J21 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t detJ = (J11*J22)-(J21*J12);
v(q,e) = w[q] * constant * detJ;
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, 0, 0, e), J12 = J(qx, qy, 1, 0, e);
const real_t J21 = J(qx, qy, 0, 1, e), J22 = J(qx, qy, 1, 1, e);
const real_t detJ = (J11 * J22) - (J21 * J12);
const real_t w_det = W(qx, qy) * detJ;
D(qx, qy, 0, e) = C(0, qx, qy, e) * w_det;
if (const_coeff) { continue; }
D(qx, qy, 1, e) = C(1, qx, qy, e) * w_det;
if (vector_coeff) { continue; }
assert(matrix_coeff);
D(qx, qy, 2, e) = C(2, qx, qy, e) * w_det;
D(qx, qy, 3, e) = C(3, qx, qy, e) * w_det;
}
}
});
}
if (dim == 3)
else if (dim == 3)
{
const real_t constant = coeff;
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const auto W = Reshape(w_r, q1d, q1d, q1d);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, q1d, coeff_vdim, ne);
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
for (int q = 0; q < NQ; ++q)
MFEM_FOREACH_THREAD(qz, z, q1d)
{
const real_t J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const real_t J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const real_t J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
v(q,e) = W[q] * constant * detJ;
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, qz, 0, 0, e),
J12 = J(qx, qy, qz, 0, 1, e),
J13 = J(qx, qy, qz, 0, 2, e);
const real_t J21 = J(qx, qy, qz, 1, 0, e),
J22 = J(qx, qy, qz, 1, 1, e),
J23 = J(qx, qy, qz, 1, 2, e);
const real_t J31 = J(qx, qy, qz, 2, 0, e),
J32 = J(qx, qy, qz, 2, 1, e),
J33 = J(qx, qy, qz, 2, 2, e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t w_det = W(qx, qy, qz) * detJ;
D(qx, qy, qz, 0, e) = C(0, qx, qy, qz, e) * w_det;
if (const_coeff) { continue; }
D(qx, qy, qz, 1, e) = C(1, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 2, e) = C(2, qx, qy, qz, e) * w_det;
if (vector_coeff) { continue; }
D(qx, qy, qz, 3, e) = C(3, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 4, e) = C(4, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 5, e) = C(5, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 6, e) = C(6, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 7, e) = C(7, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 8, e) = C(8, qx, qy, qz, e) * w_det;
}
}
}
});
}
else
{
MFEM_ABORT("Unknown VectorMassIntegrator::AssemblePA kernel for"
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
}
}
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal2D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
Vector &diag_,
const int d1d = 0,
const int q1d = 0)
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
// Use CEED backend if available
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
// Add the VectorMassAddMultPA specializations
static const auto vector_mass_kernel_specializations =
( // 2D
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 2,2>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,3>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 5,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 6,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 7,7>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 8,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 9,9>::Add(),
// 3D
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,2>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,3>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 6,7>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 7,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 8,9>::Add(),
true);
MFEM_CONTRACT_VAR(vector_mass_kernel_specializations);
VectorMassAddMultPA::Run(dim, dofs1D, quad1D,
ne, coeff_vdim, maps->B, pa_data, x, y,
dofs1D, quad1D);
}
template <const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal2D(const int NE,
const Array<real_t> &b,
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
{
constexpr int VDIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -137,7 +235,7 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
temp[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp[qx][dy] += B(qy, dy) * B(qy, dy) * op(qx, qy, e);
temp[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
}
}
}
@@ -150,33 +248,31 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
{
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
}
y(dx, dy, 0, e) = temp1;
y(dx, dy, 1, e) = temp1;
Y(dx, dy, 0, e) = temp1;
Y(dx, dy, 1, e) = temp1;
}
}
});
}
template<const int T_D1D = 0, const int T_Q1D = 0>
template <const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal3D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
Vector &diag_,
const int d1d = 0,
const int q1d = 0)
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
const auto B = Reshape(B_.Read(), Q1D, D1D);
MFEM_VERIFY(pa_data.Size() == Q1D * Q1D * Q1D * NE, "pa_data size error");
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
@@ -192,7 +288,8 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
temp[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
temp[qx][qy][dz] += B(qz, dz) * B(qz, dz) * op(qx, qy, qz, e);
temp[qx][qy][dz] +=
B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
}
}
}
@@ -207,7 +304,8 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
temp2[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp2[qx][dy][dz] += B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
temp2[qx][dy][dz] +=
B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
}
}
}
@@ -221,323 +319,42 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
real_t temp3 = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
temp3 += B(qx, dx) * B(qx, dx)
* temp2[qx][dy][dz];
temp3 += B(qx, dx) * B(qx, dx) * temp2[qx][dy][dz];
}
y(dx, dy, dz, 0, e) = temp3;
y(dx, dy, dz, 1, e) = temp3;
y(dx, dy, dz, 2, e) = temp3;
Y(dx, dy, dz, 0, e) = temp3;
Y(dx, dy, dz, 1, e) = temp3;
Y(dx, dy, dz, 2, e) = temp3;
}
}
}
});
}
static void PAVectorMassAssembleDiagonal(const int dim,
const int D1D,
const int Q1D,
const int NE,
static void PAVectorMassAssembleDiagonal(const int dim, const int D1D,
const int Q1D, const int NE,
const Array<real_t> &B,
const Array<real_t> &Bt,
const Vector &op,
Vector &y)
const Vector &pa_data,
Vector &diag)
{
if (dim == 2)
{
return PAVectorMassAssembleDiagonal2D(NE, B, Bt, op, y, D1D, Q1D);
return PAVectorMassAssembleDiagonal2D(NE, B, pa_data, diag, D1D, Q1D);
}
else if (dim == 3)
{
return PAVectorMassAssembleDiagonal3D(NE, B, Bt, op, y, D1D, Q1D);
return PAVectorMassAssembleDiagonal3D(NE, B, pa_data, diag, D1D, Q1D);
}
MFEM_ABORT("Dimension not implemented.");
}
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (DeviceCanUseCeed())
{
ceedOp->GetDiagonal(diag);
}
if (DeviceCanUseCeed()) { ceedOp->GetDiagonal(diag); }
else
{
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->Bt,
pa_data, diag);
}
}
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassApply2D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t sol_xy[max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++c)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t sol_x[max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
sol_x[qy] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t d2q = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += d2q * sol_x[qx];
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= op(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t s = sol_xy[qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const real_t q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += q2d * sol_x[dx];
}
}
}
}
});
}
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassApply3D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t sol_xyz[max_Q1D][max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++ c)
{
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
real_t sol_xy[max_Q1D][max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t sol_x[max_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] = 0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t wy = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += wy * sol_x[qx];
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t wz = B(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
real_t sol_xy[max_D1D][max_D1D];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t s = sol_xyz[qz][qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const real_t wy = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] += wy * sol_x[dx];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const real_t wz = Bt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) += wz * sol_xy[dy][dx];
}
}
}
}
}
});
}
static void PAVectorMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<real_t> &B,
const Array<real_t> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
}
if (dim == 3)
{
return PAVectorMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
if (DeviceCanUseCeed())
{
ceedOp->AddMult(x, y);
}
else
{
PAVectorMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
MFEM_VERIFY(coeff_vdim == 1, "coeff_vdim != 1");
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported");
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
}
+212
View File
@@ -0,0 +1,212 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "../kernels.hpp"
using mfem::kernels::internal::SetMaxOf;
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template <int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorMassApply2D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Vector &d,
const Vector &x,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 2, VDIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read();
const auto D = Reshape(d.Read(), Q1D, Q1D, coeff_vdim, NE);
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::v_regs2d_t<VDIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t Qx = r1[0][qy][qx];
const real_t Qy = r1[1][qy][qx];
const real_t D0 = D(qx, qy, 0, e);
if (const_coeff)
{
r0[0][qy][qx] = D0 * Qx;
r0[1][qy][qx] = D0 * Qy;
}
if (vector_coeff)
{
const real_t D1 = D(qx, qy, 1, e);
r0[0][qy][qx] = D0 * Qx;
r0[1][qy][qx] = D1 * Qy;
}
if (matrix_coeff)
{
const real_t D1 = D(qx, qy, 1, e);
const real_t D2 = D(qx, qy, 2, e);
const real_t D3 = D(qx, qy, 3, e);
r0[0][qy][qx] = D0 * Qx + D1 * Qy;
r0[1][qy][qx] = D2 * Qx + D3 * Qy;
}
}
}
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
template <int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorMassApply3D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Vector &d,
const Vector &x,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == VDIM;
const bool matrix_coeff = coeff_vdim == VDIM*VDIM;
const auto B = b.Read();
const auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, coeff_vdim, NE);
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t Qx = r1[0][qz][qy][qx];
const real_t Qy = r1[1][qz][qy][qx];
const real_t Qz = r1[2][qz][qy][qx];
const real_t D0 = D(qx, qy, qz, 0, e);
if (const_coeff)
{
r0[0][qz][qy][qx] = D0 * Qx;
r0[1][qz][qy][qx] = D0 * Qy;
r0[2][qz][qy][qx] = D0 * Qz;
}
if (vector_coeff)
{
const real_t D1 = D(qx, qy, qz, 1, e);
const real_t D2 = D(qx, qy, qz, 2, e);
r0[0][qz][qy][qx] = D0 * Qx;
r0[1][qz][qy][qx] = D1 * Qy;
r0[2][qz][qy][qx] = D2 * Qz;
}
if (matrix_coeff)
{
const real_t D1 = D(qx, qy, qz, 1, e);
const real_t D2 = D(qx, qy, qz, 2, e);
const real_t D3 = D(qx, qy, qz, 3, e);
const real_t D4 = D(qx, qy, qz, 4, e);
const real_t D5 = D(qx, qy, qz, 5, e);
const real_t D6 = D(qx, qy, qz, 6, e);
const real_t D7 = D(qx, qy, qz, 7, e);
const real_t D8 = D(qx, qy, qz, 8, e);
r0[0][qz][qy][qx] = D0 * Qx + D1 * Qy + D2 * Qz;
r0[1][qz][qy][qx] = D3 * Qx + D4 * Qy + D5 * Qz;
r0[2][qz][qy][qx] = D6 * Qx + D7 * Qy + D8 * Qz;
}
}
}
}
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
});
}
} // namespace internal
template<int DIM, int T_D1D, int T_Q1D>
VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Kernel()
{
if (DIM == 2)
{
return internal::SmemPAVectorMassApply2D<T_D1D,T_Q1D>;
}
else if (DIM == 3)
{
return internal::SmemPAVectorMassApply3D<T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int d1d, int q1d)
{
if (dim == 2)
{
return internal::SmemPAVectorMassApply2D;
}
else if (dim == 3)
{
return internal::SmemPAVectorMassApply3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+710 -3
View File
@@ -14,6 +14,7 @@
#include "../config/config.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/tensor.hpp"
namespace mfem
{
@@ -26,7 +27,713 @@ namespace kernels
namespace internal
{
/// Load B1d matrice into shared memory
// Types for tensors mapped to registers
// - N is the number of threads in each of the x and y dimensions
// - N should not be greater than 32, to have a maximum of 1024 threads
// On GPU, the last two dimensions are set to 0 to match a 2D tile of threads
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
template <int N = 0>
using s_regs2d_t = mfem::future::tensor<real_t, 0, 0>;
template <int VDIM, int N>
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, 0, 0>;
template <int VDIM, int DIM, int N = 0>
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, 0, 0>;
template <int N>
using s_regs3d_t = mfem::future::tensor<real_t, N, 0, 0>;
template <int VDIM, int N>
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, 0, 0>;
template <int VDIM, int DIM, int N>
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, 0, 0>;
// on GPU, SetMaxOf is a no-op, for minimal register usage
constexpr int SetMaxOf(int n) { return n; }
#else
template <int N>
using s_regs2d_t = mfem::future::tensor<real_t, N, N>;
template <int VDIM, int N>
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, N, N>;
template <int VDIM, int DIM, int N>
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N>;
template <int N>
using s_regs3d_t = mfem::future::tensor<real_t, N, N, N>;
template <int VDIM, int N>
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, N, N>;
template <int VDIM, int DIM, int N>
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N, N>;
// on CPU, get next multiple of 4, allowing better alignments
template <int N>
constexpr int NextMultipleOf(int n)
{
static_assert(N > 0 && (N & (N - 1)) == 0, "N must be a power of 2");
return (n + (N - 1)) & ~(N - 1);
}
constexpr int SetMaxOf(int n) { return NextMultipleOf<4>(n); }
#endif // CUDA/HIP && DEVICE_COMPILE
/// Load 2D matrix into shared memory
template <int MQ1>
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
const real_t *M, real_t (*N)[MQ1])
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
N[dy][qx] = M[dy * q1d + qx];
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input VDIM*DIM vector into given register tensor, specific component
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d, const int c,
const DeviceTensor<4, const real_t> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int d = 0; d < DIM; d++)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][d][dy][dx] = X(dx, dy, c, e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input VDIM*DIM vector into given register tensor
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c) { LoadDofs2d(e, d1d, c, X, Y); }
}
/// Load 2D input VDIM vector into given register tensor
template <int VDIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][dy][dx] = X(dx, dy, c, e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input scalar into given register tensor
template <int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<3, const real_t> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[dy][dx] = X(dx, dy, e);
}
}
MFEM_SYNC_THREAD;
}
/// Write 2D vector into given device tensor, with read (i) write (j) indices
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
const int i, const int j,
vd_regs2d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
real_t y = 0.0;
for (int d = 0; d < DIM; d++) { y += X(i, d, dy, dx); }
Y(dx, dy, j, e) += y;
}
}
MFEM_SYNC_THREAD;
}
/// Write 2D VDIM*DIM vector into given device tensor
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
vd_regs2d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c) { WriteDofs2d(e, d1d, c, c, X, Y); }
}
/// Write 2D VDIM vector into given device tensor
template <int VDIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
v_regs2d_t<VDIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y(dx, dy, c, e) += X(c, dy, dx);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input VDIM*DIM vector into given register tensor, specific component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d, const int c,
const DeviceTensor<5, const real_t> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int d = 0; d < DIM; d++)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][d][dz][dy][dx] = X(dx, dy, dz, c, e);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input VDIM*DIM vector into given register tensor
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<5, const real_t> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c) { LoadDofs3d(e, d1d, c, X, Y); }
}
/// Load 3D input VDIM vector into given register tensor
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<5, const real_t> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][dz][dy][dx] = X(dx,dy,dz,c,e);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input scalar into given register tensor
template <int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
s_regs3d_t<MQ1> &Y)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[dz][dy][dx] = X(dx,dy,dz,e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Write 3D scalar into given device tensor, with read (i) write (j) indices
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
const int i, const int j,
vd_regs3d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
real_t value = 0.0;
for (int d = 0; d < DIM; d++) { value += X(i, d, dz, dy, dx); }
Y(dx, dy, dz, j, e) += value;
}
}
}
MFEM_SYNC_THREAD;
}
/// Write 3D VDIM*DIM vector into given device tensor
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
vd_regs3d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c) { WriteDofs3d(e, d1d, c, c, X, Y); }
}
/// Write 3D VDIM vector into given device tensor
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
v_regs3d_t<VDIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y(dx, dy, dz, c, e) += X(c, dz, dy, dx);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction, X direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractX2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
{
smem[y][x] = X[y][x];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
}
Y[y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction, Y direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractY2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[y][x]; }
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
}
Y[y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar copy
template <int MQ1 = 0>
inline MFEM_HOST_DEVICE void Copy2d(const int q1d,
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { Y[y][x] = X[y][x]; }
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction: X & Y directions, with additional copy
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void Contract2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*Bx)[MQ1],
const real_t (*By)[MQ1],
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
if (!Transpose)
{
ContractX2d<false>(d1d, q1d, smem, Bx, X, Y);
ContractY2d<false>(d1d, q1d, smem, By, Y, X);
Copy2d(q1d, X, Y);
}
else
{
Copy2d(q1d, X, Y);
ContractY2d<true>(d1d, q1d, smem, By, Y, X);
ContractX2d<true>(d1d, q1d, smem, Bx, X, Y);
}
}
/// 2D scalar evaluation
template <int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
Contract2d<Transpose, MQ1>(d1d, q1d, smem, B, B, X, Y);
}
/// 2D vector evaluation
template <int VDIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs2d_t<VDIM, MQ1> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Eval2d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
}
}
/// 2D vector transposed evaluation
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void EvalTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs2d_t<VDIM, MQ1> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
Eval2d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
}
/// 2D vector gradient, with component
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
for (int d = 0; d < DIM; d++)
{
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
const real_t (*By)[MQ1] = (d == 1) ? G : B;
Contract2d<Transpose>(d1d, q1d, smem, Bx, By, X[c][d], Y[c][d]);
}
}
/// 2D vector gradient
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
}
/// 2D vector transposed gradient
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
constexpr bool Transpose = true;
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y);
}
/// 2D scalar contraction, with component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
constexpr bool Transpose = true;
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
/// 3D scalar contraction, X direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractX3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < d1d; ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
{
smem[y][x] = X[z][y][x];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
}
Y[z][y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
}
/// 3D scalar contraction, Y direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractY3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < d1d; ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[z][y][x]; }
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
}
Y[z][y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
}
/// 3D scalar contraction, Z direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractZ3d(const int d1d, const int q1d,
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < (Transpose ? d1d : q1d); ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[z][k] : B[k][z]) * X[k][y][x];
}
Y[z][y][x] = u;
}
}
}
}
/// 3D scalar contraction: X, Y & Z directions
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void Contract3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*Bx)[MQ1],
const real_t (*By)[MQ1],
const real_t (*Bz)[MQ1],
s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
if (!Transpose)
{
ContractX3d<false>(d1d, q1d, smem, Bx, X, Y);
ContractY3d<false>(d1d, q1d, smem, By, Y, X);
ContractZ3d<false>(d1d, q1d, Bz, X, Y);
}
else
{
ContractZ3d<true>(d1d, q1d, Bz, X, Y);
ContractY3d<true>(d1d, q1d, smem, By, Y, X);
ContractX3d<true>(d1d, q1d, smem, Bx, X, Y);
}
}
/// 3D scalar evaluation
template <int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
Contract3d<Transpose>(d1d, q1d, smem, B, B, B, X, Y);
}
/// 3D vector evaluation
template <int VDIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs3d_t<VDIM, MQ1> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Eval3d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
}
}
/// 3D vector transposed evaluation
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void EvalTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs3d_t<VDIM, MQ1> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
Eval3d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
}
/// 3D vector gradient, with component
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
for (int d = 0; d < DIM; d++)
{
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
const real_t (*By)[MQ1] = (d == 1) ? G : B;
const real_t (*Bz)[MQ1] = (d == 2) ? G : B;
Contract3d<Transpose>(d1d, q1d, smem, Bx, By, Bz, X[c][d], Y[c][d]);
}
}
/// 3D vector gradient
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Grad3d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
}
/// 3D vector transposed gradient
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y);
}
/// 3D vector transposed gradient, with component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y, c);
}
/// Load B1d matrix into shared memory
template<int MD1, int MQ1>
MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
const ConstDeviceMatrix &b,
@@ -48,7 +755,7 @@ MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
MFEM_SYNC_THREAD;
}
/// Load Bt1d matrices into shared memory
/// Load Bt1d matrix into shared memory
template<int MD1, int MQ1>
MFEM_HOST_DEVICE inline void LoadBt(const int D1D, const int Q1D,
const ConstDeviceMatrix &b,
@@ -1551,7 +2258,7 @@ MFEM_HOST_DEVICE inline void GradXt(const int D1D, const int Q1D,
}
}
} // namespace kernels::internal
} // namespace internal
} // namespace kernels
+139 -2
View File
@@ -203,8 +203,8 @@ void BoundaryLFIntegrator::AssembleRHSElementVect(
void BoundaryNormalLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
int dim = el.GetDim()+1;
int dof = el.GetDof();
const int dim = el.GetDim()+1;
const int dof = el.GetDof();
Vector nor(dim), Qvec;
shape.SetSize(dof);
@@ -239,6 +239,46 @@ void BoundaryNormalLFIntegrator::AssembleRHSElementVect(
}
}
void BoundaryNormalLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, FaceElementTransformations &Tr, Vector &elvect)
{
const int dim = el.GetDim();
const int dof = el.GetDof();
Vector nor(dim), Qvec;
shape.SetSize(dof);
elvect.SetSize(dof);
elvect = 0.0;
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
int intorder = oa * el.GetOrder() + ob;
ir = &IntRules.Get(el.GetGeomType(), intorder);
}
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
Tr.SetAllIntPoints(&ip);
if (dim > 1)
{
CalcOrtho(Tr.Jacobian(), nor);
}
else
{
nor[0] = 1.0;
}
Q.Eval(Qvec, *Tr.Elem1, ip);
el.CalcShape(ip, shape);
elvect.Add(ip.weight*(Qvec*nor), shape);
}
}
void BoundaryTangentialLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
@@ -918,6 +958,103 @@ void DGDirichletLFIntegrator::AssembleRHSElementVect(
}
}
void VectorFEDGDirichletLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
mfem_error("VectorFEDGDirichletLFIntegrator::AssembleRHSElementVect");
}
void VectorFEDGDirichletLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, FaceElementTransformations &Tr, Vector &elvect)
{
int dim, ndof;
bool kappa_is_nonzero = (kappa != 0.);
real_t w;
dim = el.GetDim();
ndof = el.GetDof();
nor.SetSize(dim);
nh.SetSize(dim);
ni.SetSize(dim);
if (MQ)
{
mq.SetSize(dim);
}
elvect.SetSize(ndof);
elvect = 0.0;
const IntegrationRule *ir = IntRule;
if (ir == NULL)
{
// a simple choice for the integration order; is this OK?
int order = 2*el.GetOrder();
ir = &IntRules.Get(Tr.GetGeometryType(), order);
}
Vector val(dim);
vshape.SetSize(ndof, dim);
dvshape.SetSize(ndof, dim, dim);
dvshape_dn.SetSize(ndof, dim);
dvshape.GetDenseMatrix2(dvshape_flat);
dvshape_dn_flat.SetDataAndSize(dvshape_dn.GetData(),ndof*dim);
vshape_flat.SetDataAndSize(vshape.GetData(),ndof*dim);
for (int p = 0; p < ir->GetNPoints(); p++)
{
const IntegrationPoint &ip = ir->IntPoint(p);
// Set the integration point in the face and the neighboring element
Tr.SetAllIntPoints(&ip);
// Access the neighboring element's integration point
const IntegrationPoint &eip = Tr.GetElement1IntPoint();
if (dim == 1)
{
nor(0) = 2*eip.x - 1.0;
}
else
{
CalcOrtho(Tr.Jacobian(), nor);
}
el.CalcVShape(*Tr.Elem1, vshape);
el.CalcPhysDVShape(*Tr.Elem1, dvshape);
// compute vD through the face transformation
vD->Eval(val, Tr, ip);
w = ip.weight;
if (!MQ)
{
if (Q)
{
w *= Q->Eval(*Tr.Elem1, eip);
}
ni.Set(w, nor);
}
else
{
nh.Set(w, nor);
MQ->Eval(mq, *Tr.Elem1, eip);
mq.MultTranspose(nh, ni);
}
dvshape_flat.Mult(ni, dvshape_dn_flat);
dvshape_dn.AddMult(val, elvect, sigma);
if (kappa_is_nonzero)
{
real_t hn = Tr.Elem1->Weight()/Tr.Weight();
vshape.AddMult(val, elvect, kappa*(ni*nor)/(Tr.Weight()*hn));
}
}
}
void DGElasticityDirichletLFIntegrator::AssembleRHSElementVect(
const FiniteElement &el, ElementTransformation &Tr, Vector &elvect)
{
+53
View File
@@ -241,6 +241,10 @@ public:
ElementTransformation &Tr,
Vector &elvect) override;
void AssembleRHSElementVect(const FiniteElement &el,
FaceElementTransformations &Tr,
Vector &elvect) override;
using LinearFormIntegrator::AssembleRHSElementVect;
};
@@ -602,6 +606,55 @@ public:
};
/** Boundary linear integrator for imposing non-zero Dirichlet boundary
conditions, to be used in conjunction with DGDiffusionIntegrator.
Specifically, given the Dirichlet data $u_D$, the linear form assembles the
following integrals on the boundary:
$$
\sigma \langle u_D, (Q \nabla v) \cdot n \rangle + \kappa \langle {h^{-1} Q} u_D, v \rangle,
$$
where Q is a scalar or matrix diffusion coefficient and v is the test
function. The parameters $\sigma$ and $\kappa$ should be the same as the ones
used in the DGDiffusionIntegrator. */
class VectorFEDGDirichletLFIntegrator : public LinearFormIntegrator
{
protected:
Coefficient *Q;
VectorCoefficient *vD;
MatrixCoefficient *MQ;
real_t sigma, kappa;
// these are not thread-safe!
Vector nor, nh, ni;
DenseMatrix mq;
DenseMatrix vshape, dvshape_dn;
DenseTensor dvshape;
DenseMatrix dvshape_flat;
Vector dvshape_dn_flat, vshape_flat;
public:
VectorFEDGDirichletLFIntegrator(VectorCoefficient &v, const real_t s,
const real_t k)
: Q(NULL), vD(&v), MQ(NULL), sigma(s), kappa(k) { }
VectorFEDGDirichletLFIntegrator(VectorCoefficient &v, Coefficient &q,
const real_t s, const real_t k)
: Q(&q), vD(&v), MQ(NULL), sigma(s), kappa(k) { }
VectorFEDGDirichletLFIntegrator(VectorCoefficient &v, MatrixCoefficient &q,
const real_t s, const real_t k)
: Q(NULL), vD(&v), MQ(&q), sigma(s), kappa(k) { }
void AssembleRHSElementVect(const FiniteElement &el,
ElementTransformation &Tr,
Vector &elvect) override;
void AssembleRHSElementVect(const FiniteElement &el,
FaceElementTransformations &Tr,
Vector &elvect) override;
using LinearFormIntegrator::AssembleRHSElementVect;
};
/** Boundary linear form integrator for imposing non-zero Dirichlet boundary
conditions, in a DG elasticity formulation. Specifically, the linear form is
given by
+13 -8
View File
@@ -5220,10 +5220,10 @@ void ConformingProlongationOperator::Mult(const Vector &x, Vector &y) const
for (int i = 0; i < m; i++)
{
const int end = external_ldofs[i];
std::copy(xdata+j-i, xdata+end-i, ydata+j);
if (end > j) { std::copy(xdata+j-i, xdata+end-i, ydata+j); }
j = end+1;
}
std::copy(xdata+j-m, xdata+Width(), ydata+j);
if (Width() > (j-m)) { std::copy(xdata+j-m, xdata+Width(), ydata+j); }
const int out_layout = 0; // 0 - output is ldofs array
if (!local)
@@ -5251,10 +5251,10 @@ void ConformingProlongationOperator::MultTranspose(
for (int i = 0; i < m; i++)
{
const int end = external_ldofs[i];
std::copy(xdata+j, xdata+end, ydata+j-i);
if (end > j) { std::copy(xdata+j, xdata+end, ydata+j-i); }
j = end+1;
}
std::copy(xdata+j, xdata+Height(), ydata+j-m);
if (Height() > j) { std::copy(xdata+j, xdata+Height(), ydata+j-m); }
const int out_layout = 2; // 2 - output is an array on all ltdofs
if (!local)
@@ -5271,7 +5271,8 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
MFEM_ASSERT(R->Finalized(), "");
const int tdofs = R->Height();
MFEM_ASSERT(tdofs == R->HostReadI()[tdofs], "");
ltdof_ldof = Array<int>(const_cast<int*>(R->HostReadJ()), tdofs);
ltdof_ldof.SetSize(tdofs);
ltdof_ldof.CopyFrom(R->HostReadJ());
{
Table nbr_ltdof;
gc.GetNeighborLTDofTable(nbr_ltdof);
@@ -5294,9 +5295,13 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
}
Table unique_shr;
Transpose(shared_ltdof, unique_shr, unique_ltdof.Size());
unq_ltdof = Array<int>(unique_ltdof, unique_ltdof.Size());
unq_shr_i = Array<int>(unique_shr.GetI(), unique_shr.Size()+1);
unq_shr_j = Array<int>(unique_shr.GetJ(), unique_shr.Size_of_connections());
unq_ltdof = unique_ltdof;
// Steal I and J arrays from the unique_shr table.
unq_shr_i.GetMemory() = unique_shr.GetIMemory();
unq_shr_i.SetSize(unique_shr.Size()+1);
unq_shr_j.GetMemory() = unique_shr.GetJMemory();
unq_shr_j.SetSize(unique_shr.Size_of_connections());
unique_shr.LoseData();
}
nbr_ltdof.GetJMemory().Delete();
nbr_ltdof.LoseData();
+1 -1
View File
@@ -1407,7 +1407,7 @@ real_t L2ZZErrorEstimator(BilinearFormIntegrator &flux_integrator,
}
PLBound ParGridFunction::GetBounds(Vector &lower, Vector &upper,
const int ref_factor, const int vdim)
const int ref_factor, const int vdim) const
{
PLBound plb = GridFunction::GetBounds(lower, upper, ref_factor, vdim);
int siz = vdim > 0 ? 1 : fes->GetVDim();
+1 -1
View File
@@ -587,7 +587,7 @@ public:
/// PLBound object used to compute the bounds. Note: if vdim < 1, we compute
/// the bounds for each vector dimension.
PLBound GetBounds(Vector &lower, Vector &upper,
const int ref_factor=1, const int vdim=-1) override;
const int ref_factor=1, const int vdim=-1) const override;
/** Save the local portion of the ParGridFunction. This differs from the
serial GridFunction::Save in that it takes into account the signs of
+40 -7
View File
@@ -56,6 +56,20 @@ void QuadratureFunction::Save(std::ostream &os) const
os.flush();
}
void QuadratureFunction::ProjectGridFunctionFallback(const GridFunction &gf)
{
if (gf.VectorDim() == 1)
{
GridFunctionCoefficient coeff(&gf);
coeff.Coefficient::Project(*this);
}
else
{
VectorGridFunctionCoefficient coeff(&gf);
coeff.VectorCoefficient::Project(*this);
}
}
void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
{
SetVDim(gf.VectorDim());
@@ -68,14 +82,23 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
ElementDofOrdering::LEXICOGRAPHIC :
ElementDofOrdering::NATIVE;
// Use quadrature interpolator to go from E-vector to Q-vector
const QuadratureInterpolator *qi =
gf_fes.GetQuadratureInterpolator(*qs_elem);
// If quadrature interpolator doesn't support this space, then fallback
// on slower (non-device) version, and return early.
if (!qi)
{
ProjectGridFunctionFallback(gf);
return;
}
// Use element restriction to go from L-vector to E-vector
const Operator *R = gf_fes.GetElementRestriction(ordering);
Vector e_vec(R->Height());
R->Mult(gf, e_vec);
// Use quadrature interpolator to go from E-vector to Q-vector
const QuadratureInterpolator *qi =
gf_fes.GetQuadratureInterpolator(*qs_elem);
qi->SetOutputLayout(QVectorLayout::byVDIM);
qi->DisableTensorProducts(!use_tensor_products);
qi->PhysValues(e_vec, *this);
@@ -83,12 +106,25 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
else if (auto *qs_face = dynamic_cast<FaceQuadratureSpace*>(qspace))
{
const FiniteElementSpace &gf_fes = *gf.FESpace();
const FaceType face_type = qs_face->GetFaceType();
const bool use_tensor_products = UsesTensorBasis(gf_fes);
const ElementDofOrdering ordering = use_tensor_products ?
ElementDofOrdering::LEXICOGRAPHIC :
ElementDofOrdering::NATIVE;
const FaceType face_type = qs_face->GetFaceType();
// Use quadrature interpolator to go from E-vector to Q-vector
const FaceQuadratureInterpolator *qi =
gf_fes.GetFaceQuadratureInterpolator(qspace->GetIntRule(0), face_type);
// If quadrature interpolator doesn't support this space, then fallback
// on slower (non-device) version, and return early. Also, currently,
// ElementDofOrdering::NATIVE in FaceRestriction, so fall back in that
// case too.
if (qi == nullptr || ordering == ElementDofOrdering::NATIVE)
{
ProjectGridFunctionFallback(gf);
return;
}
// Use element restriction to go from L-vector to E-vector
const Operator *R = gf_fes.GetFaceRestriction(
@@ -96,9 +132,6 @@ void QuadratureFunction::ProjectGridFunction(const GridFunction &gf)
Vector e_vec(R->Height());
R->Mult(gf, e_vec);
// Use quadrature interpolator to go from E-vector to Q-vector
const FaceQuadratureInterpolator *qi =
gf_fes.GetFaceQuadratureInterpolator(qspace->GetIntRule(0), face_type);
qi->SetOutputLayout(QVectorLayout::byVDIM);
qi->DisableTensorProducts(!use_tensor_products);
qi->Values(e_vec, *this);
+12
View File
@@ -27,6 +27,8 @@ protected:
bool own_qspace; ///< Does this own the associated QuadratureSpaceBase?
int vdim; ///< Vector dimension.
void ProjectGridFunctionFallback(const GridFunction &gf);
public:
/// Default constructor, results in an empty vector.
QuadratureFunction() : qspace(nullptr), own_qspace(false), vdim(0)
@@ -41,6 +43,12 @@ public:
qspace(&qspace_), own_qspace(false), vdim(vdim_)
{ UseDevice(true); }
/// Same as above but specify the device memory type
QuadratureFunction(QuadratureSpaceBase &qspace_, MemoryType mt, int vdim_ = 1)
: Vector(vdim_*qspace_.GetSize(), mt),
qspace(&qspace_), own_qspace(false), vdim(vdim_)
{ UseDevice(true); }
/// Create a QuadratureFunction based on the given QuadratureSpaceBase.
/** The QuadratureFunction does not assume ownership of the
QuadratureSpaceBase.
@@ -48,6 +56,10 @@ public:
QuadratureFunction(QuadratureSpaceBase *qspace_, int vdim_ = 1)
: QuadratureFunction(*qspace_, vdim_) { }
/// Same as above but specify the device memory type
QuadratureFunction(QuadratureSpaceBase *qspace_, MemoryType mt, int vdim_ = 1)
: QuadratureFunction(*qspace_, mt, vdim_) { }
/** @brief Create a QuadratureFunction based on the given QuadratureSpaceBase,
using the external (host) data, @a qf_data. */
/** The QuadratureFunction does not assume ownership of the
+12 -6
View File
@@ -69,9 +69,7 @@ QuadratureInterpolator::QuadratureInterpolator(const FiniteElementSpace &fes,
d_buffer.UseDevice(true);
if (fespace->GetNE() == 0) { return; }
const FiniteElement *fe = fespace->GetTypicalFE();
MFEM_VERIFY(fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV,
MFEM_VERIFY(SupportsFESpace(fes),
"Only elements with MapType VALUE and H_DIV are supported!");
}
@@ -86,12 +84,20 @@ QuadratureInterpolator::QuadratureInterpolator(const FiniteElementSpace &fes,
{
d_buffer.UseDevice(true);
if (fespace->GetNE() == 0) { return; }
const FiniteElement *fe = fespace->GetTypicalFE();
MFEM_VERIFY(fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV,
MFEM_VERIFY(SupportsFESpace(fes),
"Only elements with MapType VALUE and H_DIV are supported!");
}
bool QuadratureInterpolator::SupportsFESpace(const FiniteElementSpace &fespace)
{
const FiniteElement *fe = fespace.GetTypicalFE();
const Mesh &mesh = *fespace.GetMesh();
return (fe->GetMapType() == FiniteElement::MapType::VALUE ||
fe->GetMapType() == FiniteElement::MapType::H_DIV)
&& (!fespace.IsVariableOrder())
&& (!mesh.IsMixedMesh());
}
namespace internal
{
+3
View File
@@ -155,6 +155,9 @@ public:
void MultTranspose(unsigned eval_flags, const Vector &q_val,
const Vector &q_der, Vector &e_vec) const;
/// @brief Returns true if the given finite element space is supported by
/// QuadratureInterpolator.
static bool SupportsFESpace(const FiniteElementSpace &fespace);
using TensorEvalKernelType = void(*)(const int, const real_t *, const real_t *,
real_t *, const int, const int, const int);
+11 -11
View File
@@ -77,17 +77,17 @@ FaceQuadratureInterpolator::FaceQuadratureInterpolator(
if (fespace->GetNE() == 0) { return; }
GetSigns(*fespace, type, signs);
const FiniteElement *fe = fespace->GetTypicalFE();
const ScalarFiniteElement *sfe =
dynamic_cast<const ScalarFiniteElement*>(fe);
const TensorBasisElement *tfe =
dynamic_cast<const TensorBasisElement*>(fe);
MFEM_VERIFY(sfe != NULL, "Only scalar finite elements are supported");
MFEM_VERIFY(tfe != NULL &&
(tfe->GetBasisType()==BasisType::GaussLobatto ||
tfe->GetBasisType()==BasisType::Positive),
"Only Gauss-Lobatto and Bernstein basis are supported in "
"FaceQuadratureInterpolator.");
MFEM_VERIFY(SupportsFESpace(fes), "Unsupported finite element space");
}
bool FaceQuadratureInterpolator::SupportsFESpace(const FiniteElementSpace &fes)
{
const FiniteElement *fe = fes.GetTypicalFE();
const auto *sfe = dynamic_cast<const ScalarFiniteElement*>(fe);
const auto *tfe = dynamic_cast<const TensorBasisElement*>(fe);
return sfe != nullptr && tfe != nullptr && (
tfe->GetBasisType() == BasisType::GaussLobatto ||
tfe->GetBasisType() == BasisType::Positive);
}
template<const int T_VDIM, const int T_ND1D, const int T_NQ1D>
+4
View File
@@ -64,6 +64,10 @@ public:
FaceQuadratureInterpolator(const FiniteElementSpace &fes,
const IntegrationRule &ir, FaceType type);
/// @brief Returns true if the given finite element space is supported by
/// FaceQuadratureInterpolator.
static bool SupportsFESpace(const FiniteElementSpace &fes);
/** @brief Disable the use of tensor product evaluations, for tensor-product
elements, e.g. quads and hexes. */
/** Currently, tensor product evaluations are not implemented and this method
+9
View File
@@ -5368,6 +5368,15 @@ void TMOP_Integrator::ParEnableNormalization(const ParGridFunction &x)
}
#endif
void TMOP_Integrator::GetNormalizationFactors(real_t &m_normal,
real_t &l_normal,
real_t &s_normal)
{
m_normal = this->metric_normal;
l_normal = this->lim_normal;
s_normal = this->surf_fit_normal;
}
void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy,
real_t &lim_energy)
+43
View File
@@ -452,6 +452,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 9; }
};
/// 2D non-barrier Shape+Size+Orientation (VOS) metric (polyconvex).
@@ -502,6 +504,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 22; }
};
/// 2D barrier shape metric (polyconvex).
@@ -522,6 +526,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 50; }
};
/// 2D non-barrier size (V) metric (not polyconvex).
@@ -593,6 +599,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 58; }
};
/// 2D non-barrier Shape+Size (VS) metric.
@@ -675,6 +683,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 85; }
};
/// 2D compound barrier Shape+Size (VS) metric (balanced).
@@ -732,6 +742,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 98; }
};
/// 2D untangling metric.
@@ -751,6 +763,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 211; }
};
/// Shifted barrier form of metric 56 (area, ideal barrier metric), 2D
@@ -771,6 +785,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 252; }
};
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
@@ -790,6 +806,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 301; }
};
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
@@ -872,6 +890,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 311; }
};
/// 3D Shape (S) metric, untangling version of 303.
@@ -930,6 +950,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 316; }
};
/// 3D Size (V) metric.
@@ -1074,6 +1096,7 @@ public:
AddQualityMetric(sz_metric, gamma);
}
int Id() const override { return 333; }
virtual ~TMOP_Metric_333() { delete sh_metric; delete sz_metric; }
};
@@ -1136,6 +1159,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 342; }
};
/// 3D barrier Shape+Size (VS) metric, well-posed (polyconvex).
@@ -1177,6 +1202,8 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 352; }
};
/// 3D non-barrier Shape (S) metric.
@@ -1584,6 +1611,11 @@ protected:
const TargetType target_type;
bool uses_phys_coords; // see UsesPhysicalCoordinates()
/// Cached copy of GeomToPerfGeomJac used on device.
mutable DenseMatrix current_W;
/// Geometry type of current W matrix (used for cache invalidation).
mutable Geometry::Type current_W_type = Geometry::INVALID;
#ifdef MFEM_USE_MPI
MPI_Comm comm;
#endif
@@ -1963,6 +1995,12 @@ class TMOP_Integrator : public NonlinearFormIntegrator
protected:
friend class TMOPNewtonSolver;
friend class TMOPComboIntegrator;
friend class TMOPEnergyPA2D;
friend class TMOPEnergyPA3D;
friend class TMOPAssembleGradPA2D;
friend class TMOPAssembleGradPA3D;
friend class TMOPAddMultPA2D;
friend class TMOPAddMultPA3D;
// Initial positions of the mesh nodes. Not owned. The pointer is set at the
// start of the solve by TMOPNewtonSolver::Mult(), and unset at the end.
@@ -2483,6 +2521,11 @@ public:
void ParEnableNormalization(const ParGridFunction &x);
#endif
/** @brief Get the normalization factors of the metric */
void GetNormalizationFactors(real_t &metric_normal,
real_t &lim_normal,
real_t &surf_fit_normal);
/** @brief Enables FD-based approximation and computes dx. */
void EnableFiniteDifferences(const GridFunction &x);
#ifdef MFEM_USE_MPI
+180
View File
@@ -0,0 +1,180 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
/* // Original i-j assembly (old invariants code).
for (int e = 0; e < NE; e++)
{
for (int q = 0; q < nqp; q++)
{
el.CalcDShape(ip, DSh);
Mult(DSh, Jrt, DS);
for (int i = 0; i < dof; i++)
{
for (int j = 0; j < dof; j++)
{
for (int r = 0; r < dim; r++)
{
for (int c = 0; c < dim; c++)
{
for (int rr = 0; rr < dim; rr++)
{
for (int cc = 0; cc < dim; cc++)
{
const real_t H = h(r, c, rr, cc);
A(e, i + r*dof, j + rr*dof) +=
weight_q * DS(i, c) * DS(j, cc) * H;
}
}
}
}
}
}
}
}*/
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_2D(const int NE,
const ConstDeviceMatrix &B,
const ConstDeviceMatrix &G,
const DeviceTensor<5, const real_t> &J,
const DeviceTensor<7, const real_t> &H,
DeviceTensor<4> &D,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
// Takes into account Jtr by replacing H with Href at all quad points.
MFEM_SHARED real_t Href_data[2 * 2 * 2 * MQ1 * MQ1];
DeviceTensor<5, real_t> Href(Href_data, 2, 2, 2, MQ1, MQ1);
for (int v = 0; v < 2; v++)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
real_t Jrt_data[4];
ConstDeviceMatrix Jrt(Jrt_data, 2, 2);
kernels::CalcInverse<2>(Jtr, Jrt_data);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
Href(v, m, n, qx, qy) = 0.0;
for (int s = 0; s < 2; s++)
{
for (int t = 0; t < 2; t++)
{
Href(v, m, n, qx, qy) +=
Jrt(m, s) * H(v, s, v, t, qx, qy, e) * Jrt(n, t);
}
}
}
}
}
}
}
MFEM_SHARED real_t qd[2 * 2 * MQ1 * MD1];
DeviceTensor<4, real_t> QD(qd, 2, 2, MQ1, MD1);
for (int v = 0; v < 2; v++)
{
// Contract in y.
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++) { QD(m, n, qx, dy) = 0.0; }
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
const real_t L = (m == 1 ? Gy : By);
const real_t R = (n == 1 ? Gy : By);
QD(m, n, qx, dy) += L * Href(v, m, n, qx, qy) * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
// Contract in x.
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
const real_t L = (m == 0 ? Gx : Bx);
const real_t R = (n == 0 ? Gx : Bx);
d += L * QD(m, n, qx, dy) * R;
}
}
}
D(dx, dy, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag2D, TMOP_AssembleDiagPA_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag2D);
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto G = Reshape(PA.maps->G.Read(), q, d);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto H = Reshape(PA.H.Read(), 2, 2, 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiag2D::Run(d, q, NE, B, G, J, H, D, d, q);
}
} // namespace mfem
+83
View File
@@ -0,0 +1,83 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../../general/forall.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_C0_2D(const int NE,
const ConstDeviceMatrix &B,
const DeviceTensor<5, const real_t> &H0,
DeviceTensor<4> &D,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t qd[MQ1 * MD1];
DeviceTensor<2, real_t> QD(qd, MQ1, MD1);
for (int v = 0; v < 2; v++)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
QD(qx, dy) = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t bb = B(qy, dy) * B(qy, dy);
QD(qx, dy) += bb * H0(v, v, qx, qy, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t bb = B(qx, dx) * B(qx, dx);
d += bb * QD(qx, dy);
}
D(dx, dy, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef2D, TMOP_AssembleDiagPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef2D);
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto H0 = Reshape(PA.H0.Read(), 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiagCoef2D::Run(d, q, NE, B, H0, D, d, q);
}
} // namespace mfem
+229
View File
@@ -0,0 +1,229 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_3D(const int NE,
const ConstDeviceMatrix &B,
const ConstDeviceMatrix &G,
const DeviceTensor<6, const real_t> &J,
const DeviceTensor<8, const real_t> &H,
DeviceTensor<5> &D,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[3][3][MQ1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> rH, r0, r1;
for (int v = 0; v < 3; ++v)
{
// Takes into account Jtr by replacing H with Href at all quad points.
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
real_t Jrt_data[9];
ConstDeviceMatrix Jrt(Jrt_data, 3, 3);
kernels::CalcInverse<3>(Jtr, Jrt_data);
real_t h[3][3];
for (int s = 0; s < 3; s++)
{
for (int t = 0; t < 3; t++)
{
h[s][t] = H(v, s, v, t, qx, qy, qz, e);
}
}
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
rH(m, n, qz, qy, qx) = 0.0;
for (int s = 0; s < 3; s++)
{
for (int t = 0; t < 3; t++)
{
rH(m, n, qz, qy, qx) += Jrt(m, s) * h[s][t] * Jrt(n, t);
}
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in z.
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
r0(m, n, dz, qy, qx) = 0.0;
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t Bz = B(qz, dz), Gz = G(qz, dz);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 2 ? Gz : Bz);
const real_t R = (n == 2 ? Gz : Bz);
r0(m, n, dz, qy, qx) += L * rH(m, n, qz, qy, qx) * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in y.
for (int dz = 0; dz < D1D; ++dz)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[m][n][qy][qx] = r0(m, n, dz, qy, qx);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
r1(m, n, dz, dy, qx) = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 1 ? Gy : By);
const real_t R = (n == 1 ? Gy : By);
r1(m, n, dz, dy, qx) += L * smem[m][n][qy][qx] * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in x.
for (int dz = 0; dz < D1D; ++dz)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[m][n][dy][qx] = r1(m, n, dz, dy, qx);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 0 ? Gx : Bx);
const real_t R = (n == 0 ? Gx : Bx);
d += L * smem[m][n][dy][qx] * R;
}
}
}
D(dx, dy, dz, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag3D, TMOP_AssembleDiagPA_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag3D);
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto G = Reshape(PA.maps->G.Read(), q, d);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto H = Reshape(PA.H.Read(), 3, 3, 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiag3D::Run(d, q, NE, B, G, J, H, D, d, q);
}
} // namespace mfem
+131
View File
@@ -0,0 +1,131 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_C0_3D(const int NE,
const ConstDeviceMatrix &B,
const DeviceTensor<6, const real_t> &H0,
DeviceTensor<5> &D,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::s_regs3d_t<MQ1> r0, r1;
for (int v = 0; v < 3; ++v)
{
// first tensor contraction, along z direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t Bz = B(qz, dz);
u += Bz * H0(v, v, qx, qy, qz, e) * Bz;
}
r0[dz][qy][qx] = u;
}
}
MFEM_SYNC_THREAD;
}
// second tensor contraction, along y direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[qy][qx] = r0[dz][qy][qx];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
u += By * smem[qy][qx] * By;
}
r1[dz][dy][qx] = u;
}
}
MFEM_SYNC_THREAD;
}
// third tensor contraction, along x direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[dy][qx] = r1[dz][dy][qx];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
u += Bx * smem[dy][qx] * Bx;
}
D(dx, dy, dz, v, e) += u;
}
}
MFEM_SYNC_THREAD;
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef3D, TMOP_AssembleDiagPA_C0_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef3D);
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto H0 = Reshape(PA.H0.Read(), 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiagCoef3D::Run(d, q, NE, B, H0, D, d, q);
}
} // namespace mfem
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "grad2.hpp"
namespace mfem
{
void TMOP_Integrator::AssembleGradPA_2D(const Vector &x) const
{
const int mid = metric->Id();
// Calls TMOPAssembleGradPA2D::Mult for the given mid.
TMOPAssembleGradPA2D ker(this, x);
if (mid == 1) { return tmop::Kernel<1>(ker); }
if (mid == 2) { return tmop::Kernel<2>(ker); }
if (mid == 7) { return tmop::Kernel<7>(ker); }
if (mid == 56) { return tmop::Kernel<56>(ker); }
if (mid == 77) { return tmop::Kernel<77>(ker); }
if (mid == 80) { return tmop::Kernel<80>(ker); }
if (mid == 94) { return tmop::Kernel<94>(ker); }
MFEM_ABORT("Unsupported TMOP metric " << mid);
}
} // namespace mfem
+106
View File
@@ -0,0 +1,106 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
class TMOPAssembleGradPA2D
{
const mfem::TMOP_Integrator *ti; // not owned
const Vector &x;
public:
TMOPAssembleGradPA2D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
x(x) {}
int Ndof() const { return ti->PA.maps->ndof; }
int Nqpt() const { return ti->PA.maps->nqpt; }
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
static void Mult(TMOPAssembleGradPA2D &ker)
{
const mfem::TMOP_Integrator *ti = ker.ti;
const real_t metric_normal = ti->metric_normal;
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
Array<real_t> mp;
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
{
m->GetWeights(mp);
}
const real_t *w = mp.Read();
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
const auto X = Reshape(ker.x.Read(), D1D, D1D, 2, NE);
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D);
const auto J = Reshape(ti->PA.Jtr.Read(), 2, 2, Q1D, Q1D, NE);
auto H = Reshape(ti->PA.H.Write(), 2, 2, 2, 2, Q1D, Q1D, NE);
const Vector &mc = ti->PA.MC;
const bool const_m0 = mc.Size() == 1;
const auto MC = const_m0
? Reshape(mc.Read(), 1, 1, 1)
: Reshape(mc.Read(), Q1D, Q1D, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t m_coef = const_m0 ? MC(0, 0, 0) : MC(qx, qy, e);
const real_t weight = metric_normal * m_coef * W(qx, qy) * detJtr;
// Jrt = Jtr^{-1}
real_t Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X^t.DSh
const real_t Jpr[4] =
{
r1[0][0][qy][qx], r1[1][0][qy][qx],
r1[0][1][qy][qx], r1[1][1][qy][qx]
};
// Jpt = Jpr.Jrt
real_t Jpt[4];
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
METRIC{}.AssembleH(qx, qy, e, weight, Jpt, w, H);
}
}
});
}
};
} // namespace mfem
+145
View File
@@ -0,0 +1,145 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_C0_2D(const real_t lim_normal,
const ConstDeviceCube &LD,
const bool const_c0,
const DeviceTensor<3, const real_t> &C0,
const int NE,
const DeviceTensor<5, const real_t> &J,
const ConstDeviceMatrix &W,
const real_t *b,
const real_t *bld,
const DeviceTensor<4, const real_t> &X0,
const DeviceTensor<4, const real_t> &X1,
DeviceTensor<5> &H0,
const bool exp_lim,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
kernels::internal::s_regs2d_t<MQ1> rm0, rm1; // scalar LD
kernels::internal::LoadDofs2d(e, D1D, LD, rm0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, rm0, rm1);
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs2d_t<2,MQ1> r00, r01; // vector X0
kernels::internal::LoadDofs2d(e, D1D, X0, r00);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::v_regs2d_t<2,MQ1> r10, r11; // vector X1
kernels::internal::LoadDofs2d(e, D1D, X1, r10);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r10, r11);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t weight = W(qx, qy) * detJtr;
const real_t coeff0 = const_c0 ? C0(0, 0, 0) : C0(qx, qy, e);
const real_t weight_m = weight * lim_normal * coeff0;
const real_t D = rm1(qy, qx);
const real_t p0[2] = { r01(0, qy, qx), r01(1, qy, qx) };
const real_t p1[2] = { r11(0, qy, qx), r11(1, qy, qx) };
const real_t dist = D; // GetValues, default comp set to 0
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
real_t grad_grad[4];
if (!exp_lim)
{
// d2.Diag(1.0 / (dist * dist), x.Size());
const real_t c = 1.0 / (dist * dist);
kernels::Diag<2>(c, grad_grad);
}
else
{
real_t tmp[2];
kernels::Subtract<2>(1.0, p1, p0, tmp);
real_t dsq = kernels::DistanceSquared<2>(p1, p0);
real_t dist_squared = dist * dist;
real_t dist_squared_squared = dist_squared * dist_squared;
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
grad_grad[0] =
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[1] =
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
grad_grad[2] = grad_grad[1];
grad_grad[3] =
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
}
ConstDeviceMatrix gg(grad_grad, 2, 2);
for (int i = 0; i < 2; i++)
{
for (int j = 0; j < 2; j++)
{
H0(i, j, qx, qy, e) = weight_m * gg(i, j);
}
}
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef2D, TMOP_AssembleGradPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef2D);
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &x) const
{
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const real_t ln = lim_normal;
const bool const_c0 = PA.C0.Size() == 1;
const auto C0 = PA.C0.Size() == 1
? Reshape(PA.C0.Read(), 1, 1, 1)
: Reshape(PA.C0.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
const auto LD = Reshape(PA.LD.Read(), d, d, NE);
const auto XL = Reshape(PA.XL.Read(), d, d, 2, NE);
const auto X = Reshape(x.Read(), d, d, 2, NE);
auto H0 = Reshape(PA.H0.Write(), 2, 2, q, q, NE);
const auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
const bool exp_lim = el ? true : false;
TMOPAssembleGradCoef2D::Run(d, q, ln, LD, const_c0, C0, NE,
J, W, b, bld, XL, X, H0, exp_lim, d, q);
}
} // namespace mfem
+35
View File
@@ -0,0 +1,35 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "grad3.hpp"
namespace mfem
{
void TMOP_Integrator::AssembleGradPA_3D(const Vector &x) const
{
const int mid = metric->Id();
// Calls TMOPAssembleGradPA3D::Mult for the given mid.
TMOPAssembleGradPA3D ker(this, x);
if (mid == 302) { return tmop::Kernel<302>(ker); }
if (mid == 303) { return tmop::Kernel<303>(ker); }
if (mid == 315) { return tmop::Kernel<315>(ker); }
if (mid == 318) { return tmop::Kernel<318>(ker); }
if (mid == 321) { return tmop::Kernel<321>(ker); }
if (mid == 332) { return tmop::Kernel<332>(ker); }
if (mid == 338) { return tmop::Kernel<338>(ker); }
MFEM_ABORT("Unsupported TMOP metric " << mid);
}
} // namespace mfem
+111
View File
@@ -0,0 +1,111 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
class TMOPAssembleGradPA3D
{
const TMOP_Integrator *ti; // not owned
const Vector &x;
public:
TMOPAssembleGradPA3D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
x(x) {}
int Ndof() const { return ti->PA.maps->ndof; }
int Nqpt() const { return ti->PA.maps->nqpt; }
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
static void Mult(TMOPAssembleGradPA3D &ker)
{
const TMOP_Integrator *ti = ker.ti;
const real_t metric_normal = ti->metric_normal;
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
Array<real_t> mp;
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
{
m->GetWeights(mp);
}
const real_t *w = mp.Read();
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
const auto X = Reshape(ker.x.Read(), D1D, D1D, D1D, 3, NE);
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D, Q1D);
const auto J = Reshape(ti->PA.Jtr.Read(), 3, 3, Q1D, Q1D, Q1D, NE);
auto H = Reshape(ti->PA.H.Write(), 3, 3, 3, 3, Q1D, Q1D, Q1D, NE);
const Vector &mc = ti->PA.MC;
const bool const_m0 = mc.Size() == 1;
const auto MC = const_m0
? Reshape(mc.Read(), 1, 1, 1, 1)
: Reshape(mc.Read(), Q1D, Q1D, Q1D, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t m_coef = const_m0 ?
MC(0, 0, 0, 0) :
MC(qx, qy, qz, e);
const real_t weight = metric_normal * m_coef * W(qx, qy, qz) * detJtr;
// Jrt = Jtr^{-1}
real_t Jrt[9];
kernels::CalcInverse<3>(Jtr, Jrt);
// Jpr = X^T.DSh
real_t Jpr[9] =
{
r1(0, 0, qz, qy, qx), r1(1, 0, qz, qy, qx), r1(2, 0, qz, qy, qx),
r1(0, 1, qz, qy, qx), r1(1, 1, qz, qy, qx), r1(2, 1, qz, qy, qx),
r1(0, 2, qz, qy, qx), r1(1, 2, qz, qy, qx), r1(2, 2, qz, qy, qx)
};
// Jpt = X^T . DS = (X^T.DSh) . Jrt = Jpr . Jrt
real_t Jpt[9];
kernels::Mult(3, 3, 3, Jpr, Jrt, Jpt);
METRIC{}.AssembleH(qx, qy, qz, e, weight, Jrt, Jpr, Jpt, w, H);
}
}
}
});
}
};
} // namespace mfem
+167
View File
@@ -0,0 +1,167 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_C0_3D(const real_t lim_normal,
const DeviceTensor<4, const real_t> &LD,
const bool const_c0,
const DeviceTensor<4, const real_t> &C0,
const int NE,
const DeviceTensor<6, const real_t> &J,
const ConstDeviceCube &W,
const real_t *b,
const real_t *bld,
const DeviceTensor<5, const real_t> &X0,
const DeviceTensor<5, const real_t> &X1,
DeviceTensor<6> &H0,
const bool exp_lim,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
kernels::internal::s_regs3d_t<MQ1> rm0, rm1; // scalar LD
kernels::internal::LoadDofs3d(e, D1D, LD, rm0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, rm0, rm1);
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs3d_t<3,MQ1> r00, r01; // vector X0
kernels::internal::LoadDofs3d(e, D1D, X0, r00);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::v_regs3d_t<3,MQ1> r10, r11; // vector X1
kernels::internal::LoadDofs3d(e, D1D, X1, r10);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r10, r11);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t weight = W(qx, qy, qz) * detJtr;
const real_t coeff0 = const_c0
? C0(0, 0, 0, 0)
: C0(qx, qy, qz, e);
const real_t weight_m = weight * lim_normal * coeff0;
const real_t D = rm1(qz, qy, qx);
const real_t p0[3] = { r01(0, qz, qy, qx),
r01(1, qz, qy, qx),
r01(2, qz, qy, qx)
};
const real_t p1[3] = { r11(0, qz, qy, qx),
r11(1, qz, qy, qx),
r11(2, qz, qy, qx)
};
const real_t dist = D; // GetValues, default comp set to 0
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
real_t grad_grad[9];
if (!exp_lim)
{
// d2.Diag(1.0 / (dist * dist), x.Size());
const real_t c = 1.0 / (dist * dist);
kernels::Diag<3>(c, grad_grad);
}
else
{
real_t tmp[3];
kernels::Subtract<3>(1.0, p1, p0, tmp);
real_t dsq = kernels::DistanceSquared<3>(p1, p0);
real_t dist_squared = dist * dist;
real_t dist_squared_squared = dist_squared * dist_squared;
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
grad_grad[0] =
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[1] =
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
grad_grad[2] =
(400.0 * tmp[0] * tmp[2] * f) / dist_squared_squared;
grad_grad[3] = grad_grad[1];
grad_grad[4] =
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[5] =
(400.0 * tmp[1] * tmp[2] * f) / dist_squared_squared;
grad_grad[6] = grad_grad[2];
grad_grad[7] = grad_grad[5];
grad_grad[8] =
((400.0 * tmp[2] * tmp[2] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
}
ConstDeviceMatrix gg(grad_grad, 3, 3);
for (int i = 0; i < 3; i++)
{
for (int j = 0; j < 3; j++)
{
H0(i, j, qx, qy, qz, e) = weight_m * gg(i, j);
}
}
}
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef3D, TMOP_AssembleGradPA_C0_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef3D);
void TMOP_Integrator::AssembleGradPA_C0_3D(const Vector &x) const
{
const real_t ln = lim_normal;
const bool const_c0 = PA.C0.Size() == 1;
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto C0 = const_c0
? Reshape(PA.C0.Read(), 1, 1, 1, 1)
: Reshape(PA.C0.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
const auto LD = Reshape(PA.LD.Read(), d, d, d, NE);
const auto XL = Reshape(PA.XL.Read(), d, d, d, 3, NE);
const auto X = Reshape(x.Read(), d, d, d, 3, NE);
auto H0 = Reshape(PA.H0.Write(), 3, 3, q, q, q, NE);
auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
const bool exp_lim = (el) ? true : false;
TMOPAssembleGradCoef3D::Run(d, q, ln, LD, const_c0, C0, NE,
J, W, b, bld, XL, X, H0, exp_lim, d, q);
}
} // namespace mfem
+76
View File
@@ -0,0 +1,76 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_001 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1();
};
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1(dI1));
kernels::Set(2, 2, 1.0, ie.Get_dI1(), P);
}
MFEM_HOST_DEVICE void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// weight * ddI1
real_t ddI1[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t h = ddi1(r, c);
H(r, c, i, j, qx, qy, e) = weight * h;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_001;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 1);
} // namespace mfem
+75
View File
@@ -0,0 +1,75 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_002 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return 0.5 * ie.Get_I1b() - 1.0;
};
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1b(dI1b).dI2b(dI2b));
kernels::Set(2, 2, 1. / 2., ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e, const real_t weight,
const real_t (&Jpt)[4], const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// 0.5 * weight * dI1b
real_t ddI1[4], ddI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b));
const real_t half_weight = 0.5 * weight;
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t h = ddi1b(r, c);
H(r, c, i, j, qx, qy, e) = half_weight * h;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_002;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 2);
} // namespace mfem
+87
View File
@@ -0,0 +1,87 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_007 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1() * (1.0 + 1.0 / ie.Get_I2()) - 4.0;
};
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI1(dI1).dI2(dI2).dI2b(dI2b));
const real_t I2 = ie.Get_I2();
kernels::Add(2, 2, 1.0 + 1.0 / I2, ie.Get_dI1(), -ie.Get_I1() / (I2 * I2),
ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).ddI1(ddI1).ddI2(ddI2).dI1(dI1).dI2(dI2).dI2b(dI2b));
const real_t c1 = 1. / ie.Get_I2();
const real_t c2 = weight * c1 * c1;
const real_t c3 = ie.Get_I1() * c2;
ConstDeviceMatrix di1(ie.Get_dI1(), DIM, DIM);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * (1.0 + c1) * ddi1(r, c) - c3 * ddi2(r, c) -
c2 * (di1(i, j) * di2(r, c) + di2(i, j) * di1(r, c)) +
2.0 * c1 * c3 * di2(r, c) * di2(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_007;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 7);
} // namespace mfem
+81
View File
@@ -0,0 +1,81 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_056 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b + 1.0 / I2b) - 1.0;
};
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
// 0.5*(1 - 1/I2b^2)*dI2b
real_t dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b));
const real_t I2b = ie.Get_I2b();
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(), P);
}
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
// (0.5 - 0.5/I2b^2)*ddI2b + (1/I2b^3)*(dI2b x dI2b)
real_t dI2b[4], ddI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b).ddI2b(ddI2b));
const real_t I2b = ie.Get_I2b();
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * (0.5 - 0.5 / (I2b * I2b)) * ddi2b(r, c) +
weight / (I2b * I2b * I2b) * di2b(r, c) * di2b(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_056;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 56);
} // namespace mfem
+80
View File
@@ -0,0 +1,80 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_077 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4],
const real_t *) const final
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
};
MFEM_HOST_DEVICE void EvalP(const real_t (&Jpt)[4],
const real_t *w, real_t (&P)[4]) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2(dI2).dI2b(dI2b));
const real_t I2 = ie.Get_I2();
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE void AssembleH(const int qx, const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) const final
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4], ddI2[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI2(dI2).dI2b(dI2b).ddI2(ddI2));
const real_t I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r, c) +
weight * (I2inv_sq / I2) * di2(r, c) * di2(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_077;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 77);
} // namespace mfem

Some files were not shown because too many files have changed in this diff Show More