Compare commits
412
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
18e636fa54 | ||
|
|
1877a40a85 | ||
|
|
99785763af | ||
|
|
e3b4edcae1 | ||
|
|
34307512d5 | ||
|
|
e1d4c6cd40 | ||
|
|
1f8f2f1826 | ||
|
|
1c9b9beb1e | ||
|
|
13252e4734 | ||
|
|
032bf0f3dd | ||
|
|
50368046bc | ||
|
|
19733980de | ||
|
|
6f8d55d35f | ||
|
|
459f93d71c | ||
|
|
7f79fc8adc | ||
|
|
a6c6d63aa6 | ||
|
|
56689d3c0e | ||
|
|
4a5f5f8528 | ||
|
|
e30c8b53d8 | ||
|
|
73afff37cc | ||
|
|
087a29249a | ||
|
|
f2e841d7f9 | ||
|
|
dcff876e94 | ||
|
|
0555b6cc97 | ||
|
|
ea329bca35 | ||
|
|
2726959a2d | ||
|
|
2fef8ca1f0 | ||
|
|
0f5b6cc0f5 | ||
|
|
058e3414b0 | ||
|
|
e756067456 | ||
|
|
aaf2d99de8 | ||
|
|
bcf5153192 | ||
|
|
93066142ab | ||
|
|
acfa01c46c | ||
|
|
47c39d5102 | ||
|
|
4c803a8214 | ||
|
|
f460b547ea | ||
|
|
fa13480fcf | ||
|
|
48683a7f02 | ||
|
|
97cb5abb35 | ||
|
|
f09ca5df3c | ||
|
|
0db3e561a3 | ||
|
|
229c9d9fbe | ||
|
|
58c6ac70c2 | ||
|
|
7c38e8ca72 | ||
|
|
c18a0e470b | ||
|
|
b9dc31be2b | ||
|
|
6f848e00f8 | ||
|
|
ce3c13781b | ||
|
|
41378e2fa3 | ||
|
|
f19768cc70 | ||
|
|
6e67b690d9 | ||
|
|
5e40324e5d | ||
|
|
a39d48748d | ||
|
|
56a122e86a | ||
|
|
49d57f4f6f | ||
|
|
09bff27a2c | ||
|
|
525ef99753 | ||
|
|
e6180f56ee | ||
|
|
132de1fa57 | ||
|
|
e1a35d929c | ||
|
|
ac84807be1 | ||
|
|
9601a02547 | ||
|
|
b9bebd0921 | ||
|
|
d276ad6aff | ||
|
|
f3aeae7b04 | ||
|
|
2944bc5ea3 | ||
|
|
2eb244fb21 | ||
|
|
c817c3cdbe | ||
|
|
a24b5edcf0 | ||
|
|
b47f3e52a3 | ||
|
|
93f3b73790 | ||
|
|
4870ecd351 | ||
|
|
5ed9f34dd8 | ||
|
|
41c8a8e65e | ||
|
|
a2025d7693 | ||
|
|
ae629dda70 | ||
|
|
70439dcb84 | ||
|
|
1485fedec9 | ||
|
|
1ead635557 | ||
|
|
c903edd931 | ||
|
|
d8cd8f4f37 | ||
|
|
4c9769a927 | ||
|
|
b83a4184b5 | ||
|
|
042762701b | ||
|
|
3b6532e7bb | ||
|
|
8731dd171e | ||
|
|
45b1b3c535 | ||
|
|
d450cc563c | ||
|
|
4083fcc4b4 | ||
|
|
f27a7c5612 | ||
|
|
1027c6a9a8 | ||
|
|
7843bfe46a | ||
|
|
71aa2f0985 | ||
|
|
35f0d5bfe1 | ||
|
|
70a3355e90 | ||
|
|
a6b1eff541 | ||
|
|
893c28bff3 | ||
|
|
207fcd2fb6 | ||
|
|
d312108f14 | ||
|
|
409aa1cba7 | ||
|
|
454a215175 | ||
|
|
8af606b848 | ||
|
|
9a2bbec272 | ||
|
|
57f4be30c5 | ||
|
|
9b9a6cfe34 | ||
|
|
dbea0c3a0b | ||
|
|
96a1cf8c93 | ||
|
|
a447a33528 | ||
|
|
79f2ce2540 | ||
|
|
a109493821 | ||
|
|
32ce005ca0 | ||
|
|
3f0e146fb6 | ||
|
|
87c5e59228 | ||
|
|
9e91bfb376 | ||
|
|
253a5b4b79 | ||
|
|
b9105ca3e8 | ||
|
|
fdb4368212 | ||
|
|
1808ca1f3a | ||
|
|
c34de48041 | ||
|
|
e9ca0960f7 | ||
|
|
8c236973c7 | ||
|
|
8d0a2d4336 | ||
|
|
0694668c24 | ||
|
|
3492b20c70 | ||
|
|
033b29edfc | ||
|
|
5fec899806 | ||
|
|
fc010f0423 | ||
|
|
6bffbc1d2e | ||
|
|
ec6937c286 | ||
|
|
acc1b3a28c | ||
|
|
a9427e2663 | ||
|
|
72deb3b12c | ||
|
|
67d31a13fd | ||
|
|
0320e5a476 | ||
|
|
041b4ab958 | ||
|
|
3374851918 | ||
|
|
46a44c5241 | ||
|
|
641692ac4a | ||
|
|
f97bfdea7e | ||
|
|
159c89542b | ||
|
|
37fc5a3d5c | ||
|
|
6fad311e11 | ||
|
|
fcdc9c43d5 | ||
|
|
d32a63171f | ||
|
|
4fdf2cdda1 | ||
|
|
0af0b9b420 | ||
|
|
3c0f1d783e | ||
|
|
cd55bcc67b | ||
|
|
689288e18d | ||
|
|
01afc1685a | ||
|
|
4501988a3b | ||
|
|
cd82063118 | ||
|
|
cac3f72e9e | ||
|
|
d93a030431 | ||
|
|
8cfd5e0a07 | ||
|
|
63110738d4 | ||
|
|
c05d58a9c2 | ||
|
|
2647e9ef3d | ||
|
|
0efc76834a | ||
|
|
1311729f79 | ||
|
|
526f73688e | ||
|
|
aa0a33f51b | ||
|
|
34abf22be5 | ||
|
|
aa35d62e28 | ||
|
|
bfa80426bb | ||
|
|
0d6eced899 | ||
|
|
e336658f8c | ||
|
|
b276939d69 | ||
|
|
324ab0684e | ||
|
|
c5efd6a06a | ||
|
|
626cb41c1b | ||
|
|
0c167ab5e7 | ||
|
|
944217e7f3 | ||
|
|
bd31eba056 | ||
|
|
eea5d48c4e | ||
|
|
5637c98f50 | ||
|
|
fe7cce14b7 | ||
|
|
94b80fcd3c | ||
|
|
3d6c5ebd71 | ||
|
|
6e2148a698 | ||
|
|
20caca8333 | ||
|
|
37ebeed499 | ||
|
|
1f6afa0f87 | ||
|
|
a18d36e841 | ||
|
|
a1b9ad1403 | ||
|
|
47c85401cd | ||
|
|
2aa40b53e4 | ||
|
|
dd21f5469c | ||
|
|
1ee9cbcc43 | ||
|
|
9c56bc4d9e | ||
|
|
229e1b1bc6 | ||
|
|
c2b508c5fc | ||
|
|
7df869974d | ||
|
|
30859f614d | ||
|
|
b73ae540e1 | ||
|
|
56615c8fc3 | ||
|
|
5e99a705c2 | ||
|
|
8d4819b143 | ||
|
|
9d1fe4ba6b | ||
|
|
8147aeba7d | ||
|
|
ce3f215d75 | ||
|
|
8edf90f72a | ||
|
|
28590b2026 | ||
|
|
5429e98bb3 | ||
|
|
0e1e7ac5c9 | ||
|
|
1c03e51342 | ||
|
|
dea236d9ff | ||
|
|
90e9ec2d50 | ||
|
|
b9da92c5a1 | ||
|
|
d84095af57 | ||
|
|
8dba8024a1 | ||
|
|
41151d5fe9 | ||
|
|
3dfe67a219 | ||
|
|
36e2c896e4 | ||
|
|
1fa6c1ded9 | ||
|
|
419de5c398 | ||
|
|
1e62733c26 | ||
|
|
7ab83365aa | ||
|
|
60100a336e | ||
|
|
e09479465b | ||
|
|
c4da5827a2 | ||
|
|
8df19f39ef | ||
|
|
405c674d22 | ||
|
|
83362c7b4d | ||
|
|
240443abfb | ||
|
|
c189f50ab4 | ||
|
|
7d7d4cf6f4 | ||
|
|
061a92067f | ||
|
|
0ddb02c7e7 | ||
|
|
427406d1b8 | ||
|
|
5026d6ca8a | ||
|
|
04d7e8a62f | ||
|
|
915853cee0 | ||
|
|
a52599d4cc | ||
|
|
0d5b13c4aa | ||
|
|
dda6b0dbe1 | ||
|
|
c579f28b3f | ||
|
|
947694ae63 | ||
|
|
e97704d91c | ||
|
|
1f8f7e44bc | ||
|
|
1aba4e0bb8 | ||
|
|
03cf0edb5b | ||
|
|
0838384b78 | ||
|
|
3e321e6c9e | ||
|
|
9c20900150 | ||
|
|
52014e215e | ||
|
|
2daa072b55 | ||
|
|
ed31cb7bbd | ||
|
|
7811775ab1 | ||
|
|
5edabb64b7 | ||
|
|
319b870ae1 | ||
|
|
ba70885008 | ||
|
|
2e706883d0 | ||
|
|
614530599f | ||
|
|
acbe258939 | ||
|
|
7dba46021e | ||
|
|
189c201792 | ||
|
|
0742475bc7 | ||
|
|
716c201fd3 | ||
|
|
f9972a57c0 | ||
|
|
b3e9ddd846 | ||
|
|
f3443f92c3 | ||
|
|
c50905e492 | ||
|
|
7a37cc2448 | ||
|
|
82c2e01678 | ||
|
|
2ead488818 | ||
|
|
190c4a091d | ||
|
|
6bb5f8ae0b | ||
|
|
3b88332c98 | ||
|
|
ea25a9bb7b | ||
|
|
62dd1aee3b | ||
|
|
e41c782974 | ||
|
|
eb6843443f | ||
|
|
51c28ade7b | ||
|
|
686feffc1b | ||
|
|
795a121f4e | ||
|
|
6a04555556 | ||
|
|
580dc6f50c | ||
|
|
d27fa842b1 | ||
|
|
3fb13a3614 | ||
|
|
3c2f98bf64 | ||
|
|
f4bbc69ae6 | ||
|
|
484249d94c | ||
|
|
2a3c1c5bde | ||
|
|
031cd67c76 | ||
|
|
7f7e5f61e2 | ||
|
|
582773213e | ||
|
|
3086daecee | ||
|
|
c707a4d3f6 | ||
|
|
ba4befcfb9 | ||
|
|
72d28e0cdb | ||
|
|
1c04593e79 | ||
|
|
0892f190cf | ||
|
|
b08d12b5f1 | ||
|
|
d2e283419a | ||
|
|
3949decb69 | ||
|
|
93dbe1404a | ||
|
|
68ae66c3ed | ||
|
|
feda9f0125 | ||
|
|
003b4175df | ||
|
|
fe83a192e9 | ||
|
|
746fbca164 | ||
|
|
e7f71da564 | ||
|
|
7d9048f9f0 | ||
|
|
3b2fe7ce7f | ||
|
|
dd64dcc10b | ||
|
|
f89e89e9fa | ||
|
|
2e27af385c | ||
|
|
f1025ecdeb | ||
|
|
09b5dcab6d | ||
|
|
22e1f000b9 | ||
|
|
8884c31461 | ||
|
|
04b868a47b | ||
|
|
8b99a6a847 | ||
|
|
4868222660 | ||
|
|
82aa3a135c | ||
|
|
f4ef788e40 | ||
|
|
8f333155cc | ||
|
|
06d9613f59 | ||
|
|
5499eb8939 | ||
|
|
5ccdd4b53d | ||
|
|
c50a55f05c | ||
|
|
1c146ff523 | ||
|
|
00d3b1ca49 | ||
|
|
0f13bc76f5 | ||
|
|
0c2f087bb9 | ||
|
|
b377eac8cd | ||
|
|
175a4302c3 | ||
|
|
e31f662f8b | ||
|
|
903ddbf97d | ||
|
|
9c44c45a7e | ||
|
|
4da44357d0 | ||
|
|
bf55c1cc46 | ||
|
|
96d188ac12 | ||
|
|
c0a1e91237 | ||
|
|
788c1a15da | ||
|
|
bb72f2c0c6 | ||
|
|
edeb7ab0f9 | ||
|
|
55f954e6c3 | ||
|
|
587ef0ef47 | ||
|
|
23a18a46b3 | ||
|
|
8e00d7c191 | ||
|
|
b242fde7ec | ||
|
|
a09c9ef266 | ||
|
|
a2e5ffc9db | ||
|
|
6bebe4caf9 | ||
|
|
97e554fb13 | ||
|
|
3a30286567 | ||
|
|
bdc919f084 | ||
|
|
8cc6821cad | ||
|
|
f2851bbb61 | ||
|
|
49e19c2414 | ||
|
|
73cbdc75e1 | ||
|
|
81b91619ae | ||
|
|
070ab14951 | ||
|
|
bbbb50ff5b | ||
|
|
204c9c60d1 | ||
|
|
e7f8898546 | ||
|
|
9c64303464 | ||
|
|
28bf478b68 | ||
|
|
9b0edad337 | ||
|
|
d6cbd76c46 | ||
|
|
2c8a7548cb | ||
|
|
33ce5f1765 | ||
|
|
84ce97a114 | ||
|
|
c508781257 | ||
|
|
df994a4890 | ||
|
|
db63f71f40 | ||
|
|
c4b05bd83e | ||
|
|
6155e148a7 | ||
|
|
b8e0bd0692 | ||
|
|
675ecb994f | ||
|
|
e3d5c35614 | ||
|
|
953a063626 | ||
|
|
535e39bbb1 | ||
|
|
cfa41e0d9c | ||
|
|
0ee13cc107 | ||
|
|
4635107f2b | ||
|
|
e5c1b4412c | ||
|
|
a3cd66cf84 | ||
|
|
7abddc527a | ||
|
|
6afec618be | ||
|
|
fff5a50f21 | ||
|
|
257ab6837b | ||
|
|
e8b229bb54 | ||
|
|
f7db5cbc24 | ||
|
|
1a1df8de48 | ||
|
|
3f7fddb188 | ||
|
|
94ccc4f364 | ||
|
|
d86da2d2c5 | ||
|
|
46d81f4742 | ||
|
|
81786e6671 | ||
|
|
70a1fd7679 | ||
|
|
feefc7d62a | ||
|
|
3061b78689 | ||
|
|
a4f4e00ba0 | ||
|
|
ba938e8a97 | ||
|
|
1cc174cd7f | ||
|
|
db840eb015 | ||
|
|
77d32658fa | ||
|
|
d55cacbd6e | ||
|
|
c42e6e7687 | ||
|
|
a96d4cdf33 | ||
|
|
ec402f6b24 | ||
|
|
2a6f96bb90 | ||
|
|
6398fa198b | ||
|
|
f1f1b420b7 | ||
|
|
ad989ab425 | ||
|
|
7caa15acde | ||
|
|
04009fedb0 | ||
|
|
098db1b70c |
@@ -19,9 +19,15 @@ CMakeFiles/
|
||||
# Clangd server cache
|
||||
*.cache*
|
||||
|
||||
#vscode settings
|
||||
/.vscode/
|
||||
|
||||
# Backup files
|
||||
*~
|
||||
|
||||
# clangd index
|
||||
/.cache/
|
||||
|
||||
# Default install location
|
||||
/mfem/
|
||||
|
||||
@@ -79,6 +85,7 @@ examples/sol_u.*
|
||||
examples/sol_p.*
|
||||
examples/sol_r.*
|
||||
examples/sol_i.*
|
||||
examples/sol_z.*
|
||||
examples/ex6p-checkpoint.*
|
||||
examples/order.*
|
||||
examples/ex9.mesh
|
||||
|
||||
+529
-70
@@ -9,91 +9,550 @@
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# DESCRIPTION:
|
||||
###############################################################################
|
||||
# General GitLab pipelines configurations for supercomputers and Linux clusters
|
||||
# at Lawrence Livermore National Laboratory (LLNL). This entire pipeline is
|
||||
# LLNL-specific!
|
||||
|
||||
include:
|
||||
- project: 'lc-templates/id_tokens'
|
||||
file: 'id_tokens.yml'
|
||||
|
||||
# The pipeline is divided into stages. Usually, jobs in a given stage wait for
|
||||
# the preceding stages to complete before to start. However, we sometimes use
|
||||
# the "needs" keyword and express the DAG of jobs for more efficiency.
|
||||
# - We use setup and setup_baseline phases to download content outside of mfem
|
||||
# directory.
|
||||
# - Allocate/Release is where Dane resource are allocated/released once for all.
|
||||
# - Build and Test is where we build and MFEM for multiple toolchains.
|
||||
# - Baseline_checks gathers baseline-type test suites execution
|
||||
# - Baseline_publish, only available on master, allows to update baseline
|
||||
# results
|
||||
stages:
|
||||
- sub-pipelines
|
||||
# at Lawrence Livermore National Laboratory (LLNL).
|
||||
# This entire pipeline is LLNL-specific
|
||||
#
|
||||
# Important note: This file is a template provided by llnl/radiuss-shared-ci.
|
||||
# Remains to set variable values, change the reference to the radiuss-shared-ci
|
||||
# repo, opt-in and out optional features. The project can then extend it with
|
||||
# additional stages.
|
||||
#
|
||||
# In addition, each project should copy over and complete:
|
||||
# - .gitlab/custom-jobs-and-variables.yml
|
||||
# - .gitlab/subscribed-pipelines.yml
|
||||
#
|
||||
# The jobs should be specified in a file local to the project,
|
||||
# - .gitlab/jobs/${CI_MACHINE}.yml
|
||||
# or generated (see LLNL/Umpire for an example).
|
||||
###############################################################################
|
||||
# MAP OF GITLAB CI
|
||||
#######################
|
||||
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
# File dependencies: direct, through jobs, through variables
|
||||
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
# .gitlab-ci.yml
|
||||
# ├── .build-and-test [job]
|
||||
# │ ├── .gitlab/custom-jobs-and-variables.yml
|
||||
# │ │ ├── .custom_job [job]
|
||||
# │ │ ├── .reproducer_vars [job]
|
||||
# │ │ ├── .report_job_success [job]
|
||||
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
|
||||
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
|
||||
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
|
||||
# │ │ ├── .report_job_failure [job]
|
||||
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
|
||||
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
|
||||
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
|
||||
# │ │ └── JOB_CMD [var]
|
||||
# │ │ └── tests/gitlab/build_and_test [script]
|
||||
# │ │ └── tests/gitlab/get_mfem_uberenv [script]
|
||||
# │ ├── <radiuss-shared-ci>/pipelines/matrix.yml [conditional]
|
||||
# │ │ ├── .on_matrix [job]
|
||||
# │ │ ├── .matrix_reproducer_init [job]
|
||||
# │ │ ├── .matrix_reproducer_vars [job]
|
||||
# │ │ ├── .matrix_reproducer_job [job]
|
||||
# │ │ ├── .matrix_job_command [job]
|
||||
# │ │ └── .job_on_matrix [job]
|
||||
# │ ├── <radiuss-shared-ci>/pipelines/dane.yml [conditional]
|
||||
# │ │ ├── .on_dane [job]
|
||||
# │ │ ├── .dane_reproducer_init [job]
|
||||
# │ │ ├── .dane_reproducer_vars [job]
|
||||
# │ │ ├── .dane_reproducer_job [job]
|
||||
# │ │ ├── .dane_job_command [job]
|
||||
# │ │ ├── .job_on_dane [job]
|
||||
# │ │ ├── allocate_resources [job]
|
||||
# │ │ └── release_resources [job]
|
||||
# │ ├── <radiuss-shared-ci>/pipelines/tioga.yml [conditional]
|
||||
# │ │ ├── .on_tioga [job]
|
||||
# │ │ ├── .tioga_reproducer_init [job]
|
||||
# │ │ ├── .tioga_reproducer_vars [job]
|
||||
# │ │ ├── .tioga_reproducer_job [job]
|
||||
# │ │ ├── .tioga_job_command [job]
|
||||
# │ │ ├── .job_on_tioga [job]
|
||||
# │ │ ├── allocate_resources [job]
|
||||
# │ │ └── release_resources [job]
|
||||
# │ ├── <artifact>/matrix-jobs.yml [conditional, from 'generate-job-lists']
|
||||
# │ │ ├── .gitlab/jobs/matrix.yml
|
||||
# │ │ │ ├── .matrix_reproducer_vars [job]
|
||||
# │ │ │ ├── setup [job]
|
||||
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
|
||||
# │ │ │ ├── opt_mpi_cuda_gcc [job]
|
||||
# │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc [job]
|
||||
# │ │ └── .gitlab/jobs/matrix-reports.yml [used conditionally]
|
||||
# │ │ ├── report_job_success
|
||||
# │ │ └── report_job_failure
|
||||
# │ ├── <artifact>/dane-jobs.yml [conditional, from 'generate-job-lists']
|
||||
# │ │ ├── .gitlab/jobs/dane.yml
|
||||
# │ │ │ ├── .dane_reproducer_vars [job]
|
||||
# │ │ │ ├── setup [job]
|
||||
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
|
||||
# │ │ │ ├── debug_ser_gcc_10 [job]
|
||||
# │ │ │ ├── debug_par_gcc_10 [job]
|
||||
# │ │ │ ├── opt_ser_gcc_10 [job]
|
||||
# │ │ │ ├── opt_par_gcc_10 [job]
|
||||
# │ │ │ ├── opt_par_gcc_10_sundials [job]
|
||||
# │ │ │ ├── opt_par_gcc_10_petsc [job]
|
||||
# │ │ │ └── opt_par_gcc_10_pumi [job]
|
||||
# │ │ └── .gitlab/jobs/dane-reports.yml [used conditionally]
|
||||
# │ │ ├── report_job_success
|
||||
# │ │ └── report_job_failure
|
||||
# │ └── <artifact>/tioga-jobs.yml [conditional, from 'generate-job-lists']
|
||||
# │ ├── .gitlab/jobs/tioga.yml
|
||||
# │ │ ├── .tioga_reproducer_vars [job]
|
||||
# │ │ ├── setup [job]
|
||||
# │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
|
||||
# │ │ └── cce_16_0_1 [job]
|
||||
# │ └── .gitlab/jobs/tioga-reports.yml [used conditionally]
|
||||
# │ ├── report_job_success
|
||||
# │ └── report_job_failure
|
||||
# └── .gitlab/subscribed-pipelines.yml
|
||||
# ├── .machine-check [job]
|
||||
# ├── generate-job-lists [job]
|
||||
# ├── dane-up-check [job]
|
||||
# ├── dane-build-and-test [job]
|
||||
# ├── dane-baseline [job]
|
||||
# │ └── .gitlab/dane-baseline.yml
|
||||
# │ ├── .on_dane [job]
|
||||
# │ ├── baselinecheck_mfem_intel_dane [job]
|
||||
# │ │ └── .gitlab/scripts/baseline [script]
|
||||
# │ ├── cleanup [job]
|
||||
# │ ├── report_baseline [job]
|
||||
# │ │ ├── .gitlab/scripts/safe_create_rundir [script]
|
||||
# │ │ └── .gitlab/scripts/git_try_to_push [script]
|
||||
# │ ├── baselinepublish_mfem_dane [job]
|
||||
# │ │ └── .gitlab/scripts/rebaseline [script]
|
||||
# │ ├── .gitlab/custom-jobs-and-variables.yml
|
||||
# │ │ └── <same as above: see .gitlab-ci.yml/.build-and-test>
|
||||
# │ └── .gitlab/configs/setup-baseline.yml
|
||||
# │ └── setup_baseline [job]
|
||||
# ├── tioga-up-check [job]
|
||||
# ├── tioga-build-and-test [job]
|
||||
# ├── matrix-up-check [job]
|
||||
# └── matrix-build-and-test [job]
|
||||
#
|
||||
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
# File tree hierarchy with file contents highlights
|
||||
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
# In addition to the files in the MFEM repo, the Gitlab CI uses files from the
|
||||
# radiuss/radiuss-shared-ci project, see below, after the <mfem-root> tree.
|
||||
#
|
||||
# <mfem root>
|
||||
# ├── .gitlab-ci.yml [this file]
|
||||
# │ ├── <jobs>
|
||||
# │ │ └── .build-and-test
|
||||
# │ ├── <included files>
|
||||
# │ │ ├── .gitlab/subscribed-pipelines.yml
|
||||
# │ │ ├── .gitlab/custom-jobs-and-variables.yml [by ".build-and-test"]
|
||||
# │ │ ├── <artifact> [by ".build-and-test"]
|
||||
# │ │ │ ├── artifact: '${CI_MACHINE}-jobs.yml'
|
||||
# │ │ │ └── job: 'generate-job-lists'
|
||||
# │ │ └── <external> [by ".build-and-test"]
|
||||
# │ │ ├── project: 'radiuss/radiuss-shared-ci'
|
||||
# │ │ ├── ref: 'v2025.09.1'
|
||||
# │ │ └── file: 'pipelines/${CI_MACHINE}.yml'
|
||||
# │ └── <defined variables>
|
||||
# │ ├── CUSTOM_CI_BUILDS_DIR
|
||||
# │ ├── USER_CI_TOP_DIR
|
||||
# │ ├── SHARED_REPOS_DIR
|
||||
# │ ├── AUTOTEST_ROOT
|
||||
# │ ├── MFEM_DATA_DIR
|
||||
# │ ├── AUTOTEST
|
||||
# │ ├── AUTOTEST_COMMIT
|
||||
# │ ├── REBASELINE
|
||||
# │ ├── GITHUB_PROJECT_NAME
|
||||
# │ └── GITHUB_PROJECT_ORG
|
||||
# ├── .gitlab
|
||||
# │ ├── configs
|
||||
# │ │ └── setup-baseline.yml
|
||||
# │ │ ├── <jobs>
|
||||
# │ │ │ └── setup_baseline
|
||||
# │ │ └── <used variables>
|
||||
# │ │ ├── MACHINE_NAME
|
||||
# │ │ ├── REBASELINE
|
||||
# │ │ ├── AUTOTEST
|
||||
# │ │ ├── AUTOTEST_COMMIT
|
||||
# │ │ ├── BUILD_ROOT
|
||||
# │ │ ├── TPLS_REPO
|
||||
# │ │ ├── TESTS_REPO
|
||||
# │ │ ├── AUTOTEST_ROOT
|
||||
# │ │ └── AUTOTEST_REPO
|
||||
# │ ├── jobs
|
||||
# │ │ ├── matrix-reports.yml
|
||||
# │ │ │ ├── <jobs>
|
||||
# │ │ │ │ ├── report_job_success
|
||||
# │ │ │ │ └── report_job_failure
|
||||
# │ │ │ └── <used jobs>
|
||||
# │ │ │ ├── .on_matrix
|
||||
# │ │ │ ├── .report_job_success
|
||||
# │ │ │ └── .report_job_failure
|
||||
# │ │ ├── matrix.yml
|
||||
# │ │ │ ├── <jobs>
|
||||
# │ │ │ │ ├── .matrix_reproducer_vars
|
||||
# │ │ │ │ ├── setup
|
||||
# │ │ │ │ ├── opt_mpi_cuda_gcc
|
||||
# │ │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc
|
||||
# │ │ │ ├── <used jobs>
|
||||
# │ │ │ │ ├── .reproducer_vars
|
||||
# │ │ │ │ ├── .on_matrix
|
||||
# │ │ │ │ └── .job_on_matrix
|
||||
# │ │ │ ├── <included and used files>
|
||||
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
|
||||
# │ │ │ └── <defined variables>
|
||||
# │ │ │ └── SPEC
|
||||
# │ │ ├── dane-reports.yml
|
||||
# │ │ │ ├── <jobs>
|
||||
# │ │ │ │ ├── report_job_success
|
||||
# │ │ │ │ └── report_job_failure
|
||||
# │ │ │ └── <used jobs>
|
||||
# │ │ │ ├── .on_dane
|
||||
# │ │ │ ├── .report_job_success
|
||||
# │ │ │ └── .report_job_failure
|
||||
# │ │ ├── dane.yml
|
||||
# │ │ │ ├── <jobs>
|
||||
# │ │ │ │ ├── .dane_reproducer_vars
|
||||
# │ │ │ │ ├── setup
|
||||
# │ │ │ │ ├── debug_ser_gcc_10
|
||||
# │ │ │ │ ├── debug_par_gcc_10
|
||||
# │ │ │ │ ├── opt_ser_gcc_10
|
||||
# │ │ │ │ ├── opt_par_gcc_10
|
||||
# │ │ │ │ ├── opt_par_gcc_10_sundials
|
||||
# │ │ │ │ ├── opt_par_gcc_10_petsc
|
||||
# │ │ │ │ └── opt_par_gcc_10_pumi
|
||||
# │ │ │ ├── <used jobs>
|
||||
# │ │ │ │ ├── .reproducer_vars
|
||||
# │ │ │ │ ├── .on_dane
|
||||
# │ │ │ │ └── .job_on_dane
|
||||
# │ │ │ ├── <included and used files>
|
||||
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
|
||||
# │ │ │ └── <defined variables>
|
||||
# │ │ │ ├── SPEC
|
||||
# │ │ │ └── THREADS
|
||||
# │ │ ├── tioga-reports.yml
|
||||
# │ │ │ ├── <jobs>
|
||||
# │ │ │ │ ├── report_job_success
|
||||
# │ │ │ │ └── report_job_failure
|
||||
# │ │ │ └── <used jobs>
|
||||
# │ │ │ ├── .on_tioga
|
||||
# │ │ │ ├── .report_job_success
|
||||
# │ │ │ └── .report_job_failure
|
||||
# │ │ └── tioga.yml
|
||||
# │ │ ├── <jobs>
|
||||
# │ │ │ ├── .tioga_reproducer_vars
|
||||
# │ │ │ ├── setup
|
||||
# │ │ │ └── opt_mpi_rocm_hypre_rocm
|
||||
# │ │ ├── <used jobs>
|
||||
# │ │ │ ├── .reproducer_vars
|
||||
# │ │ │ ├── .on_tioga
|
||||
# │ │ │ └── .job_on_tioga
|
||||
# │ │ ├── <included and used files>
|
||||
# │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
|
||||
# │ │ └── <defined variables>
|
||||
# │ │ ├── SPEC
|
||||
# │ │ └── THREADS
|
||||
# │ ├── scripts
|
||||
# │ │ ├── baseline
|
||||
# │ │ │ └── <used variables>
|
||||
# │ │ │ ├── BASELINE_TEST
|
||||
# │ │ │ ├── SYS_TYPE
|
||||
# │ │ │ ├── MACHINE_NAME
|
||||
# │ │ │ ├── CI_PROJECT_DIR
|
||||
# │ │ │ ├── ARTIFACTS_DIR
|
||||
# │ │ │ ├── BUILD_ROOT
|
||||
# │ │ │ └── TPLS_DIR
|
||||
# │ │ ├── git_try_to_push
|
||||
# │ │ ├── rebaseline
|
||||
# │ │ │ └── <used variables>
|
||||
# │ │ │ ├── CI_PROJECT_DIR
|
||||
# │ │ │ ├── ARTIFACTS_DIR
|
||||
# │ │ │ ├── SYS_TYPE
|
||||
# │ │ │ ├── BUILD_ROOT
|
||||
# │ │ │ ├── MACHINE_NAME
|
||||
# │ │ │ └── CI_PIPELINE_ID
|
||||
# │ │ ├── report_build_and_test
|
||||
# │ │ │ ├── <used files>
|
||||
# │ │ │ │ ├── .gitlab/scripts/safe_create_rundir
|
||||
# │ │ │ │ └── .gitlab/scripts/git_try_to_push
|
||||
# │ │ │ └── <used variables>
|
||||
# │ │ │ ├── AUTOTEST_ROOT
|
||||
# │ │ │ ├── CI_COMMIT_REF_SLUG
|
||||
# │ │ │ ├── CI_PROJECT_DIR
|
||||
# │ │ │ ├── CI_PIPELINE_URL
|
||||
# │ │ │ ├── AUTOTEST_COMMIT
|
||||
# │ │ │ └── CI_MACHINE
|
||||
# │ │ └── safe_create_rundir
|
||||
# │ ├── custom-jobs-and-variables.yml
|
||||
# │ │ ├── <jobs>
|
||||
# │ │ │ ├── .custom_job
|
||||
# │ │ │ ├── .reproducer_vars
|
||||
# │ │ │ ├── .report_job_success
|
||||
# │ │ │ └── .report_job_failure
|
||||
# │ │ ├── <used files>
|
||||
# │ │ │ ├── tests/gitlab/build_and_test [in JOB_CMD]
|
||||
# │ │ │ └── .gitlab/scripts/report_build_and_test [by .report_job_*]
|
||||
# │ │ ├── <defined variables>
|
||||
# │ │ │ ├── JOB_CMD
|
||||
# │ │ │ ├── BUILD_ROOT
|
||||
# │ │ │ ├── ALLOC_NAME
|
||||
# │ │ │ ├── TPLS_REPO
|
||||
# │ │ │ ├── TESTS_REPO
|
||||
# │ │ │ ├── AUTOTEST_REPO
|
||||
# │ │ │ ├── MFEM_DATA_REPO
|
||||
# │ │ │ ├── ARTIFACTS_DIR: artifacts
|
||||
# │ │ │ ├── SLURM_OVERLAP: 1
|
||||
# │ │ │ ├── DANE_SHARED_ALLOC
|
||||
# │ │ │ ├── DANE_JOB_ALLOC
|
||||
# │ │ │ ├── TIOGA_SHARED_ALLOC
|
||||
# │ │ │ ├── TIOGA_JOB_ALLOC
|
||||
# │ │ │ └── MATRIX_JOB_ALLOC
|
||||
# │ │ └── <used variables>
|
||||
# │ │ ├── SPEC
|
||||
# │ │ ├── BUILD_ROOT
|
||||
# │ │ └── ...
|
||||
# │ ├── dane-baseline.yml
|
||||
# │ │ ├── <jobs>
|
||||
# │ │ │ ├── .on_dane
|
||||
# │ │ │ ├── baselinecheck_mfem_intel_dane
|
||||
# │ │ │ ├── cleanup
|
||||
# │ │ │ ├── report_baseline
|
||||
# │ │ │ └── baselinepublish_mfem_dane
|
||||
# │ │ ├── <included and used files>
|
||||
# │ │ │ ├── .gitlab/custom-jobs-and-variables.yml
|
||||
# │ │ │ ├── .gitlab/configs/setup-baseline.yml
|
||||
# │ │ │ ├── .gitlab/scripts/rebaseline
|
||||
# │ │ │ ├── .gitlab/scripts/baseline
|
||||
# │ │ │ └── .gitlab/scripts/git_try_to_push
|
||||
# │ │ ├── <defined variables>
|
||||
# │ │ │ ├── BASELINE_TEST: baseline
|
||||
# │ │ │ ├── MACHINE_NAME: dane
|
||||
# │ │ │ ├── TPLS_DIR
|
||||
# │ │ │ └── export MFEM_TEST_NP
|
||||
# │ │ └── <used variables>
|
||||
# │ │ ├── ON_DANE
|
||||
# │ │ ├── AUTOTEST [defined by .gitlab-ci.yml]
|
||||
# │ │ ├── BUILD_ROOT [defined by custom-jobs-and-variables.yml]
|
||||
# │ │ ├── TPLS_DIR [defined by this file]
|
||||
# │ │ ├── ARTIFACTS_DIR [defined by custom-jobs-and-variables.yml]
|
||||
# │ │ ├── MACHINE_NAME [defined by this file]
|
||||
# │ │ ├── AUTOTEST_COMMIT [defined by .gitlab-ci.yml]
|
||||
# │ │ ├── AUTOTEST_ROOT [defined by .gitlab-ci.yml]
|
||||
# │ │ ├── BASELINE_TEST [defined by this file]
|
||||
# │ │ └── REBASELINE [defined by .gitlab-ci.yml]
|
||||
# │ └── subscribed-pipelines.yml
|
||||
# │ ├── <jobs>
|
||||
# │ │ ├── .machine-check
|
||||
# │ │ ├── generate-job-lists
|
||||
# │ │ ├── dane-up-check
|
||||
# │ │ ├── dane-build-and-test
|
||||
# │ │ ├── dane-baseline
|
||||
# │ │ ├── tioga-up-check
|
||||
# │ │ ├── tioga-build-and-test
|
||||
# │ │ ├── matrix-up-check
|
||||
# │ │ └── matrix-build-and-test
|
||||
# │ ├── <used jobs>
|
||||
# │ │ └── .build-and-test [from ".gitlab-ci.yml"]
|
||||
# │ ├── <included files>
|
||||
# │ │ └── .gitlab/dane-baseline.yml [by "dane-baseline"]
|
||||
# │ └── <used variables>
|
||||
# │ ├── GITHUB_PROJECT_ORG
|
||||
# │ ├── GITHUB_PROJECT_NAME
|
||||
# │ ├── AUTOTEST
|
||||
# │ ├── AUTOTEST_COMMIT
|
||||
# │ └── REBASELINE
|
||||
# └── tests
|
||||
# ├── gitlab
|
||||
# │ ├── build_and_test
|
||||
# │ │ ├── <builds and tests a given MFEM spec with uberenv>
|
||||
# │ │ ├── <used files>
|
||||
# │ │ │ ├── tests/uberenv/uberenv.py [deps mode, cloned]
|
||||
# │ │ │ └── tests/gitlab/get_mfem_uberenv [deps mode]
|
||||
# │ │ └── <used variables>
|
||||
# │ │ ├── SYS_TYPE
|
||||
# │ │ ├── THREADS [num. parallel jobs to build MFEM]
|
||||
# │ │ ├── MODULE_LIST [modules to load]
|
||||
# │ │ ├── CI_JOB_ID
|
||||
# │ │ ├── USE_DEV_SHM
|
||||
# │ │ ├── SPACK_DEBUG
|
||||
# │ │ ├── DEBUG_MODE
|
||||
# │ │ ├── REGISTRY_TOKEN
|
||||
# │ │ ├── CI_REGISTRY_USER (defined by Gitlab)
|
||||
# │ │ ├── USER
|
||||
# │ │ ├── CI_REGISTRY_IMAGE (defined by Gitlab)
|
||||
# │ │ └── CI_JOB_TOKEN (defined by Gitlab)
|
||||
# │ ├── build_and_test_setup
|
||||
# │ │ ├── <updates MFEM_DATA_REPO and AUTOTEST_REPO using locks>
|
||||
# │ │ └── <used variables>
|
||||
# │ │ ├── MFEM_DATA_REPO
|
||||
# │ │ ├── SHARED_REPOS_DIR
|
||||
# │ │ ├── AUTOTEST_REPO
|
||||
# │ │ └── AUTOTEST_ROOT
|
||||
# │ └── get_mfem_uberenv
|
||||
# │ ├── <github.com/mfem/mfem-uberenv.git -> tests/uberenv>
|
||||
# │ └── <defines the uberenv hash to use>
|
||||
# └── uberenv [cloned by tests/gitlab/get_mfem_uberenv]
|
||||
# └── uberenv.py
|
||||
#
|
||||
# <root of radiuss/radiuss-shared-ci, ref: 'v2025.09.1'>
|
||||
# └── pipelines
|
||||
# ├── matrix.yml
|
||||
# │ ├── <jobs>
|
||||
# │ │ ├── .on_matrix
|
||||
# │ │ ├── .matrix_reproducer_init
|
||||
# │ │ ├── .matrix_reproducer_vars
|
||||
# │ │ ├── .matrix_reproducer_job
|
||||
# │ │ ├── .matrix_job_command
|
||||
# │ │ └── .job_on_matrix
|
||||
# │ ├── <used jobs>
|
||||
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
|
||||
# │ └── <used variables>
|
||||
# │ ├── ON_MATRIX
|
||||
# │ ├── ADVANCED_JOB
|
||||
# │ ├── ALL_TARGETS
|
||||
# │ ├── SYS_TYPE
|
||||
# │ ├── LLNL_SERVICE_USER
|
||||
# │ ├── USER
|
||||
# │ ├── GITHUB_PROJECT_NAME
|
||||
# │ ├── GITHUB_PROJECT_ORG
|
||||
# │ ├── MATRIX_JOB_ALLOC
|
||||
# │ └── JOB_CMD
|
||||
# ├── dane.yml
|
||||
# │ ├── <jobs>
|
||||
# │ │ ├── .on_dane
|
||||
# │ │ ├── .dane_reproducer_init
|
||||
# │ │ ├── .dane_reproducer_vars
|
||||
# │ │ ├── .dane_reproducer_job
|
||||
# │ │ ├── .dane_job_command
|
||||
# │ │ ├── .job_on_dane
|
||||
# │ │ ├── allocate_resources
|
||||
# │ │ └── release_resources
|
||||
# │ ├── <used jobs>
|
||||
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
|
||||
# │ ├── <defined variables>
|
||||
# │ │ └── export JOBID
|
||||
# │ └── <used variables>
|
||||
# │ ├── ON_DANE
|
||||
# │ ├── ADVANCED_JOB
|
||||
# │ ├── ALL_TARGETS
|
||||
# │ ├── SYS_TYPE
|
||||
# │ ├── LLNL_SERVICE_USER
|
||||
# │ ├── USER
|
||||
# │ ├── GITHUB_PROJECT_NAME
|
||||
# │ ├── GITHUB_PROJECT_ORG
|
||||
# │ ├── DANE_JOB_ALLOC
|
||||
# │ ├── JOB_CMD
|
||||
# │ ├── JOBID
|
||||
# │ ├── ALLOC_NAME
|
||||
# │ └── DANE_SHARED_ALLOC
|
||||
# └── tioga.yml
|
||||
# ├── <jobs>
|
||||
# │ ├── .on_tioga
|
||||
# │ ├── .tioga_reproducer_init
|
||||
# │ ├── .tioga_reproducer_vars
|
||||
# │ ├── .tioga_reproducer_job
|
||||
# │ ├── .tioga_job_command
|
||||
# │ ├── .job_on_tioga
|
||||
# │ ├── allocate_resources
|
||||
# │ └── release_resources
|
||||
# ├── <used jobs>
|
||||
# │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
|
||||
# ├── <defined variables>
|
||||
# │ └── PROXY
|
||||
# └── <used variables>
|
||||
# ├── ON_TIOGA
|
||||
# ├── ADVANCED_JOB
|
||||
# ├── ALL_TARGETS
|
||||
# ├── SYS_TYPE
|
||||
# ├── LLNL_SERVICE_USER
|
||||
# ├── USER
|
||||
# ├── GITHUB_PROJECT_NAME
|
||||
# ├── GITHUB_PROJECT_ORG
|
||||
# ├── TIOGA_JOB_ALLOC
|
||||
# ├── JOB_CMD
|
||||
# ├── PROXY
|
||||
# ├── ALLOC_NAME
|
||||
# └── TIOGA_SHARED_ALLOC
|
||||
|
||||
###############################################################################
|
||||
# We define the following GitLab pipeline variables:
|
||||
variables:
|
||||
##### LC GITLAB CONFIGURATION
|
||||
CUSTOM_CI_BUILDS_DIR: "/usr/workspace/mfem/gitlab-runner"
|
||||
|
||||
##### PROJECT VARIABLES
|
||||
USER_CI_TOP_DIR: "${CUSTOM_CI_BUILDS_DIR}/${GITLAB_USER_LOGIN}"
|
||||
SHARED_REPOS_DIR: "${USER_CI_TOP_DIR}/repos"
|
||||
AUTOTEST_ROOT: "${SHARED_REPOS_DIR}"
|
||||
# MFEM_DATA_DIR is setup in '.gitlab/configs/setup-build-and-test.yml' and
|
||||
# used in '.gitlab/configs/<machine>-config.yml':
|
||||
MFEM_DATA_DIR: "${SHARED_REPOS_DIR}/mfem-data"
|
||||
|
||||
# AUTOTEST: enable (ON/YES) or disable (any other value) test reporting. See
|
||||
# also AUTOTEST_COMMIT.
|
||||
AUTOTEST: "OFF"
|
||||
# AUTOTEST_COMMIT: used only when AUTOTEST is set to ON/YES.
|
||||
# * If AUTOTEST_COMMIT is set to ON/YES, reporting jobs will commit their
|
||||
# files to the MFEM/autotest repo.
|
||||
# * If AUTOTEST_COMMIT is NOT set to ON/YES, reporting jobs will NOT commit
|
||||
# their files to the MFEM/autotest repo. Instead they will just show the
|
||||
# contents of the report files and remove them.
|
||||
AUTOTEST_COMMIT: "ON"
|
||||
# REBASELINE:
|
||||
# Defines the default choice for updating the saved baseline results. By default
|
||||
# the baseline can only be updated from the master branch. This variable offers
|
||||
# the option to manually ask for rebaselining from another branch if necessary.
|
||||
REBASELINE: "NO"
|
||||
AUTOTEST: "NO"
|
||||
# AUTOTEST_COMMIT: used only when AUTOTEST is set to YES.
|
||||
# * If AUTOTEST_COMMIT is NOT set to NO, reporting jobs will commit their
|
||||
# files to the MFEM/autotest repo.
|
||||
# * If AUTOTEST_COMMIT is set to NO, reporting jobs will NOT commit their
|
||||
# files to the MFEM/autotest repo. Instead they will just show the contents
|
||||
# of the report files and remove them.
|
||||
AUTOTEST_COMMIT: "YES"
|
||||
REBASELINE: "OFF"
|
||||
|
||||
# Trigger subpipelines:
|
||||
dane-build-and-test:
|
||||
stage: sub-pipelines
|
||||
##### SHARED_CI CONFIGURATION
|
||||
# Required information about GitHub repository
|
||||
GITHUB_PROJECT_NAME: "mfem"
|
||||
GITHUB_PROJECT_ORG: "MFEM"
|
||||
# Override the pattern describing branches that will skip the "draft PR filter
|
||||
# test". Add protected branches here. See default value in
|
||||
# preliminary-ignore-draft-pr.yml.
|
||||
# ALWAYS_RUN_PATTERN: ""
|
||||
|
||||
###############################################################################
|
||||
##### High level stages
|
||||
# We organize the test-pipelines stage with sub-pipelines. Each sub-pipeline
|
||||
# corresponds to a test batch on a given machine.
|
||||
stages:
|
||||
- prerequisites
|
||||
- test-pipelines
|
||||
|
||||
###############################################################################
|
||||
# Template for jobs triggering a build-and-test sub-pipeline:
|
||||
.build-and-test:
|
||||
stage: test-pipelines
|
||||
variables:
|
||||
# Explicitly pass down values that we want to be able to set when triggering
|
||||
# pipelines manually or using scheduling
|
||||
# Explicitly pass down values that are not always propagated to child
|
||||
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
|
||||
# interface (project variables).
|
||||
# Note: in some cases, this does not work as expected, e.g. when the
|
||||
# variable is not re-defined in the web interface; in such cases, the child
|
||||
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
|
||||
# AUTOTEST is undefined, even though there is a default value in
|
||||
# .gitlab-ci.yml.
|
||||
AUTOTEST: "${AUTOTEST}"
|
||||
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
|
||||
trigger:
|
||||
include: .gitlab/dane-build-and-test.yml
|
||||
include:
|
||||
- local: '.gitlab/custom-jobs-and-variables.yml'
|
||||
- project: 'radiuss/radiuss-shared-ci'
|
||||
ref: 'v2025.09.1'
|
||||
file: 'pipelines/${CI_MACHINE}.yml'
|
||||
- artifact: '${CI_MACHINE}-jobs.yml'
|
||||
job: 'generate-job-lists'
|
||||
strategy: depend
|
||||
forward:
|
||||
pipeline_variables: true
|
||||
|
||||
dane-baseline:
|
||||
stage: sub-pipelines
|
||||
variables:
|
||||
# Explicitly pass down values that we want to be able to set when triggering
|
||||
# pipelines manually or using scheduling
|
||||
REBASELINE: "${REBASELINE}"
|
||||
AUTOTEST: "${AUTOTEST}"
|
||||
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
|
||||
trigger:
|
||||
include: .gitlab/dane-baseline.yml
|
||||
strategy: depend
|
||||
|
||||
lassen-build-and-test:
|
||||
stage: sub-pipelines
|
||||
variables:
|
||||
# Explicitly pass down values that we want to be able to set when triggering
|
||||
# pipelines manually or using scheduling
|
||||
AUTOTEST: "${AUTOTEST}"
|
||||
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
|
||||
trigger:
|
||||
include: .gitlab/lassen-build-and-test.yml
|
||||
strategy: depend
|
||||
|
||||
corona-build-and-test:
|
||||
stage: sub-pipelines
|
||||
variables:
|
||||
# Explicitly pass down values that we want to be able to set when triggering
|
||||
# pipelines manually or using scheduling
|
||||
AUTOTEST: "${AUTOTEST}"
|
||||
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
|
||||
trigger:
|
||||
include: .gitlab/corona-build-and-test.yml
|
||||
strategy: depend
|
||||
###############################################################################
|
||||
include:
|
||||
# Sets ID tokens for every job using `default:`
|
||||
- project: 'lc-templates/id_tokens'
|
||||
file: 'id_tokens.yml'
|
||||
# [Optional] checks preliminary to running the actual CI test
|
||||
#- project: 'radiuss/radiuss-shared-ci'
|
||||
# ref: 'v2025.09.1'
|
||||
# file: 'preliminary-ignore-draft-pr.yml'
|
||||
# pipelines subscribed by the project
|
||||
- local: '.gitlab/subscribed-pipelines.yml'
|
||||
|
||||
+36
-12
@@ -8,6 +8,8 @@
|
||||
https://mfem.org
|
||||
|
||||
|
||||
FIXME: this file needs to be updated
|
||||
|
||||
This directory contains most of the GitLab CI configuration. MFEM runs both PR
|
||||
and nightly testing on GitLab.
|
||||
|
||||
@@ -15,9 +17,9 @@ and nightly testing on GitLab.
|
||||
|
||||
## Top level
|
||||
|
||||
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo.
|
||||
This file only defines one stage, in which we trigger several
|
||||
sub-pipelines.
|
||||
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo. This
|
||||
file only defines three stages, a prerequisites one, and two main stages in
|
||||
which we trigger several sub-pipelines.
|
||||
|
||||
We use sub-pipelines to isolate the test for one combination of `machine`
|
||||
and `test type`.
|
||||
@@ -25,8 +27,8 @@ and `test type`.
|
||||
Machines typically include:
|
||||
|
||||
* Dane: Intel Sapphire Rapids
|
||||
* Lassen: Power9 + Nvidia GPU
|
||||
* Corona: AMD GPU
|
||||
* Matrix: Intel Sapphire Rapids + Nvidia H100 GPU
|
||||
* Tioga: AMD MI250X GPU
|
||||
|
||||
Test types include:
|
||||
|
||||
@@ -39,9 +41,31 @@ altering the scheduling, execution and displaying of the others.
|
||||
|
||||
## Sub-pipelines
|
||||
|
||||
Each file is this directory is the root configuration file for one
|
||||
sub-pipeline. The naming reflects the corresponding couple (`machine`,
|
||||
`test_type`).
|
||||
### build-and-test
|
||||
|
||||
The build-and-test sub-pipelines leverage RADIUSS Shared CI to share most of
|
||||
the CI implementation. RADIUSS Shared CI provides a shared CI infrastructure
|
||||
vetted on most LC systems of interest and efficiently leveraging each machine
|
||||
scheduler to increase CI throughput. The maintenance of RADIUSS Shared CI is
|
||||
shared among several RADIUSS projects.
|
||||
|
||||
Jobs for the build-and-test sub-pipelines are defined in the jobs directory.
|
||||
Because build-and-test jobs leverage Uberenv and Spack to build the
|
||||
dependencies automatically, the jobs essentially consists in a `spack spec`
|
||||
defined in the jobs files, and some scheduling parameters defined in the
|
||||
`.gitlab/custom-jobs-and-variables.yml` file.
|
||||
|
||||
Build-and-test jobs all run the `tests/gitlab/build_and_test` script.
|
||||
|
||||
The build-and-test pipelines are controlled by the
|
||||
`.gitlab/subscribed-pipelines.yml` which defines which machines to run on and
|
||||
implements additional features like machine availability check, and job list
|
||||
generation.
|
||||
|
||||
### baseline
|
||||
|
||||
Baseline sub-pipelines are described by files with names reflecting the
|
||||
machine it runs on, e.g. `dane-baseline`.
|
||||
|
||||
Those files define the *stages* and the *jobs* for the sub-pipeline. They
|
||||
also contain any configuration that cannot be shared. For the most part
|
||||
@@ -63,11 +87,11 @@ usage function. This should be improved.
|
||||
|
||||
# More testing
|
||||
|
||||
## Adding a new target to a build_and_test pipeline
|
||||
## Adding a new target to a build-and-test pipeline
|
||||
|
||||
`build_and_test` pipelines rely on Spack to install dependencies. Spack is
|
||||
`build-and-test` pipelines rely on Spack to install dependencies. Spack is
|
||||
driven by Uberenv which helps freezing Spack configuration: the goal being to
|
||||
point to specific commit in Spack and isolate its configuration so that it is
|
||||
point to a specific commit in Spack and isolate its configuration so that it is
|
||||
not influenced by the user environment. More documentation about this can be
|
||||
found in `tests/gitlab`.
|
||||
|
||||
@@ -82,7 +106,7 @@ spack spec to use. Adding a job on Dane for example resumes to:
|
||||
<job_name>:
|
||||
variables:
|
||||
SPEC: "<spack_spec>"
|
||||
extends: .build_and_test_on_dane
|
||||
extends: .job_on_dane
|
||||
```
|
||||
|
||||
The remaining and non trivial work is to make sure this spec is working. To
|
||||
|
||||
@@ -1,40 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
include:
|
||||
- project: 'lc-templates/id_tokens'
|
||||
file: 'id_tokens.yml'
|
||||
|
||||
# We define the following GitLab pipeline variables:
|
||||
variables:
|
||||
|
||||
# The path to the shared resource between all jobs. For example, external
|
||||
# repositories like 'tests' and 'tpls' are cloned here. Also, 'tpls' is built
|
||||
# once for all targets, so that build happen here. The BUILD_ROOT is unique to
|
||||
# the pipeline, preventing any form of concurrency with other pipelines. This
|
||||
# also means that the BUILD_ROOT directory will never be cleaned.
|
||||
# TODO: add a clean-up mechanism
|
||||
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${MACHINE_NAME}-pipeline-${CI_PIPELINE_ID}
|
||||
|
||||
# On LLNL's Dane, there is only one allocation shared among jobs in order to
|
||||
# save time and resource. This allocation has to be uniquely named so that we
|
||||
# are sure to retrieve it.
|
||||
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
|
||||
|
||||
# Git repositories used in the pipeline
|
||||
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
|
||||
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
|
||||
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
|
||||
MFEM_DATA_REPO: https://github.com/mfem/data.git
|
||||
|
||||
# Directory used to place artifacts.
|
||||
ARTIFACTS_DIR: artifacts
|
||||
SLURM_OVERLAP: 1
|
||||
@@ -1,59 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# GitLab pipeline configuration for the Corona machine at LLNL
|
||||
variables:
|
||||
MACHINE_NAME: corona
|
||||
|
||||
.on_corona:
|
||||
tags:
|
||||
- shell
|
||||
- corona
|
||||
rules:
|
||||
# Don't run corona jobs if...
|
||||
# Note: This makes corona an "opt-in" machine. To activate builds on corona
|
||||
# for a given GitLab clone of MFEM, go to Setting/CI-CD/variables, and set
|
||||
# "ON_CORONA" to "ON". An LC account on for corona is required to trigger a
|
||||
# pipeline there.
|
||||
- if: '$CI_COMMIT_BRANCH =~ /_cnone/ || $ON_CORONA != "ON"'
|
||||
when: never
|
||||
# Don't run autotest update if...
|
||||
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Report success on success status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
|
||||
when: on_success
|
||||
# Report failure on failure status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
|
||||
when: on_failure
|
||||
# Always release resource
|
||||
- if: '$CI_JOB_NAME =~ /release_resource/'
|
||||
when: always
|
||||
# Always cleanup
|
||||
- if: '$CI_JOB_NAME =~ /cleanup/'
|
||||
when: always
|
||||
# Default is to run if previous stage succeeded
|
||||
- when: on_success
|
||||
|
||||
# Spack helped builds
|
||||
# Generic corona build job, extending build script
|
||||
.build_and_test_on_corona:
|
||||
extends: [.on_corona]
|
||||
stage: build_and_test
|
||||
script:
|
||||
# THREADS is used by 'tests/gitlab/build_and_test', run below
|
||||
- export THREADS=12
|
||||
- echo ${ALLOC_NAME}
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- echo ${JOBID}
|
||||
- echo ${MFEM_DATA_DIR}
|
||||
- echo ${SPEC}
|
||||
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) -t 15 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
|
||||
@@ -1,56 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# GitLab pipelines configurations for the Dane machine at LLNL
|
||||
variables:
|
||||
MACHINE_NAME: dane
|
||||
|
||||
.on_dane:
|
||||
tags:
|
||||
- shell
|
||||
- dane
|
||||
rules:
|
||||
# Don't run dane jobs if...
|
||||
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_DANE == "OFF"'
|
||||
when: never
|
||||
# Don't run autotest update if...
|
||||
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Report success on success status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
|
||||
when: on_success
|
||||
# Report failure on failure status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
|
||||
when: on_failure
|
||||
# Always release resource
|
||||
- if: '$CI_JOB_NAME =~ /release_resource/'
|
||||
when: always
|
||||
# Always cleanup
|
||||
- if: '$CI_JOB_NAME =~ /cleanup/'
|
||||
when: always
|
||||
# Default is to run if previous stage succeeded
|
||||
- when: on_success
|
||||
|
||||
# Spack helped builds
|
||||
# Generic dane build job, extending build script
|
||||
.build_and_test_on_dane:
|
||||
extends: [.on_dane]
|
||||
stage: build_and_test
|
||||
script:
|
||||
# THREADS is used by 'tests/gitlab/build_and_test', run below
|
||||
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
|
||||
- export THREADS=28
|
||||
- echo ${ALLOC_NAME}
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- echo ${JOBID}
|
||||
- echo ${MFEM_DATA_DIR}
|
||||
- echo ${SPEC}
|
||||
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) --reservation=ci -t 60 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
|
||||
@@ -1,48 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# GitLab pipelines configurations for the Lassen machine at LLNL
|
||||
variables:
|
||||
MACHINE_NAME: lassen
|
||||
|
||||
.on_lassen:
|
||||
tags:
|
||||
- shell
|
||||
- lassen
|
||||
rules:
|
||||
- if: '$CI_COMMIT_BRANCH =~ /_lnone/ || $ON_LASSEN == "OFF"' #run except if ...
|
||||
when: never
|
||||
# Don't run autotest update if...
|
||||
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Report success on success status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
|
||||
when: on_success
|
||||
# Report failure on failure status
|
||||
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
|
||||
when: on_failure
|
||||
# Always cleanup
|
||||
- if: '$CI_JOB_NAME =~ /cleanup/'
|
||||
when: always
|
||||
- when: on_success
|
||||
|
||||
# Lassen uses a different job scheduler (spectrum lsf) that does not allow
|
||||
# pre-allocation the same way slurm does. We use the pci queue on lassen
|
||||
# to speed-up the allocation.
|
||||
.build_and_test_on_lassen:
|
||||
extends: [.on_lassen]
|
||||
stage: build_and_test
|
||||
script:
|
||||
- echo ${MFEM_DATA_DIR}
|
||||
- echo ${SPEC}
|
||||
# Next script uses 'THREADS': leaving it empty --> it uses 'make all -j'
|
||||
- lalloc 1 -W 45 -q pci --atsdisable tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
|
||||
needs: [setup]
|
||||
@@ -1,77 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Jobs report
|
||||
.report_job_success:
|
||||
script:
|
||||
- echo ${MACHINE_NAME}
|
||||
- echo ${AUTOTEST}
|
||||
- echo ${AUTOTEST_COMMIT}
|
||||
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
|
||||
- cd ${AUTOTEST_ROOT}
|
||||
- |
|
||||
(
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
# Report SUCCESS while holding the file lock on 'autotest.lock'.
|
||||
# The next script uses the following environment variables:
|
||||
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
|
||||
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
|
||||
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
|
||||
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_success
|
||||
err=$?
|
||||
# sleep for a period to allow NFS to propagate the above changes;
|
||||
# clearly, there is no guarantee that other NFS clients will see the
|
||||
# changes even after the timeout
|
||||
sleep 10
|
||||
exit $err
|
||||
) 9> autotest.lock
|
||||
|
||||
.report_job_failure:
|
||||
script:
|
||||
- echo ${MACHINE_NAME}
|
||||
- echo ${AUTOTEST}
|
||||
- echo ${AUTOTEST_COMMIT}
|
||||
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
|
||||
- cd ${AUTOTEST_ROOT}
|
||||
- |
|
||||
(
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
# Report FAILURE while holding the file lock on 'autotest.lock'.
|
||||
# The next script uses the following environment variables:
|
||||
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
|
||||
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
|
||||
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
|
||||
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_failure
|
||||
err=$?
|
||||
# sleep for a period to allow NFS to propagate the above changes;
|
||||
# clearly, there is no guarantee that other NFS clients will see the
|
||||
# changes even after the timeout
|
||||
sleep 10
|
||||
exit $err
|
||||
) 9> autotest.lock
|
||||
@@ -1,90 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
|
||||
# script then symlinks the repo to the parent directory of the MFEM source
|
||||
# directory. Unit tests that depend on the mfem/data repo will then detect that
|
||||
# this directory is present and be enabled.
|
||||
setup:
|
||||
tags:
|
||||
- shell
|
||||
- dane
|
||||
stage: setup
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
script:
|
||||
#
|
||||
# Setup MFEM_DATA_DIR=${SHARED_REPOS_DIR}/mfem-data, see '.gitlab-ci.yml'
|
||||
# and '.gitlab/configs/<machine>-config.yml'
|
||||
#
|
||||
- echo "MACHINE_NAME = ${MACHINE_NAME}"
|
||||
- echo "AUTOTEST = ${AUTOTEST}"
|
||||
- echo "AUTOTEST_COMMIT = ${AUTOTEST_COMMIT}"
|
||||
- echo "SHARED_REPOS_DIR ${SHARED_REPOS_DIR}"
|
||||
- mkdir -p ${SHARED_REPOS_DIR} && cd ${SHARED_REPOS_DIR}
|
||||
- command -v flock || echo "Required command 'flock' not found"
|
||||
- |
|
||||
(
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/mfem-data.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
|
||||
# try every 5 seconds; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
done
|
||||
echo "Acquired lock on '$PWD/mfem-data.lock'"
|
||||
date
|
||||
# clone/update the mfem/data repo while holding the file lock on
|
||||
# 'mfem-data.lock'
|
||||
err=0
|
||||
if [[ ! -d "mfem-data" ]]; then
|
||||
git clone ${MFEM_DATA_REPO} "mfem-data"
|
||||
else
|
||||
cd "mfem-data" && git pull && cd ..
|
||||
fi || err=1
|
||||
# sleep for a period to allow NFS to propagate the above changes;
|
||||
# clearly, there is no guarantee that other NFS clients will see the
|
||||
# changes even after the timeout
|
||||
sleep 10
|
||||
exit $err
|
||||
) 9> mfem-data.lock
|
||||
#
|
||||
# Setup ${AUTOTEST_ROOT}/autotest:
|
||||
#
|
||||
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
|
||||
- mkdir -p ${AUTOTEST_ROOT} && cd ${AUTOTEST_ROOT}
|
||||
- |
|
||||
(
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
# clone/update the autotest repo while holding the file lock on
|
||||
# 'autotest.lock'
|
||||
err=0
|
||||
if [[ ! -d "autotest" ]]; then
|
||||
git clone ${AUTOTEST_REPO}
|
||||
else
|
||||
cd autotest && git pull && cd ..
|
||||
fi || err=1
|
||||
# sleep for a period to allow NFS to propagate the above changes;
|
||||
# clearly, there is no guarantee that other NFS clients will see the
|
||||
# changes even after the timeout
|
||||
sleep 10
|
||||
exit $err
|
||||
) 9> autotest.lock
|
||||
@@ -1,67 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
stages:
|
||||
- setup
|
||||
- allocate_resource
|
||||
- build_and_test
|
||||
- release_resource_and_report
|
||||
|
||||
# Slurm shared allocation
|
||||
allocate_resource:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_corona
|
||||
stage: allocate_resource
|
||||
script:
|
||||
- echo ${ALLOC_NAME}
|
||||
- salloc --exclusive --nodes=1 --partition=mi60 --time=45 --no-shell --job-name=${ALLOC_NAME}
|
||||
timeout: 6h
|
||||
needs: [setup]
|
||||
|
||||
# Build and test jobs, simply provide a spec
|
||||
rocm_gcc_8.3.1:
|
||||
variables:
|
||||
SPEC: "@develop%gcc@8.3.1+rocm amdgpu_target=gfx906"
|
||||
extends: .build_and_test_on_corona
|
||||
needs: [allocate_resource]
|
||||
|
||||
# Release slurm allocation
|
||||
release_resource:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_corona
|
||||
stage: release_resource_and_report
|
||||
script:
|
||||
- echo ${ALLOC_NAME}
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- echo ${JOBID}
|
||||
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
|
||||
needs: [rocm_gcc_8.3.1]
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
stage: release_resource_and_report
|
||||
extends:
|
||||
- .on_corona
|
||||
- .report_job_success
|
||||
|
||||
report_job_failure:
|
||||
stage: release_resource_and_report
|
||||
extends:
|
||||
- .on_corona
|
||||
- .report_job_failure
|
||||
|
||||
include:
|
||||
- local: .gitlab/configs/common.yml
|
||||
- local: .gitlab/configs/corona-config.yml
|
||||
- local: .gitlab/configs/setup-build-and-test.yml
|
||||
- local: .gitlab/configs/report-build-and-test.yml
|
||||
@@ -0,0 +1,132 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
include:
|
||||
- project: 'lc-templates/id_tokens'
|
||||
file: 'id_tokens.yml'
|
||||
|
||||
# We define the following GitLab pipeline variables:
|
||||
variables:
|
||||
# Set the build-and-test command.
|
||||
# Nested variables are allowed and useful to customize the job command. We
|
||||
# protect variables with quotes so that their value may remain a string even if
|
||||
# they contain whitespaces.
|
||||
JOB_CMD:
|
||||
value: tests/gitlab/build_and_test --spec \"${SPEC}\" --data-dir ${MFEM_DATA_DIR} --data
|
||||
# The path to the shared resource between all jobs in the 'dane-baseline'
|
||||
# pipeline. For example, external repositories like 'tests' and 'tpls' are
|
||||
# cloned here. Also, 'tpls' is built once for all targets, so that build happens
|
||||
# here. The BUILD_ROOT is unique to the pipeline, preventing any form of
|
||||
# concurrency with other pipelines. This directory is removed by the 'cleanup'
|
||||
# stage in the 'dane-baseline' pipeline.
|
||||
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${CI_MACHINE}-pipeline-${CI_PIPELINE_ID}
|
||||
|
||||
# On LLNL's dane and tioga, the 'build-and-test' pipelines creates only one
|
||||
# allocation shared among jobs in the pipeline in order to save time and
|
||||
# resources. This allocation has to be uniquely named so that we are sure to
|
||||
# retrieve it and avoid collisions.
|
||||
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
|
||||
|
||||
# Git repositories used in the pipelines:
|
||||
# - TPLS_REPO and TESTS_REPO are used only by the 'dane-baseline' pipeline
|
||||
# - AUTOTEST_REPO is used by all pipelines
|
||||
# - MFEM_DATA_REPO is used only by the 'build-and-test' pipelines
|
||||
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
|
||||
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
|
||||
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
|
||||
MFEM_DATA_REPO: https://github.com/mfem/data.git
|
||||
|
||||
# Directory used to place artifacts:
|
||||
# - ARTIFACTS_DIR is only used by the 'dane-baseline' pipeline
|
||||
ARTIFACTS_DIR: artifacts
|
||||
SLURM_OVERLAP: 1
|
||||
|
||||
# Dane
|
||||
# Arguments for top level allocation
|
||||
DANE_SHARED_ALLOC: "--exclusive --reservation=ci --time=60 --nodes=1"
|
||||
# Arguments for job level allocation
|
||||
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
|
||||
DANE_JOB_ALLOC: "--reservation=ci --overlap --nodes=1"
|
||||
|
||||
# Tioga
|
||||
# Arguments for top level allocation
|
||||
TIOGA_SHARED_ALLOC: "--queue=pci --exclusive --time-limit=45m --nodes=1"
|
||||
# Arguments for job level allocation
|
||||
TIOGA_JOB_ALLOC: "--nodes=1 --begin-time=+5s"
|
||||
|
||||
# Matrix
|
||||
# Arguments for top level allocation
|
||||
MATRIX_SHARED_ALLOC: "-p pdebug --exclusive --time=45 --nodes=1 -G 4"
|
||||
# Arguments for job level allocation
|
||||
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
|
||||
MATRIX_JOB_ALLOC: "--overlap --nodes=1"
|
||||
|
||||
# Configuration shared by build and test jobs specific to this project.
|
||||
# Not all configuration can be shared. Here projects can fine tune the
|
||||
# CI behavior.
|
||||
# See Umpire for an example (export junit test reports).
|
||||
.custom_job:
|
||||
artifacts:
|
||||
reports:
|
||||
|
||||
# Note: this part is not used by the 'dane-baseline' pipeline.
|
||||
# FIXME: BUILD_ROOT, TPLS_REPO, TESTS_REPO are not needed here.
|
||||
# Also, the definition of SHARED_REPOS_DIR is wrong.
|
||||
.reproducer_vars:
|
||||
script:
|
||||
- |
|
||||
echo -e "
|
||||
# Variables \n
|
||||
export SPEC=\"${SPEC//\"/\\\"}\" \n
|
||||
# Directories \n
|
||||
export BUILD_ROOT=\"\${working_dir}\" \n
|
||||
export SHARED_REPOS_DIR=\"\${BUILD_ROOT}/..\" \n
|
||||
export MFEM_DATA_DIR=\"\${SHARED_REPOS_DIR}/mfem-data\" \n
|
||||
# Repositories \n
|
||||
export TPLS_REPO=\"${TPLS_REPO//\"/\\\"}\" \n
|
||||
export TESTS_REPO=\"${TESTS_REPO//\"/\\\"}\" \n
|
||||
export AUTOTEST_REPO=\"${AUTOTEST_REPO//\"/\\\"}\" \n
|
||||
export MFEM_DATA_REPO=\"${MFEM_DATA_REPO//\"/\\\"}\" \n
|
||||
# Setup directories \n
|
||||
./tests/gitlab/build_and_test_setup \n
|
||||
# Using the CI build cache is optional and requires a token. Set it like so: \n
|
||||
# export REGISTRY_TOKEN=\"<your token here>\" \n"
|
||||
#
|
||||
|
||||
# Jobs report
|
||||
.report_job_success:
|
||||
script:
|
||||
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test SUCCESS
|
||||
rules:
|
||||
- when: on_success
|
||||
|
||||
.report_job_failure:
|
||||
script:
|
||||
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test FAILURE
|
||||
rules:
|
||||
- when: on_failure
|
||||
|
||||
# Keep the following for debugging purposes: renaming this job from
|
||||
# '.show_variables' to 'show_variables' will insert this debug job at the
|
||||
# beginning of all child pipelines.
|
||||
.show_variables:
|
||||
tags: [shell, oslic]
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
stage: .pre
|
||||
script:
|
||||
- |
|
||||
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
|
||||
echo "AUTOTEST=${AUTOTEST}"
|
||||
echo "AUTOTEST_COMMIT=${AUTOTEST_COMMIT}"
|
||||
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
|
||||
# Fail the job on purpose to prevent the rest of the pipeline from running
|
||||
false
|
||||
@@ -11,6 +11,7 @@
|
||||
|
||||
variables:
|
||||
BASELINE_TEST: baseline
|
||||
MACHINE_NAME: dane
|
||||
|
||||
stages:
|
||||
- setup
|
||||
@@ -19,6 +20,25 @@ stages:
|
||||
- cleanup
|
||||
- baseline_publish
|
||||
|
||||
.on_dane:
|
||||
tags:
|
||||
- shell
|
||||
- dane
|
||||
rules:
|
||||
# Don't run dane jobs if...
|
||||
- if: '$ON_DANE == "OFF"'
|
||||
when: never
|
||||
# Don't run autotest update if...
|
||||
# Note: in some cases, the content of AUTOTEST can be '${AUTOTEST}', so we
|
||||
# need to treat that value as the default value of 'OFF'.
|
||||
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "ON" && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Always cleanup
|
||||
- if: '$CI_JOB_NAME =~ /cleanup/'
|
||||
when: always
|
||||
# Default is to run if previous stage succeeded
|
||||
- when: on_success
|
||||
|
||||
baselinecheck_mfem_intel_dane:
|
||||
extends: [.on_dane]
|
||||
stage: baseline_check
|
||||
@@ -29,6 +49,9 @@ baselinecheck_mfem_intel_dane:
|
||||
# .gitlab/configs/setup-baseline.yml.
|
||||
TPLS_DIR: ${BUILD_ROOT}/tpls
|
||||
script:
|
||||
- echo "AUTOTEST=$AUTOTEST"
|
||||
- echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
|
||||
- echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
|
||||
- echo ${BUILD_ROOT}
|
||||
- echo ${TPLS_DIR}
|
||||
# Used by the tests in MFEM/tests, dane has 224 threads/node:
|
||||
@@ -89,7 +112,13 @@ report_baseline:
|
||||
cp ${rundir}/pipeline.txt ${rundir}/autotest-email.html
|
||||
fi
|
||||
msg="GitLab CI log for ${BASELINE_TEST} on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
|
||||
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
|
||||
# Note: in some cases, the content of AUTOTEST_COMMIT can be
|
||||
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
|
||||
# value of 'ON'.
|
||||
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
|
||||
AUTOTEST_COMMIT="ON"
|
||||
fi
|
||||
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
|
||||
git pull && \
|
||||
git add ${rundir} && \
|
||||
git commit -m "${msg}" && \
|
||||
@@ -117,8 +146,8 @@ baselinepublish_mfem_dane:
|
||||
extends: [.on_dane]
|
||||
stage: baseline_publish
|
||||
rules:
|
||||
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "YES"'
|
||||
- if: '$REBASELINE == "YES"'
|
||||
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "ON"'
|
||||
- if: '$REBASELINE == "ON"'
|
||||
when: manual
|
||||
script:
|
||||
- echo ${BUILD_ROOT}
|
||||
@@ -128,6 +157,5 @@ baselinepublish_mfem_dane:
|
||||
- .gitlab/scripts/rebaseline
|
||||
|
||||
include:
|
||||
- local: .gitlab/configs/common.yml
|
||||
- local: .gitlab/configs/dane-config.yml
|
||||
- local: .gitlab/custom-jobs-and-variables.yml
|
||||
- local: .gitlab/configs/setup-baseline.yml
|
||||
|
||||
@@ -1,94 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
stages:
|
||||
- setup
|
||||
- allocate_resource
|
||||
- build_and_test
|
||||
- release_resource_and_report
|
||||
|
||||
# Allocate
|
||||
allocate_resource:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_dane
|
||||
stage: allocate_resource
|
||||
script:
|
||||
- echo ${ALLOC_NAME}
|
||||
- salloc --exclusive --nodes=1 --reservation=ci --time=60 --no-shell --job-name=${ALLOC_NAME}
|
||||
timeout: 6h
|
||||
|
||||
# GitLab jobs for the Dane machine at LLNL
|
||||
debug_ser_gcc_10:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug~mpi"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
debug_par_gcc_10:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug+mpi"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
opt_ser_gcc_10:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 ~mpi"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
opt_par_gcc_10:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
opt_par_gcc_10_sundials:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +sundials"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
opt_par_gcc_10_petsc:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
opt_par_gcc_10_pumi:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +pumi"
|
||||
extends: .build_and_test_on_dane
|
||||
|
||||
# Release
|
||||
release_resource:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_dane
|
||||
stage: release_resource_and_report
|
||||
script:
|
||||
- echo ${ALLOC_NAME}
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- echo ${JOBID}
|
||||
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
stage: release_resource_and_report
|
||||
extends:
|
||||
- .on_dane
|
||||
- .report_job_success
|
||||
|
||||
report_job_failure:
|
||||
stage: release_resource_and_report
|
||||
extends:
|
||||
- .on_dane
|
||||
- .report_job_failure
|
||||
|
||||
include:
|
||||
- local: .gitlab/configs/common.yml
|
||||
- local: .gitlab/configs/dane-config.yml
|
||||
- local: .gitlab/configs/setup-build-and-test.yml
|
||||
- local: .gitlab/configs/report-build-and-test.yml
|
||||
@@ -0,0 +1,19 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
extends: [.on_dane, .report_job_success]
|
||||
stage: jobs-stage-3
|
||||
|
||||
report_job_failure:
|
||||
extends: [.on_dane, .report_job_failure]
|
||||
stage: jobs-stage-3
|
||||
@@ -0,0 +1,87 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Override reproducer section to define MFEM specific variables.
|
||||
.dane_reproducer_vars:
|
||||
script:
|
||||
- !reference [.reproducer_vars, script]
|
||||
|
||||
# TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY
|
||||
# cannot be "none" anymore).
|
||||
|
||||
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
|
||||
# script then symlinks the repo to the parent directory of the MFEM source
|
||||
# directory. Unit tests that depend on the mfem/data repo will then detect that
|
||||
# this directory is present and be enabled.
|
||||
setup:
|
||||
extends: .on_dane
|
||||
stage: jobs-stage-1
|
||||
script:
|
||||
- ./tests/gitlab/build_and_test_setup
|
||||
|
||||
|
||||
########################
|
||||
# Overridden shared jobs
|
||||
########################
|
||||
# When using shared jobs, we can duplicate them here to override description and
|
||||
# add necessary changes.
|
||||
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
|
||||
# the comparison with the original job is easier.
|
||||
|
||||
|
||||
############
|
||||
# Extra jobs
|
||||
############
|
||||
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
|
||||
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
|
||||
# describe the spec here.
|
||||
|
||||
.mfem_job_on_dane:
|
||||
extends: .job_on_dane
|
||||
stage: jobs-stage-2
|
||||
variables:
|
||||
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
|
||||
THREADS: 28
|
||||
|
||||
debug_ser_gcc_10:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug~mpi"
|
||||
|
||||
debug_par_gcc_10:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug+mpi"
|
||||
|
||||
opt_ser_gcc_10:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 ~mpi"
|
||||
|
||||
opt_par_gcc_10:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1"
|
||||
|
||||
opt_par_gcc_10_sundials:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +sundials"
|
||||
|
||||
opt_par_gcc_10_petsc:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
|
||||
|
||||
opt_par_gcc_10_pumi:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +pumi"
|
||||
@@ -0,0 +1,19 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
extends: [.on_matrix, .report_job_success]
|
||||
stage: jobs-stage-3
|
||||
|
||||
report_job_failure:
|
||||
extends: [.on_matrix, .report_job_failure]
|
||||
stage: jobs-stage-3
|
||||
@@ -0,0 +1,65 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Override reproducer section to define UMPIRE specific variables.
|
||||
.matrix_reproducer_vars:
|
||||
script:
|
||||
- !reference [.reproducer_vars, script]
|
||||
|
||||
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
|
||||
|
||||
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
|
||||
# script then symlinks the repo to the parent directory of the MFEM source
|
||||
# directory. Unit tests that depend on the mfem/data repo will then detect that
|
||||
# this directory is present and be enabled.
|
||||
setup:
|
||||
extends: .on_matrix
|
||||
stage: jobs-stage-1
|
||||
script:
|
||||
- ./tests/gitlab/build_and_test_setup
|
||||
|
||||
|
||||
########################
|
||||
# Overridden shared jobs
|
||||
########################
|
||||
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
|
||||
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
|
||||
# the comparison with the original job is easier.
|
||||
|
||||
|
||||
############
|
||||
# Extra jobs
|
||||
############
|
||||
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
|
||||
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
|
||||
# describe the spec here.
|
||||
|
||||
.mfem_job_on_matrix:
|
||||
extends: .job_on_matrix
|
||||
stage: jobs-stage-2
|
||||
variables:
|
||||
# We run 2 jobs on 1 node that has 112 threads
|
||||
THREADS: 48
|
||||
# These modules need to be consistent with the uberenv configurations:
|
||||
MODULE_LIST: "gcc/10.3.1-magic cuda/12.9.1"
|
||||
|
||||
allocate_resources:
|
||||
timeout: 4h
|
||||
|
||||
opt_mpi_cuda_gcc:
|
||||
extends: .mfem_job_on_matrix
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90"
|
||||
|
||||
opt_mpi_cuda_hypre_cuda_gcc:
|
||||
extends: .mfem_job_on_matrix
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
|
||||
@@ -0,0 +1,20 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
extends: [.on_tioga, .report_job_success]
|
||||
stage: jobs-stage-3
|
||||
|
||||
report_job_failure:
|
||||
extends: [.on_tioga, .report_job_failure]
|
||||
stage: jobs-stage-3
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Override reproducer section to define UMPIRE specific variables.
|
||||
.tioga_reproducer_vars:
|
||||
script:
|
||||
- !reference [.reproducer_vars, script]
|
||||
|
||||
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
|
||||
|
||||
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
|
||||
# script then symlinks the repo to the parent directory of the MFEM source
|
||||
# directory. Unit tests that depend on the mfem/data repo will then detect that
|
||||
# this directory is present and be enabled.
|
||||
setup:
|
||||
extends: .on_tioga
|
||||
stage: jobs-stage-1
|
||||
script:
|
||||
- ./tests/gitlab/build_and_test_setup
|
||||
|
||||
########################
|
||||
# Overridden shared jobs
|
||||
########################
|
||||
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
|
||||
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
|
||||
# the comparison with the original job is easier.
|
||||
|
||||
|
||||
############
|
||||
# Extra jobs
|
||||
############
|
||||
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
|
||||
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
|
||||
# describe the spec here.
|
||||
|
||||
# Build and test jobs, simply provide a spec
|
||||
|
||||
#.tioga_job_command:
|
||||
# script:
|
||||
# - echo PROXY="${PROXY}"
|
||||
# - echo TIOGA_JOB_ALLOC="${TIOGA_JOB_ALLOC}"
|
||||
# - "printf '#!/bin/bash\n%s\n' \"${JOB_CMD}\" > flux_script.sh"
|
||||
# - cat flux_script.sh
|
||||
# - ${PROXY} flux watch $( ${PROXY} flux batch -o output.stdout.type=kvs ${TIOGA_JOB_ALLOC} flux_script.sh )
|
||||
# - rm -f flux_script.sh
|
||||
|
||||
.mfem_job_on_tioga:
|
||||
extends: .job_on_tioga
|
||||
stage: jobs-stage-2
|
||||
variables:
|
||||
# We run 1 job on 1 node that has 64 threads
|
||||
THREADS: 64
|
||||
|
||||
opt_mpi_rocm_hypre_rocm:
|
||||
extends: .mfem_job_on_tioga
|
||||
variables:
|
||||
SPEC: "%rocmcc@=6.3.1 +rocm amdgpu_target=gfx90a ^hypre+rocm"
|
||||
|
||||
# cce_16_0_1:
|
||||
# extends: .mfem_job_on_tioga
|
||||
# variables:
|
||||
# SPEC: "%cce@=16.0.1"
|
||||
@@ -1,44 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
stages:
|
||||
- setup
|
||||
- build_and_test
|
||||
- report
|
||||
|
||||
opt_mpi_cuda_gcc:
|
||||
variables:
|
||||
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70"
|
||||
extends: .build_and_test_on_lassen
|
||||
|
||||
opt_mpi_cuda_hypre_cuda_gcc:
|
||||
variables:
|
||||
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70 ^hypre+cuda~shared cuda_arch=70"
|
||||
extends: .build_and_test_on_lassen
|
||||
|
||||
# Jobs report
|
||||
report_job_success:
|
||||
stage: report
|
||||
extends:
|
||||
- .on_lassen
|
||||
- .report_job_success
|
||||
|
||||
report_job_failure:
|
||||
stage: report
|
||||
extends:
|
||||
- .on_lassen
|
||||
- .report_job_failure
|
||||
|
||||
include:
|
||||
- local: .gitlab/configs/common.yml
|
||||
- local: .gitlab/configs/lassen-config.yml
|
||||
- local: .gitlab/configs/setup-build-and-test.yml
|
||||
- local: .gitlab/configs/report-build-and-test.yml
|
||||
@@ -35,8 +35,6 @@ if [[ "${MACHINE_NAME}" == "dane" ]]; then
|
||||
salloc --nodes=1 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
elif [[ ${MACHINE_NAME} == "corona" ]]; then
|
||||
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
elif [[ ${MACHINE_NAME} == "lassen" ]]; then
|
||||
lalloc 1 -q pci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
else
|
||||
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
|
||||
exit 1
|
||||
|
||||
Executable
+118
@@ -0,0 +1,118 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
function info_msg ()
|
||||
{
|
||||
echo "[Information:] ${1}"
|
||||
}
|
||||
|
||||
function error_msg ()
|
||||
{
|
||||
echo "[Error:] ${1}"
|
||||
}
|
||||
|
||||
# Perform a report while holding a lock file to prevent concurrency on
|
||||
# the destination.
|
||||
# Usage:
|
||||
# locked_clone <report_function> <lock_name>
|
||||
function locked_report ()
|
||||
{
|
||||
if ! command -v flock
|
||||
then
|
||||
error_msg "Required command 'flock' not found"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
info_msg "Will report ${1} while holding a lock in ${2}"
|
||||
|
||||
( date; info_msg "Waiting to acquire lock on '${PWD}/${2}.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
|
||||
# try every 5 seconds; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do sleep 5; done
|
||||
date; info_msg "Acquired lock on '${PWD}/${2}.lock'"
|
||||
|
||||
report ${1}
|
||||
err=$?
|
||||
|
||||
# sleep for a period to allow NFS to propagate the above changes;
|
||||
# clearly, there is no guarantee that other NFS clients will see the
|
||||
# changes even after the timeout
|
||||
sleep 10
|
||||
exit $err
|
||||
) 9> ${2}.lock
|
||||
}
|
||||
|
||||
function report ()
|
||||
{
|
||||
if [[ "${1}" == "SUCCESS" ]]
|
||||
then
|
||||
info_msg "All the ${MACHINE_NAME} jobs passed"
|
||||
status_msg="The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL."
|
||||
elif [[ "${1}" == "FAILURE" ]]
|
||||
then
|
||||
info_msg "At least one failure on ${MACHINE_NAME}"
|
||||
status_msg="Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED."
|
||||
else
|
||||
error_msg "Unknown status: ${1} ... aborting"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cd ${AUTOTEST_ROOT}/autotest || \
|
||||
{ error_msg "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
|
||||
mkdir -p ${MACHINE_NAME}
|
||||
|
||||
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
|
||||
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
|
||||
|
||||
printf "%s\n" "${status_msg}" \
|
||||
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
|
||||
|
||||
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
|
||||
|
||||
if [[ "${1}" == "FAILURE" ]]
|
||||
then
|
||||
# Create 'autotest-email.html' to indicate failure:
|
||||
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
|
||||
fi
|
||||
|
||||
# Note: in some cases, the content of AUTOTEST_COMMIT can be
|
||||
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
|
||||
# value of 'ON'.
|
||||
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
|
||||
AUTOTEST_COMMIT="ON"
|
||||
fi
|
||||
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
|
||||
git pull && \
|
||||
git add ${rundir} && \
|
||||
git commit -m "${msg}" && \
|
||||
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
|
||||
else
|
||||
for file in ${rundir}/*; do
|
||||
echo "------------------------------"
|
||||
echo "Content of '$file'"
|
||||
echo "******************************"
|
||||
cat $file
|
||||
echo "******************************"
|
||||
done
|
||||
rm -rf ${rundir} || true
|
||||
fi
|
||||
}
|
||||
|
||||
export MACHINE_NAME=${CI_MACHINE}
|
||||
info_msg "MACHINE_NAME is ${MACHINE_NAME}"
|
||||
info_msg "AUTOTEST_ROOT is ${AUTOTEST_ROOT}"
|
||||
info_msg "AUTOTEST=$AUTOTEST"
|
||||
info_msg "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
|
||||
|
||||
cd ${AUTOTEST_ROOT} && locked_report ${1} autotest
|
||||
@@ -1,45 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
echo "Runs if there was at least one failure on ${MACHINE_NAME}"
|
||||
|
||||
cd ${AUTOTEST_ROOT}/autotest || \
|
||||
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
|
||||
mkdir -p ${MACHINE_NAME}
|
||||
|
||||
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
|
||||
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
|
||||
|
||||
printf "%s\n" "Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED." \
|
||||
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
|
||||
|
||||
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
|
||||
|
||||
# Create 'autotest-email.html' to indicate failure:
|
||||
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
|
||||
|
||||
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
|
||||
git pull && \
|
||||
git add ${rundir} && \
|
||||
git commit -m "${msg}" && \
|
||||
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
|
||||
else
|
||||
for file in ${rundir}/*; do
|
||||
echo "------------------------------"
|
||||
echo "Content of '$file'"
|
||||
echo "******************************"
|
||||
cat $file
|
||||
echo "******************************"
|
||||
done
|
||||
rm -rf ${rundir} || true
|
||||
fi
|
||||
@@ -1,42 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
echo "Can only run if all the ${MACHINE_NAME} jobs passed"
|
||||
|
||||
cd ${AUTOTEST_ROOT}/autotest || \
|
||||
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
|
||||
mkdir -p ${MACHINE_NAME}
|
||||
|
||||
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
|
||||
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
|
||||
|
||||
printf "%s\n" "The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL." \
|
||||
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.out
|
||||
|
||||
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
|
||||
|
||||
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
|
||||
git pull && \
|
||||
git add ${rundir} && \
|
||||
git commit -m "${msg}" && \
|
||||
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
|
||||
else
|
||||
for file in ${rundir}/*; do
|
||||
echo "------------------------------"
|
||||
echo "Content of '$file'"
|
||||
echo "******************************"
|
||||
cat $file
|
||||
echo "******************************"
|
||||
done
|
||||
rm -rf ${rundir} || true
|
||||
fi
|
||||
@@ -0,0 +1,130 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# The template job to test whether a machine is up.
|
||||
# Expects CI_MACHINE defined to machine name.
|
||||
.machine-check:
|
||||
stage: prerequisites
|
||||
tags: [shell, oslic]
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
script:
|
||||
- |
|
||||
if [[ $(jq '.[env.CI_MACHINE].total_nodes_up' /usr/global/tools/lorenz/data/loginnodeStatus) == 0 ]]
|
||||
then
|
||||
echo -e "\e[31mNo node available on ${CI_MACHINE}\e[0m"
|
||||
false && \
|
||||
curl --url "https://api.github.com/repos/${GITHUB_PROJECT_ORG}/${GITHUB_PROJECT_NAME}/statuses/${CI_COMMIT_SHA}" \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "authorization: Bearer ${GITHUB_TOKEN}" \
|
||||
--data "{ \"state\": \"failure\", \"target_url\": \"${CI_PIPELINE_URL}\", \"description\": \"GitLab ${CI_MACHINE} down\", \"context\": \"ci/gitlab/${CI_MACHINE}\" }"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
###
|
||||
# Trigger a build-and-test pipeline for a machine.
|
||||
# Comment the jobs for machines you don’t need.
|
||||
###
|
||||
|
||||
# One job to generate the job list for all the subpipelines
|
||||
generate-job-lists:
|
||||
stage: prerequisites
|
||||
tags: [shell, oslic]
|
||||
variables:
|
||||
LOCAL_JOBS_PATH: ".gitlab/jobs"
|
||||
script:
|
||||
- |
|
||||
echo "AUTOTEST=$AUTOTEST"
|
||||
echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
|
||||
echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
|
||||
- |
|
||||
cat ${LOCAL_JOBS_PATH}/dane.yml > dane-jobs.yml
|
||||
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
|
||||
then
|
||||
cat ${LOCAL_JOBS_PATH}/dane-reports.yml >> dane-jobs.yml
|
||||
fi
|
||||
- |
|
||||
cat ${LOCAL_JOBS_PATH}/matrix.yml > matrix-jobs.yml
|
||||
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
|
||||
then
|
||||
cat ${LOCAL_JOBS_PATH}/matrix-reports.yml >> matrix-jobs.yml
|
||||
fi
|
||||
- |
|
||||
cat ${LOCAL_JOBS_PATH}/tioga.yml > tioga-jobs.yml
|
||||
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
|
||||
then
|
||||
cat ${LOCAL_JOBS_PATH}/tioga-reports.yml >> tioga-jobs.yml
|
||||
fi
|
||||
artifacts:
|
||||
paths:
|
||||
- dane-jobs.yml
|
||||
- matrix-jobs.yml
|
||||
- tioga-jobs.yml
|
||||
|
||||
|
||||
# DANE
|
||||
dane-up-check:
|
||||
variables:
|
||||
CI_MACHINE: "dane"
|
||||
extends: [.machine-check]
|
||||
|
||||
dane-build-and-test:
|
||||
variables:
|
||||
CI_MACHINE: "dane"
|
||||
needs: [dane-up-check, generate-job-lists]
|
||||
extends: [.build-and-test]
|
||||
|
||||
# DANE, MFEM Specific
|
||||
dane-baseline:
|
||||
stage: test-pipelines
|
||||
variables:
|
||||
# Explicitly pass down values that are not always propagated to child
|
||||
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
|
||||
# interface (project variables).
|
||||
# Note: in some cases, this does not work as expected, e.g. when the
|
||||
# variable is not re-defined in the web interface; in such cases, the child
|
||||
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
|
||||
# AUTOTEST is undefined, even though there is a default value in
|
||||
# .gitlab-ci.yml.
|
||||
AUTOTEST: "${AUTOTEST}"
|
||||
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
|
||||
trigger:
|
||||
include: .gitlab/dane-baseline.yml
|
||||
strategy: depend
|
||||
forward:
|
||||
pipeline_variables: true
|
||||
needs: [dane-up-check]
|
||||
|
||||
|
||||
# TIOGA
|
||||
tioga-up-check:
|
||||
variables:
|
||||
CI_MACHINE: "tioga"
|
||||
extends: [.machine-check]
|
||||
|
||||
tioga-build-and-test:
|
||||
variables:
|
||||
CI_MACHINE: "tioga"
|
||||
needs: [tioga-up-check, generate-job-lists]
|
||||
extends: [.build-and-test]
|
||||
|
||||
|
||||
# Matrix
|
||||
matrix-up-check:
|
||||
variables:
|
||||
CI_MACHINE: "matrix"
|
||||
extends: [.machine-check]
|
||||
|
||||
matrix-build-and-test:
|
||||
variables:
|
||||
CI_MACHINE: "matrix"
|
||||
needs: [matrix-up-check, generate-job-lists]
|
||||
extends: [.build-and-test]
|
||||
@@ -43,6 +43,13 @@ Discretization improvements
|
||||
Meshing improvements
|
||||
--------------------
|
||||
|
||||
- The TMOP kernel hierarchy has been restructured to reduce compilation time.
|
||||
Most large kernels have been split into smaller, specific ones, with kernels
|
||||
for each metric. The directory structure has been updated with assemble,
|
||||
metrics, mult and tools subdirectories. The new kernel dispatch and
|
||||
specialization system has also been integrated.
|
||||
Unit tests have been revised to ensure --all tests pass.
|
||||
|
||||
- Introduced NC-patch NURBS meshes, which are conforming element-wise but allow
|
||||
for nonconforming patch topology. This new mesh format supports element
|
||||
spacing formulas for refinement, as well as local refinement factors for a
|
||||
@@ -76,6 +83,15 @@ GPU computing
|
||||
- Added GPU support in GradientGridFunctionCoefficient and
|
||||
InnerProductCoefficient by implementing their Project methods.
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added `FilteredSolver`: a base class for solvers with filtering. It handles cases
|
||||
where a solver performs well except in small subspaces, by adding a filtering step
|
||||
formulated as a subspace correction.
|
||||
- Added `AMGFSolver`: a derived class of `FilteredSolver`, specialized for
|
||||
AMG with Filtering (AMGF), providing robust preconditioning for linear systems
|
||||
arising in constrained optimization problems such as frictionless contact.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added miniapps to demonstrate an implementation of the absolute-value
|
||||
|
||||
+57
-24
@@ -133,33 +133,49 @@ if (MFEM_USE_CUDA)
|
||||
if (NOT CMAKE_CUDA_HOST_COMPILER)
|
||||
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
|
||||
endif()
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18.0)
|
||||
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
|
||||
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
|
||||
string(REGEX REPLACE "^sm_" "" ARCH_NUMBER "${CUDA_ARCH}")
|
||||
if ("${CUDA_ARCH}" STREQUAL "sm_${ARCH_NUMBER}")
|
||||
set(CMAKE_CUDA_ARCHITECTURES "${ARCH_NUMBER}")
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown CUDA_ARCH: ${CUDA_ARCH}")
|
||||
endif()
|
||||
if (NOT CMAKE_CUDA_ARCHITECTURES)
|
||||
# make CUDA_ARCH resemble the same form as CMAKE_CUDA_ARCHITECTURES
|
||||
string(REPLACE "sm_" "" CUDA_ARCH_TMP "${CUDA_ARCH}")
|
||||
string(REPLACE "," ";" CUDA_ARCH "${CUDA_ARCH_TMP}")
|
||||
set(CMAKE_CUDA_ARCHITECTURES "${CUDA_ARCH}")
|
||||
else()
|
||||
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
|
||||
endif()
|
||||
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
|
||||
enable_language(CUDA)
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18.0)
|
||||
# backup try to detect if this is clang or nvcc
|
||||
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
endif()
|
||||
# backup try to detect if this is clang or nvcc
|
||||
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
if ("all" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
|
||||
OR "native" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
|
||||
OR "all-major" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}")
|
||||
set(CUDA_FLAGS "-arch=${CMAKE_CUDA_ARCHITECTURES} ${CUDA_FLAGS}")
|
||||
else()
|
||||
# build -gencode sequence for multiple architectures
|
||||
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
|
||||
set(CUDA_FLAGS
|
||||
"-gencode arch=compute_${ENTRY},code=sm_${ENTRY} ${CUDA_FLAGS}")
|
||||
endforeach()
|
||||
endif()
|
||||
else()
|
||||
# build cuda-gpu-arch sequence for multiple architectures
|
||||
# does not support all/all-major/native
|
||||
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
|
||||
set(CUDA_FLAGS "-cuda-gpu-arch=sm_${ENTRY} ${CUDA_FLAGS}")
|
||||
endforeach()
|
||||
endif()
|
||||
else()
|
||||
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
endif()
|
||||
# TODO: all, native, all-major require CMake 3.24+
|
||||
# backport support for CMake 3.18 to 3.24
|
||||
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
|
||||
# nvcc
|
||||
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
|
||||
set(CUDA_FLAGS
|
||||
"${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
|
||||
endif()
|
||||
endif()
|
||||
set(CMAKE_CUDA_STANDARD ${CMAKE_CXX_STANDARD} CACHE STRING
|
||||
"CUDA standard to use.")
|
||||
@@ -242,10 +258,16 @@ endif()
|
||||
|
||||
# AMD HIP
|
||||
if (MFEM_USE_HIP)
|
||||
if (HIP_ARCH)
|
||||
message(STATUS "Using HIP architecture: ${HIP_ARCH}")
|
||||
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for")
|
||||
if (NOT CMAKE_HIP_ARCHITECTURES)
|
||||
if (HIP_ARCH)
|
||||
set(CMAKE_HIP_ARCHITECTURES CACHE STRING "HIP targets to compile for" "${HIP_ARCH}")
|
||||
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for" FORCE)
|
||||
endif()
|
||||
else()
|
||||
set(HIP_ARCH CACHE STRING "HIP targets to compile for" "${CMAKE_HIP_ARCHITECTURES}")
|
||||
set(GPU_TARGETS "${CMAKE_HIP_ARCHITECTURES}" CACHE STRING "HIP targets to compile for" FORCE)
|
||||
endif()
|
||||
message(STATUS "Using HIP architecture: ${CMAKE_HIP_ARCHITECTURES}")
|
||||
if (ROCM_PATH)
|
||||
list(INSERT CMAKE_PREFIX_PATH 0 ${ROCM_PATH})
|
||||
endif()
|
||||
@@ -278,8 +300,19 @@ if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Umpire (must be included before hypre, so hypre can use it if needed)
|
||||
# Warn user if deprecated FETCH_TPLS is provided
|
||||
if (DEFINED FETCH_TPLS)
|
||||
message(STATUS "Setting MFEM_FETCH_TPLS to user-provided value of FETCH_TPLS (i.e., MFEM_FETCH_TPLS=${FETCH_TPLS})")
|
||||
set (MFEM_FETCH_TPLS FETCH_TPLS)
|
||||
message(DEPRECATION "The use of FETCH_TPLS is deprecated and will be removed in future verison. Please use MFEM_FETCH_TPLS instead.")
|
||||
endif()
|
||||
|
||||
# Umpire (must be included before hypre, so hypre can use it if needed)
|
||||
if (MFEM_USE_UMPIRE)
|
||||
# umpire uses FindCUDA, which needs CMP0146=OLD in CMake >= 3.27
|
||||
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.27.0)
|
||||
cmake_policy(SET CMP0146 OLD)
|
||||
endif()
|
||||
find_package(UMPIRE REQUIRED)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -123,7 +123,7 @@ Parallel build:
|
||||
|
||||
Parallel build with fetching of hypre and METIS:
|
||||
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DFETCH_TPLS=YES
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
|
||||
make -j 4
|
||||
|
||||
CUDA build:
|
||||
@@ -1081,9 +1081,10 @@ The following options are CMake specific:
|
||||
MFEM_ENABLE_TESTING - Enable the ctest framework for testing.
|
||||
MFEM_ENABLE_EXAMPLES - Build all of the examples by default.
|
||||
MFEM_ENABLE_MINIAPPS - Build all of the miniapps by default.
|
||||
FETCH_TPLS - Enable fetching of all supported third-party libraries.
|
||||
HYPRE_FETCH - Enable fetching of hypre.
|
||||
METIS_FETCH - Enable fetching of metis.
|
||||
MFEM_FETCH_TPLS - Enable fetching of all supported third-party libraries.
|
||||
MFEM_FETCH_GSLIB - Enable fetching of gslib.
|
||||
MFEM_FETCH_HYPRE - Enable fetching of hypre.
|
||||
MFEM_FETCH_METIS - Enable fetching of metis.
|
||||
|
||||
External libraries (CMake):
|
||||
---------------------------
|
||||
@@ -1149,6 +1150,7 @@ The MFEM CMake build system also provides fetching (automated building) for the
|
||||
packages/libraries listed below. Note that when fetching is enabled, any related
|
||||
auto-detection functionality is disabled.
|
||||
|
||||
- GSLIB
|
||||
- HYPRE
|
||||
- METIS
|
||||
|
||||
|
||||
@@ -9,10 +9,47 @@
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Defines the following variables:
|
||||
# Defines the following variables if fetching of TPLs is disabled (default):
|
||||
# - GSLIB_FOUND
|
||||
# - GSLIB_LIBRARIES
|
||||
# - GSLIB_INCLUDE_DIRS
|
||||
# otherwise, the following are defined:
|
||||
# - GSLIB (imported library target)
|
||||
|
||||
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
|
||||
enable_language(C)
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
|
||||
set(GSLIB_FETCH_VERSION 1.0.9)
|
||||
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
|
||||
if (CMAKE_C_FLAGS)
|
||||
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
|
||||
endif()
|
||||
if (BUILD_SHARED_LIBS)
|
||||
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
|
||||
endif()
|
||||
add_library(GSLIB STATIC IMPORTED)
|
||||
# define external project and create future include directory so it is present
|
||||
# to pass CMake checks at end of MFEM configuration step
|
||||
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
|
||||
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
|
||||
include(ExternalProject)
|
||||
ExternalProject_Add(gslib
|
||||
GIT_REPOSITORY https://github.com/Nek5000/gslib
|
||||
GIT_TAG v${GSLIB_FETCH_VERSION}
|
||||
GIT_SHALLOW TRUE
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
PREFIX ${PREFIX}
|
||||
CONFIGURE_COMMAND ""
|
||||
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
|
||||
INSTALL_COMMAND "")
|
||||
file(MAKE_DIRECTORY ${PREFIX}/include)
|
||||
# set imported library target properties
|
||||
add_dependencies(GSLIB gslib)
|
||||
set_target_properties(GSLIB PROPERTIES
|
||||
IMPORTED_LOCATION ${PREFIX}/lib/libgs.a
|
||||
INTERFACE_INCLUDE_DIRECTORIES ${PREFIX}/include)
|
||||
return()
|
||||
endif()
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(GSLIB GSLIB GSLIB_DIR "include" gslib.h "lib" gs
|
||||
|
||||
@@ -37,21 +37,21 @@ if (HYPRE_FOUND OR TARGET HYPRE)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (HYPRE_FETCH OR FETCH_TPLS)
|
||||
# Collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
|
||||
if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
|
||||
set(HYPRE_FETCH_VERSION 2.33.0)
|
||||
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
|
||||
add_library(HYPRE STATIC IMPORTED)
|
||||
# set options and associated dependencies
|
||||
set(HYPRE_CMAKE_OPTIONS "")
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
|
||||
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
|
||||
get_cmake_property(all_vars VARIABLES)
|
||||
foreach(var ${all_vars})
|
||||
if(var MATCHES "^HYPRE_ENABLE")
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS "-D${var}:BOOL=${${var}}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(HYPRE_FETCH_VERSION 2.33.0)
|
||||
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
|
||||
add_library(HYPRE STATIC IMPORTED)
|
||||
# set options and associated dependencies
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
|
||||
# process all MFEM_USE variables that impact hypre
|
||||
if (MFEM_USE_CUDA)
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_CUDA:BOOL=ON -DCMAKE_CUDA_ARCHITECTURES:STRING=${CMAKE_CUDA_ARCHITECTURES})
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
# - METIS (imported library target)
|
||||
# - METIS_VERSION_5 (cache variable)
|
||||
|
||||
if (METIS_FETCH OR FETCH_TPLS)
|
||||
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
|
||||
set(METIS_FETCH_VERSION 4.0.3)
|
||||
add_library(METIS STATIC IMPORTED)
|
||||
# define external project
|
||||
|
||||
@@ -91,9 +91,10 @@ option(MFEM_ENABLE_BENCHMARKS "Build all of the benchmarks" OFF)
|
||||
|
||||
# Allow a user to specify fetching of certain third-party libraries instead of
|
||||
# searching for existing installations.
|
||||
option(FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
|
||||
option(HYPRE_FETCH "Enable fetching of hypre" OFF)
|
||||
option(METIS_FETCH "Enable fetching of METIS" OFF)
|
||||
option(MFEM_FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
|
||||
option(MFEM_FETCH_GSLIB "Enable fetching of GSLIB" OFF)
|
||||
option(MFEM_FETCH_HYPRE "Enable fetching of hypre" OFF)
|
||||
option(MFEM_FETCH_METIS "Enable fetching of METIS" OFF)
|
||||
|
||||
# Setting CXX/MPICXX on the command line or in user.cmake will overwrite the
|
||||
# autodetected C++ compiler.
|
||||
|
||||
@@ -89,6 +89,7 @@ if (MFEM_USE_MPI)
|
||||
ex37p.cpp
|
||||
ex39p.cpp
|
||||
ex40p.cpp
|
||||
ex999p.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -471,10 +471,13 @@ int main(int argc, char *argv[])
|
||||
|
||||
ofstream sol_r_ofs("sol_r.gf");
|
||||
ofstream sol_i_ofs("sol_i.gf");
|
||||
ofstream sol_z_ofs("sol_z.gf");
|
||||
sol_r_ofs.precision(8);
|
||||
sol_i_ofs.precision(8);
|
||||
sol_z_ofs.precision(8);
|
||||
u.real().Save(sol_r_ofs);
|
||||
u.imag().Save(sol_i_ofs);
|
||||
u.Save(sol_z_ofs);
|
||||
}
|
||||
|
||||
// 14. Send the solution by socket to a GLVis server.
|
||||
|
||||
+5
-1
@@ -507,10 +507,11 @@ int main(int argc, char *argv[])
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can be
|
||||
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
{
|
||||
ostringstream mesh_name, sol_r_name, sol_i_name;
|
||||
ostringstream mesh_name, sol_r_name, sol_i_name, sol_z_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
|
||||
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
|
||||
sol_z_name << "sol_z." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
@@ -518,10 +519,13 @@ int main(int argc, char *argv[])
|
||||
|
||||
ofstream sol_r_ofs(sol_r_name.str().c_str());
|
||||
ofstream sol_i_ofs(sol_i_name.str().c_str());
|
||||
ofstream sol_z_ofs(sol_z_name.str().c_str());
|
||||
sol_r_ofs.precision(8);
|
||||
sol_i_ofs.precision(8);
|
||||
sol_z_ofs.precision(8);
|
||||
u.real().Save(sol_r_ofs);
|
||||
u.imag().Save(sol_i_ofs);
|
||||
u.Save(sol_z_ofs);
|
||||
}
|
||||
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
#include <mfem.hpp>
|
||||
#include "nlohmann/json.hpp"
|
||||
#include "minja.hpp"
|
||||
#include "myqfunction.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using namespace mfem::future;
|
||||
|
||||
template<class T>
|
||||
struct remove_cvref
|
||||
{
|
||||
using type = std::remove_cv_t<std::remove_reference_t<T>>;
|
||||
};
|
||||
|
||||
template <typename qf_t>
|
||||
auto process(qf_t qf)
|
||||
{
|
||||
using qfsig = typename create_function_signature<qf_t>::type;
|
||||
using qfpar_t = typename qfsig::parameter_ts;
|
||||
using qfout_t = typename qfsig::return_t;
|
||||
|
||||
auto qfparams = decay_tuple<qfpar_t> {};
|
||||
|
||||
auto in_str = apply([](auto&&... arg)
|
||||
{
|
||||
return std::vector<std::string>
|
||||
{
|
||||
std::string(get_type_name<typename remove_cvref<decltype(arg)>::type>())...
|
||||
};
|
||||
}, qfparams);
|
||||
|
||||
std::vector<std::string> out_str
|
||||
{
|
||||
std::string(get_type_name<typename remove_cvref<qfout_t>::type>())
|
||||
};
|
||||
|
||||
return std::tuple{in_str, out_str};
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
// load the kernel template
|
||||
std::ifstream
|
||||
kernel_istream("/Users/andrej1/repos/mfem/examples/kernel_skeleton.jinja");
|
||||
if (!kernel_istream.is_open())
|
||||
{
|
||||
std::cerr << "error opening jinja template file" << std::endl;
|
||||
return 1;
|
||||
}
|
||||
std::stringstream buffer;
|
||||
buffer << kernel_istream.rdbuf();
|
||||
|
||||
std::string fileContent = buffer.str();
|
||||
auto kernel_tmpl = minja::Parser::parse(buffer.str(), /* options= */ {});
|
||||
|
||||
auto [in_str, out_str] = process(myqfunction0);
|
||||
|
||||
for (auto &v : in_str)
|
||||
{
|
||||
std::cout << v << " ";
|
||||
}
|
||||
std::cout << std::endl;
|
||||
|
||||
const size_t DUMMY_STRIDE = 64*32*32;
|
||||
const size_t basis_p_1d = 2;
|
||||
|
||||
json context_json{};
|
||||
context_json["kernel_name"] = "demo";
|
||||
|
||||
context_json["spaces"].push_back(
|
||||
{
|
||||
{"P_1D", basis_p_1d},
|
||||
{"dim", 3},
|
||||
{"needs_value", true},
|
||||
{"needs_grad", true},
|
||||
});
|
||||
|
||||
context_json["spaces"].push_back(
|
||||
{
|
||||
{"P_1D", basis_p_1d},
|
||||
});
|
||||
|
||||
context_json["inputs"].push_back(
|
||||
{
|
||||
{"name", "potential"},
|
||||
{"space_idx", 0},
|
||||
{"num_comp", 1},
|
||||
{"comp_stride", DUMMY_STRIDE},
|
||||
{"eval_grad", true},
|
||||
});
|
||||
|
||||
context_json["inputs"].push_back(
|
||||
{
|
||||
{"name", "weights"},
|
||||
{"space_idx", 0},
|
||||
{"num_comp", 1},
|
||||
{"comp_stride", DUMMY_STRIDE},
|
||||
{"is_qdata", true},
|
||||
});
|
||||
|
||||
context_json["outputs"].push_back(
|
||||
{
|
||||
{"name", "solution"},
|
||||
{"space_idx", 0},
|
||||
{"num_comp", 1},
|
||||
{"comp_stride", DUMMY_STRIDE},
|
||||
{"eval_grad", true},
|
||||
});
|
||||
|
||||
const size_t nqf = 1;
|
||||
const std::vector<std::string> qfunc_names = {"myqfunction0"};
|
||||
const std::vector<std::vector<size_t>> qfunc_inputs = {{0, 1, 2}};
|
||||
|
||||
for (size_t i = 0; i < nqf; i++)
|
||||
{
|
||||
json inarr = json::array();
|
||||
for (size_t j = 0; j < qfunc_inputs[i].size(); j++)
|
||||
{
|
||||
inarr.push_back(
|
||||
{
|
||||
{"index", j},
|
||||
{"datatype", in_str[j]}
|
||||
});
|
||||
}
|
||||
|
||||
context_json["qfuncs"].push_back(
|
||||
{
|
||||
{"name", qfunc_names[i]},
|
||||
{"inputs", inarr}
|
||||
});
|
||||
}
|
||||
|
||||
std::cout << context_json.dump(2) << std::endl;
|
||||
|
||||
auto context = minja::Context::make(context_json);
|
||||
auto kernel_source = kernel_tmpl->render(context);
|
||||
|
||||
std::cout << ">>> generated kernel source\n"
|
||||
<< kernel_source
|
||||
<< "\n<<< generated kernel source\n"
|
||||
<< std::endl;
|
||||
|
||||
{
|
||||
// test casting
|
||||
std::vector<real_t> d(4);
|
||||
int i = 0;
|
||||
for (auto &v : d)
|
||||
{
|
||||
v = ++i;
|
||||
}
|
||||
|
||||
mfem::future::tensor<real_t, 2, 2> *dudxi =
|
||||
reinterpret_cast<mfem::future::tensor<real_t, 2, 2> *>(d.data());
|
||||
|
||||
std::cout << *dudxi << std::endl;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
#include "util.hpp"
|
||||
|
||||
#define NUM_SPACES {{ spaces | count }}
|
||||
#define NUM_INPUTS {{ inputs | count }}
|
||||
#define NUM_OUTPUTS {{ outputs | count }}
|
||||
|
||||
extern "C" __global__ void dfem_jit_{{kernel_name}}(int num_entities, const real_t *fields[NUM_INPUTS], real_t *outputs[NUM_OUTPUTS], const real_t *B[NUM_SPACES]) {
|
||||
// transform fields
|
||||
const real_t *inputs = ...;
|
||||
|
||||
// call qfunctions
|
||||
{% for qf in qfuncs -%}
|
||||
{
|
||||
{%- for qfinput in qf.inputs %}
|
||||
{{ qfinput.datatype }}* in{{ loop.index0 }} =
|
||||
reinterpret_cast<{{ qfinput.datatype }}>(inputs[{{ qfinput.index }}]);
|
||||
{% endfor %}
|
||||
{{ qf.name }}({% for qfinput in qf.inputs %}*in{{ loop.index0 }}{{ "," if not loop.last else "" }}{% endfor %});
|
||||
}
|
||||
{% endfor %}
|
||||
}
|
||||
+2
-2
@@ -195,8 +195,8 @@ clean-build:
|
||||
clean-exec:
|
||||
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh ex6p-checkpoint.*
|
||||
@rm -rf Example5* Example9* Example15* Example16* Example23* ParaView
|
||||
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* order.*
|
||||
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
|
||||
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* sol_z.*
|
||||
@rm -f order.* ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
|
||||
@rm -f deformed.* velocity.* elastic_energy.* mode_* mode_deriv_* flux.*
|
||||
@rm -f ex5-p-*.bp ex9-p-*.bp ex12-p-*.bp ex16-p-*.bp
|
||||
@rm -f ex16.mesh ex16-mesh.* ex16-init.* ex16-final.*
|
||||
|
||||
+4137
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,17 @@
|
||||
#include <mfem.hpp>
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::future::tensor;
|
||||
|
||||
constexpr int dim = 2;
|
||||
|
||||
tensor<real_t, dim, dim> myqfunction0(
|
||||
const tensor<real_t, dim, dim> &dvdxi,
|
||||
const tensor<real_t, dim, dim> &J,
|
||||
const real_t &w)
|
||||
{
|
||||
const auto invJ = inv(J);
|
||||
const auto dvdx = dvdxi * invJ;
|
||||
const auto test_function_terms = inv(J);
|
||||
return dot(dvdx, J) * det(J) * w * test_function_terms;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,183 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#ifndef INCLUDE_NLOHMANN_JSON_FWD_HPP_
|
||||
#define INCLUDE_NLOHMANN_JSON_FWD_HPP_
|
||||
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <map> // map
|
||||
#include <memory> // allocator
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
// #include <nlohmann/detail/abi_macros.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// This file contains all macro definitions affecting or depending on the ABI
|
||||
|
||||
#ifndef JSON_SKIP_LIBRARY_VERSION_CHECK
|
||||
#if defined(NLOHMANN_JSON_VERSION_MAJOR) && \
|
||||
defined(NLOHMANN_JSON_VERSION_MINOR) && \
|
||||
defined(NLOHMANN_JSON_VERSION_PATCH)
|
||||
#if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || \
|
||||
NLOHMANN_JSON_VERSION_PATCH != 0
|
||||
#warning "Already included a different version of the library!"
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#define NLOHMANN_JSON_VERSION_MAJOR 3 // NOLINT(modernize-macro-to-enum)
|
||||
#define NLOHMANN_JSON_VERSION_MINOR 12 // NOLINT(modernize-macro-to-enum)
|
||||
#define NLOHMANN_JSON_VERSION_PATCH 0 // NOLINT(modernize-macro-to-enum)
|
||||
|
||||
#ifndef JSON_DIAGNOSTICS
|
||||
#define JSON_DIAGNOSTICS 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_DIAGNOSTIC_POSITIONS
|
||||
#define JSON_DIAGNOSTIC_POSITIONS 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0
|
||||
#endif
|
||||
|
||||
#if JSON_DIAGNOSTICS
|
||||
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag
|
||||
#else
|
||||
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS
|
||||
#endif
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS _dp
|
||||
#else
|
||||
#define NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS
|
||||
#endif
|
||||
|
||||
#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON _ldvcmp
|
||||
#else
|
||||
#define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON
|
||||
#endif
|
||||
|
||||
#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION
|
||||
#define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0
|
||||
#endif
|
||||
|
||||
// Construct the namespace ABI tags component
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi##a##b##c
|
||||
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \
|
||||
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c)
|
||||
|
||||
#define NLOHMANN_JSON_ABI_TAGS \
|
||||
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
|
||||
NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \
|
||||
NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \
|
||||
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS)
|
||||
|
||||
// Construct the namespace version component
|
||||
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \
|
||||
_v##major##_##minor##_##patch
|
||||
#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT(major, minor, patch) \
|
||||
NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch)
|
||||
|
||||
#if NLOHMANN_JSON_NAMESPACE_NO_VERSION
|
||||
#define NLOHMANN_JSON_NAMESPACE_VERSION
|
||||
#else
|
||||
#define NLOHMANN_JSON_NAMESPACE_VERSION \
|
||||
NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT(NLOHMANN_JSON_VERSION_MAJOR, \
|
||||
NLOHMANN_JSON_VERSION_MINOR, \
|
||||
NLOHMANN_JSON_VERSION_PATCH)
|
||||
#endif
|
||||
|
||||
// Combine namespace components
|
||||
#define NLOHMANN_JSON_NAMESPACE_CONCAT_EX(a, b) a##b
|
||||
#define NLOHMANN_JSON_NAMESPACE_CONCAT(a, b) \
|
||||
NLOHMANN_JSON_NAMESPACE_CONCAT_EX(a, b)
|
||||
|
||||
#ifndef NLOHMANN_JSON_NAMESPACE
|
||||
#define NLOHMANN_JSON_NAMESPACE \
|
||||
nlohmann::NLOHMANN_JSON_NAMESPACE_CONCAT(NLOHMANN_JSON_ABI_TAGS, \
|
||||
NLOHMANN_JSON_NAMESPACE_VERSION)
|
||||
#endif
|
||||
|
||||
#ifndef NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
#define NLOHMANN_JSON_NAMESPACE_BEGIN \
|
||||
namespace nlohmann { \
|
||||
inline namespace NLOHMANN_JSON_NAMESPACE_CONCAT( \
|
||||
NLOHMANN_JSON_ABI_TAGS, NLOHMANN_JSON_NAMESPACE_VERSION) {
|
||||
#endif
|
||||
|
||||
#ifndef NLOHMANN_JSON_NAMESPACE_END
|
||||
#define NLOHMANN_JSON_NAMESPACE_END \
|
||||
} /* namespace (inline namespace) NOLINT(readability/namespace) */ \
|
||||
} // namespace nlohmann
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief namespace for Niels Lohmann
|
||||
@see https://github.com/nlohmann
|
||||
@since version 1.0.0
|
||||
*/
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
|
||||
/*!
|
||||
@brief default JSONSerializer template argument
|
||||
|
||||
This serializer ignores the template arguments and uses ADL
|
||||
([argument-dependent lookup](https://en.cppreference.com/w/cpp/language/adl))
|
||||
for serialization.
|
||||
*/
|
||||
template <typename T = void, typename SFINAE = void> struct adl_serializer;
|
||||
|
||||
/// a class to store JSON values
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/
|
||||
template <template <typename U, typename V, typename... Args> class ObjectType =
|
||||
std::map,
|
||||
template <typename U, typename... Args> class ArrayType = std::vector,
|
||||
class StringType = std::string, class BooleanType = bool,
|
||||
class NumberIntegerType = std::int64_t,
|
||||
class NumberUnsignedType = std::uint64_t,
|
||||
class NumberFloatType = double,
|
||||
template <typename U> class AllocatorType = std::allocator,
|
||||
template <typename T, typename SFINAE = void> class JSONSerializer =
|
||||
adl_serializer,
|
||||
class BinaryType =
|
||||
std::vector<std::uint8_t>, // cppcheck-suppress syntaxError
|
||||
class CustomBaseClass = void>
|
||||
class basic_json;
|
||||
|
||||
/// @brief JSON Pointer defines a string syntax for identifying a specific value
|
||||
/// within a JSON document
|
||||
/// @sa https://json.nlohmann.me/api/json_pointer/
|
||||
template <typename RefStringType> class json_pointer;
|
||||
|
||||
/*!
|
||||
@brief default specialization
|
||||
@sa https://json.nlohmann.me/api/json/
|
||||
*/
|
||||
using json = basic_json<>;
|
||||
|
||||
/// @brief a minimal map-like container that preserves insertion order
|
||||
/// @sa https://json.nlohmann.me/api/ordered_map/
|
||||
template <class Key, class T, class IgnoredLess, class Allocator>
|
||||
struct ordered_map;
|
||||
|
||||
/// @brief specialization that maintains the insertion order of object keys
|
||||
/// @sa https://json.nlohmann.me/api/ordered_json/
|
||||
using ordered_json = basic_json<nlohmann::ordered_map>;
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
#endif // INCLUDE_NLOHMANN_JSON_FWD_HPP_
|
||||
+48
-27
@@ -128,32 +128,46 @@ set(SRCS
|
||||
normal_deriv_restriction.cpp
|
||||
staticcond.cpp
|
||||
tmop.cpp
|
||||
tmop/tmop_pa.cpp
|
||||
tmop/tmop_pa_da3.cpp
|
||||
tmop/tmop_pa_h2d.cpp
|
||||
tmop/tmop_pa_h2d_c0.cpp
|
||||
tmop/tmop_pa_h2m.cpp
|
||||
tmop/tmop_pa_h2m_c0.cpp
|
||||
tmop/tmop_pa_h2s.cpp
|
||||
tmop/tmop_pa_h2s_c0.cpp
|
||||
tmop/tmop_pa_h3d.cpp
|
||||
tmop/tmop_pa_h3d_c0.cpp
|
||||
tmop/tmop_pa_h3m.cpp
|
||||
tmop/tmop_pa_h3m_c0.cpp
|
||||
tmop/tmop_pa_h3s.cpp
|
||||
tmop/tmop_pa_h3s_c0.cpp
|
||||
tmop/tmop_pa_jp2.cpp
|
||||
tmop/tmop_pa_jp3.cpp
|
||||
tmop/tmop_pa_p2.cpp
|
||||
tmop/tmop_pa_p2_c0.cpp
|
||||
tmop/tmop_pa_p3.cpp
|
||||
tmop/tmop_pa_p3_c0.cpp
|
||||
tmop/tmop_pa_tc2.cpp
|
||||
tmop/tmop_pa_tc3.cpp
|
||||
tmop/tmop_pa_w2.cpp
|
||||
tmop/tmop_pa_w2_c0.cpp
|
||||
tmop/tmop_pa_w3.cpp
|
||||
tmop/tmop_pa_w3_c0.cpp
|
||||
tmop/pa.cpp
|
||||
tmop/assemble/diag2_limit.cpp
|
||||
tmop/assemble/diag2.cpp
|
||||
tmop/assemble/grad2_limit.cpp
|
||||
tmop/assemble/grad2.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3.cpp
|
||||
tmop/assemble/grad3_limit.cpp
|
||||
tmop/assemble/grad3.cpp
|
||||
tmop/metrics/001.cpp
|
||||
tmop/metrics/002.cpp
|
||||
tmop/metrics/007.cpp
|
||||
tmop/metrics/056.cpp
|
||||
tmop/metrics/077.cpp
|
||||
tmop/metrics/080.cpp
|
||||
tmop/metrics/094.cpp
|
||||
tmop/metrics/302.cpp
|
||||
tmop/metrics/303.cpp
|
||||
tmop/metrics/315.cpp
|
||||
tmop/metrics/318.cpp
|
||||
tmop/metrics/321.cpp
|
||||
tmop/metrics/332.cpp
|
||||
tmop/metrics/338.cpp
|
||||
tmop/mult/grad2_limit.cpp
|
||||
tmop/mult/grad2.cpp
|
||||
tmop/mult/mult2_limit.cpp
|
||||
tmop/mult/mult2.cpp
|
||||
tmop/mult/grad3_limit.cpp
|
||||
tmop/mult/grad3.cpp
|
||||
tmop/mult/mult3_limit.cpp
|
||||
tmop/mult/mult3.cpp
|
||||
tmop/tools/det2_jpr.cpp
|
||||
tmop/tools/det3_jpr.cpp
|
||||
tmop/tools/discrete.cpp
|
||||
tmop/tools/energy2_limit.cpp
|
||||
tmop/tools/energy2.cpp
|
||||
tmop/tools/energy3_limit.cpp
|
||||
tmop/tools/energy3.cpp
|
||||
tmop/tools/target2.cpp
|
||||
tmop/tools/target3.cpp
|
||||
tmop_tools.cpp
|
||||
tmop_amr.cpp
|
||||
gslib.cpp
|
||||
@@ -182,6 +196,8 @@ set(HDRS
|
||||
integ/bilininteg_hdiv_kernels.hpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.hpp
|
||||
integ/bilininteg_mass_kernels.hpp
|
||||
integ/bilininteg_vecdiffusion_pa.hpp
|
||||
integ/bilininteg_vecmass_pa.hpp
|
||||
coefficient.hpp
|
||||
complex_fem.hpp
|
||||
convergence.hpp
|
||||
@@ -279,7 +295,12 @@ set(HDRS
|
||||
tfespace.hpp
|
||||
tintrules.hpp
|
||||
tmop.hpp
|
||||
tmop/tmop_pa.hpp
|
||||
tmop/pa.hpp
|
||||
tmop/assemble/grad2.hpp
|
||||
tmop/assemble/grad2.hpp
|
||||
tmop/mult/mult2.hpp
|
||||
tmop/mult/mult3.hpp
|
||||
tmop/tools/energy2.hpp
|
||||
tmop_tools.hpp
|
||||
tmop_amr.hpp
|
||||
gslib.hpp
|
||||
|
||||
@@ -3066,7 +3066,6 @@ void VectorDiffusionIntegrator::AssembleElementMatrix(
|
||||
|
||||
for (int i = 0; i < ir -> GetNPoints(); i++)
|
||||
{
|
||||
|
||||
const IntegrationPoint &ip = ir->IntPoint(i);
|
||||
el.CalcDShape(ip, dshape);
|
||||
|
||||
|
||||
+44
-41
@@ -2596,41 +2596,40 @@ public:
|
||||
by scalar FE through standard transformation. */
|
||||
class VectorMassIntegrator: public BilinearFormIntegrator
|
||||
{
|
||||
private:
|
||||
int vdim;
|
||||
int vdim = -1, Q_order = 0;
|
||||
Vector shape, te_shape, vec;
|
||||
DenseMatrix partelmat;
|
||||
DenseMatrix mcoeff;
|
||||
int Q_order;
|
||||
|
||||
protected:
|
||||
Coefficient *Q;
|
||||
VectorCoefficient *VQ;
|
||||
MatrixCoefficient *MQ;
|
||||
Coefficient *Q = nullptr;
|
||||
VectorCoefficient *VQ = nullptr;
|
||||
MatrixCoefficient *MQ = nullptr;
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, nq, dofs1D, quad1D;
|
||||
int ne, dim, dofs1D, quad1D, coeff_vdim;
|
||||
Vector pa_data;
|
||||
|
||||
public:
|
||||
/// Construct an integrator with coefficient 1.0
|
||||
VectorMassIntegrator()
|
||||
: vdim(-1), Q_order(0), Q(NULL), VQ(NULL), MQ(NULL) { }
|
||||
VectorMassIntegrator() = default;
|
||||
|
||||
/** Construct an integrator with scalar coefficient q. If possible, save
|
||||
memory by using a scalar integrator since the resulting matrix is block
|
||||
diagonal with the same diagonal block repeated. */
|
||||
VectorMassIntegrator(Coefficient &q, int qo = 0)
|
||||
: vdim(-1), Q_order(qo), Q(&q), VQ(NULL), MQ(NULL) { }
|
||||
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir), vdim(-1), Q_order(0), Q(&q), VQ(NULL),
|
||||
MQ(NULL) { }
|
||||
VectorMassIntegrator(Coefficient &q, int qo = 0): Q_order(qo), Q(&q) { }
|
||||
|
||||
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir):
|
||||
BilinearFormIntegrator(ir), Q(&q) { }
|
||||
|
||||
/// Construct an integrator with diagonal coefficient q
|
||||
VectorMassIntegrator(VectorCoefficient &q, int qo = 0)
|
||||
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(&q), MQ(NULL) { }
|
||||
VectorMassIntegrator(VectorCoefficient &q, int qo = 0):
|
||||
vdim(q.GetVDim()), Q_order(qo), VQ(&q) { }
|
||||
|
||||
/// Construct an integrator with matrix coefficient q
|
||||
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0)
|
||||
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(NULL), MQ(&q) { }
|
||||
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0):
|
||||
vdim(q.GetVDim()), Q_order(qo), MQ(&q) { }
|
||||
|
||||
int GetVDim() const { return vdim; }
|
||||
void SetVDim(int vdim_) { vdim = vdim_; }
|
||||
@@ -2642,6 +2641,7 @@ public:
|
||||
const FiniteElement &test_fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &elmat) override;
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
void AssemblePA(const FiniteElementSpace &fes) override;
|
||||
void AssembleMF(const FiniteElementSpace &fes) override;
|
||||
@@ -2650,6 +2650,15 @@ public:
|
||||
void AddMultPA(const Vector &x, Vector &y) const override;
|
||||
void AddMultMF(const Vector &x, Vector &y) const override;
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
using VectorMassAddMultPAType =
|
||||
void(*)(const int, const int,
|
||||
const Array<real_t>&, const Vector&,
|
||||
const Vector&, Vector&, const int, const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(VectorMassAddMultPA,
|
||||
VectorMassAddMultPAType,
|
||||
(int, int, int));
|
||||
};
|
||||
|
||||
|
||||
@@ -3120,23 +3129,21 @@ public:
|
||||
to be the spatial dimension (i.e. 2-dimension or 3-dimension). */
|
||||
class VectorDiffusionIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
Coefficient *Q = NULL;
|
||||
VectorCoefficient *VQ = NULL;
|
||||
MatrixCoefficient *MQ = NULL;
|
||||
int vdim = -1;
|
||||
DenseMatrix dshape, dshapedxt, pelmat;
|
||||
DenseMatrix mcoeff;
|
||||
Vector vcoeff;
|
||||
|
||||
protected:
|
||||
Coefficient *Q = nullptr;
|
||||
VectorCoefficient *VQ = nullptr;
|
||||
MatrixCoefficient *MQ = nullptr;
|
||||
// PA extension
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, sdim, ne, dofs1D, quad1D;
|
||||
int ne, dim, sdim, dofs1D, quad1D, coeff_vdim;
|
||||
Vector pa_data;
|
||||
|
||||
private:
|
||||
DenseMatrix dshape, dshapedxt, pelmat;
|
||||
int vdim = -1;
|
||||
DenseMatrix mcoeff;
|
||||
Vector vcoeff;
|
||||
|
||||
public:
|
||||
VectorDiffusionIntegrator(const IntegrationRule *ir = nullptr);
|
||||
|
||||
@@ -3189,6 +3196,7 @@ public:
|
||||
void AssembleElementVector(const FiniteElement &el,
|
||||
ElementTransformation &Tr,
|
||||
const Vector &elfun, Vector &elvect) override;
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
void AssemblePA(const FiniteElementSpace &fes) override;
|
||||
void AssembleMF(const FiniteElementSpace &fes) override;
|
||||
@@ -3198,13 +3206,11 @@ public:
|
||||
void AddMultMF(const Vector &x, Vector &y) const override;
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
/// arguments: ne, B, G, Bt, Gt, pa_data, x, y, d1d, q1d, vdim
|
||||
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &, const Vector &,
|
||||
const Vector &, Vector &, const int,
|
||||
const int, const int);
|
||||
/// arguments: ne, coeff_vdim, B, G, pa_data, x, y, d1d, q1d, vdim
|
||||
using ApplyKernelType = void (*)(const int, const int,
|
||||
const Array<real_t> &, const Array<real_t> &,
|
||||
const Vector &, const Vector &, Vector &,
|
||||
const int, const int, const int);
|
||||
|
||||
/// arguments: dim, vdim, d1d, q1d
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int, int));
|
||||
@@ -3215,10 +3221,7 @@ public:
|
||||
ApplyPAKernels::Specialization<DIM, VDIM, D1D, Q1D>::Add();
|
||||
}
|
||||
|
||||
struct Kernels
|
||||
{
|
||||
Kernels();
|
||||
};
|
||||
// struct Kernels { Kernels(); };
|
||||
};
|
||||
|
||||
/** Integrator for the linear elasticity form:
|
||||
|
||||
@@ -1085,6 +1085,29 @@ void SumCoefficient::SetTime(real_t t)
|
||||
this->Coefficient::SetTime(t);
|
||||
}
|
||||
|
||||
void SumCoefficient::Project(QuadratureFunction &qf)
|
||||
{
|
||||
if (a == nullptr)
|
||||
{
|
||||
// qf = alpha*aConst + beta * b
|
||||
const real_t d_alpha_a = aConst*alpha;
|
||||
const real_t d_beta = beta;
|
||||
b->Project(qf);
|
||||
auto d_qf = qf.ReadWrite();
|
||||
mfem::forall(qf.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_qf[i] = d_alpha_a + d_beta*d_qf[i];
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
a->Project(qf);
|
||||
QuadratureFunction qf_b(*qf.GetSpace());
|
||||
b->Project(qf_b);
|
||||
add(alpha, qf, beta, qf_b, qf);
|
||||
}
|
||||
}
|
||||
|
||||
void ProductCoefficient::SetTime(real_t t)
|
||||
{
|
||||
if (a) { a->SetTime(t); }
|
||||
@@ -1092,6 +1115,23 @@ void ProductCoefficient::SetTime(real_t t)
|
||||
this->Coefficient::SetTime(t);
|
||||
}
|
||||
|
||||
void ProductCoefficient::Project(QuadratureFunction &qf)
|
||||
{
|
||||
if (a == nullptr)
|
||||
{
|
||||
// qf = aConst * b
|
||||
b->Project(qf);
|
||||
qf *= aConst;
|
||||
}
|
||||
else
|
||||
{
|
||||
a->Project(qf);
|
||||
QuadratureFunction qf_b(qf.GetSpace());
|
||||
b->Project(qf_b);
|
||||
qf *= qf_b;
|
||||
}
|
||||
}
|
||||
|
||||
void RatioCoefficient::SetTime(real_t t)
|
||||
{
|
||||
if (a) { a->SetTime(t); }
|
||||
@@ -1099,6 +1139,38 @@ void RatioCoefficient::SetTime(real_t t)
|
||||
this->Coefficient::SetTime(t);
|
||||
}
|
||||
|
||||
void RatioCoefficient::Project(QuadratureFunction &qf)
|
||||
{
|
||||
if (b == nullptr)
|
||||
{
|
||||
if (a == nullptr)
|
||||
{
|
||||
qf = aConst / bConst;
|
||||
}
|
||||
else
|
||||
{
|
||||
a->Project(qf);
|
||||
qf *= 1.0/bConst;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (a == nullptr)
|
||||
{
|
||||
b->Project(qf);
|
||||
qf.Reciprocal();
|
||||
qf *= aConst;
|
||||
}
|
||||
else
|
||||
{
|
||||
a->Project(qf);
|
||||
QuadratureFunction qf_b(qf.GetSpace());
|
||||
b->Project(qf_b);
|
||||
qf /= qf_b;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void PowerCoefficient::SetTime(real_t t)
|
||||
{
|
||||
if (a) { a->SetTime(t); }
|
||||
|
||||
@@ -1456,6 +1456,9 @@ public:
|
||||
/// Set the time for internally stored coefficients
|
||||
void SetTime(real_t t) override;
|
||||
|
||||
/// @copydoc Coefficient::Project(QuadratureFunction &)
|
||||
void Project(QuadratureFunction &qf) override;
|
||||
|
||||
/// Reset the first term in the linear combination as a constant
|
||||
void SetAConst(real_t A) { a = NULL; aConst = A; }
|
||||
/// Return the first term in the linear combination
|
||||
@@ -1637,6 +1640,9 @@ public:
|
||||
/// Set the time for internally stored coefficients
|
||||
void SetTime(real_t t) override;
|
||||
|
||||
/// @copydoc Coefficient::Project(QuadratureFunction &)
|
||||
void Project(QuadratureFunction &qf) override;
|
||||
|
||||
/// Reset the first term in the product as a constant
|
||||
void SetAConst(real_t A) { a = NULL; aConst = A; }
|
||||
/// Return the first term in the product
|
||||
@@ -1685,6 +1691,9 @@ public:
|
||||
/// Set the time for internally stored coefficients
|
||||
void SetTime(real_t t) override;
|
||||
|
||||
/// @copydoc Coefficient::Project(QuadratureFunction &)
|
||||
void Project(QuadratureFunction &qf) override;
|
||||
|
||||
/// Reset the numerator in the ratio as a constant
|
||||
void SetAConst(real_t A) { a = NULL; aConst = A; }
|
||||
/// Return the numerator of the ratio
|
||||
|
||||
+281
-8
@@ -11,14 +11,15 @@
|
||||
|
||||
#include "complex_fem.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../general/text.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
|
||||
: Vector(2*(fes->GetVSize()))
|
||||
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *f)
|
||||
: Vector(2*(f->GetVSize())), fes(f), fec_owned(NULL)
|
||||
{
|
||||
UseDevice(true);
|
||||
this->Vector::operator=(0.0);
|
||||
@@ -28,12 +29,88 @@ ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
|
||||
|
||||
gfi = new GridFunction();
|
||||
gfi->MakeRef(fes, *this, fes->GetVSize());
|
||||
|
||||
fes_sequence = fes->GetSequence();
|
||||
}
|
||||
|
||||
ComplexGridFunction::ComplexGridFunction(Mesh *m, std::istream &input)
|
||||
: Vector(), fes(NULL), fec_owned(NULL)
|
||||
{
|
||||
string buff;
|
||||
|
||||
// Grid functions are stored on the device
|
||||
UseDevice(true);
|
||||
|
||||
input >> std::ws;
|
||||
getline(input, buff); // 'ComplexGridFunction'
|
||||
filter_dos(buff);
|
||||
if (buff != "ComplexGridFunction")
|
||||
{
|
||||
MFEM_ABORT("unrecognized file header: " << buff);
|
||||
}
|
||||
|
||||
fes = new FiniteElementSpace;
|
||||
fec_owned = fes->Load(m, input);
|
||||
|
||||
skip_comment_lines(input, '#');
|
||||
istream::int_type next_char = input.peek();
|
||||
if (next_char == 'N') // First letter of "NURBS_patches"
|
||||
{
|
||||
getline(input, buff);
|
||||
filter_dos(buff);
|
||||
if (buff == "NURBS_patches")
|
||||
{
|
||||
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("unknown section: " << buff);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
Vector::Load(input, 2*fes->GetVSize());
|
||||
|
||||
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
|
||||
if (fes->Nonconforming() &&
|
||||
fes->GetMesh()->ncmesh->IsLegacyLoaded())
|
||||
{
|
||||
// LegacyNCReorder();
|
||||
MFEM_ABORT("LegacyNCReorder not supported for "
|
||||
"ComplexGridFunction objects");
|
||||
}
|
||||
}
|
||||
|
||||
gfr = new GridFunction();
|
||||
gfr->MakeRef(fes, *this, 0);
|
||||
|
||||
gfi = new GridFunction();
|
||||
gfi->MakeRef(fes, *this, fes->GetVSize());
|
||||
|
||||
fes_sequence = fes->GetSequence();
|
||||
}
|
||||
|
||||
void ComplexGridFunction::Destroy()
|
||||
{
|
||||
delete gfr; delete gfi;
|
||||
|
||||
if (fec_owned)
|
||||
{
|
||||
delete fes;
|
||||
delete fec_owned;
|
||||
fec_owned = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ComplexGridFunction::Update()
|
||||
{
|
||||
FiniteElementSpace *fes = gfr->FESpace();
|
||||
if (fes->GetSequence() == fes_sequence)
|
||||
{
|
||||
return; // space and grid function are in sync, no-op
|
||||
}
|
||||
fes_sequence = fes->GetSequence();
|
||||
|
||||
const int vsize = fes->GetVSize();
|
||||
|
||||
const Operator *T = fes->GetUpdateOperator();
|
||||
@@ -84,6 +161,17 @@ ComplexGridFunction::Update()
|
||||
}
|
||||
}
|
||||
|
||||
int ComplexGridFunction::VectorDim() const
|
||||
{
|
||||
const FiniteElement *fe = fes->GetTypicalFE();
|
||||
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
|
||||
{
|
||||
return fes->GetVDim();
|
||||
}
|
||||
return fes->GetVDim()*std::max(fes->GetMesh()->SpaceDimension(),
|
||||
fe->GetRangeDim());
|
||||
}
|
||||
|
||||
void
|
||||
ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
|
||||
Coefficient &imag_coeff)
|
||||
@@ -149,6 +237,35 @@ ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
|
||||
gfi->SyncAliasMemory(*this);
|
||||
}
|
||||
|
||||
void ComplexGridFunction::Save(std::ostream &os) const
|
||||
{
|
||||
os << "ComplexGridFunction\n";
|
||||
fes->Save(os);
|
||||
os << '\n';
|
||||
if (fes->GetOrdering() == Ordering::byNODES)
|
||||
{
|
||||
Vector::Print(os, 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
Vector::Print(os, fes->GetVDim());
|
||||
}
|
||||
os.flush();
|
||||
}
|
||||
|
||||
void ComplexGridFunction::Save(const char *fname, int precision) const
|
||||
{
|
||||
ofstream ofs(fname);
|
||||
ofs.precision(precision);
|
||||
Save(ofs);
|
||||
}
|
||||
|
||||
std::ostream &operator<<(std::ostream &os, const ComplexGridFunction &sol)
|
||||
{
|
||||
sol.Save(os);
|
||||
return os;
|
||||
}
|
||||
|
||||
|
||||
ComplexLinearForm::ComplexLinearForm(FiniteElementSpace *fes,
|
||||
ComplexOperator::Convention convention)
|
||||
@@ -654,8 +771,8 @@ SesquilinearForm::Update(FiniteElementSpace *nfes)
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
|
||||
: Vector(2*(pfes->GetVSize()))
|
||||
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pf)
|
||||
: Vector(2*(pf->GetVSize())), pfes(pf), fec_owned(NULL)
|
||||
{
|
||||
UseDevice(true);
|
||||
this->Vector::operator=(0.0);
|
||||
@@ -665,12 +782,105 @@ ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
|
||||
|
||||
pgfi = new ParGridFunction();
|
||||
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
|
||||
|
||||
fes_sequence = pfes->GetSequence();
|
||||
}
|
||||
|
||||
ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
|
||||
: Vector(), pfes(NULL), fec_owned(NULL)
|
||||
{
|
||||
string buff;
|
||||
|
||||
// Grid functions are stored on the device
|
||||
UseDevice(true);
|
||||
|
||||
input >> std::ws;
|
||||
getline(input, buff); // 'ParComplexGridFunction'
|
||||
filter_dos(buff);
|
||||
if (buff != "ParComplexGridFunction")
|
||||
{
|
||||
MFEM_ABORT("unrecognized file header: " << buff);
|
||||
}
|
||||
|
||||
FiniteElementSpace *fes = new FiniteElementSpace;
|
||||
fec_owned = fes->Load(m, input);
|
||||
|
||||
pfes = new ParFiniteElementSpace(m, fec_owned, fes->GetVDim(),
|
||||
fes->GetOrdering());
|
||||
|
||||
delete fes;
|
||||
|
||||
skip_comment_lines(input, '#');
|
||||
istream::int_type next_char = input.peek();
|
||||
if (next_char == 'N') // First letter of "NURBS_patches"
|
||||
{
|
||||
getline(input, buff);
|
||||
filter_dos(buff);
|
||||
if (buff == "NURBS_patches")
|
||||
{
|
||||
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("unknown section: " << buff);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
int vsize = pfes->GetVSize();
|
||||
Vector::Load(input, 2*vsize);
|
||||
|
||||
real_t *data_ = const_cast<real_t*>(HostRead());
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
|
||||
if (pfes->Nonconforming() &&
|
||||
pfes->GetMesh()->ncmesh->IsLegacyLoaded())
|
||||
{
|
||||
// LegacyNCReorder();
|
||||
MFEM_ABORT("LegacyNCReorder not supported for "
|
||||
"ComplexGridFunction objects");
|
||||
}
|
||||
}
|
||||
|
||||
pgfr = new ParGridFunction();
|
||||
pgfr->MakeRef(pfes, *this, 0);
|
||||
|
||||
pgfi = new ParGridFunction();
|
||||
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
|
||||
|
||||
fes_sequence = pfes->GetSequence();
|
||||
}
|
||||
|
||||
void ParComplexGridFunction::Destroy()
|
||||
{
|
||||
delete pgfr; delete pgfi;
|
||||
|
||||
if (fec_owned)
|
||||
{
|
||||
delete pfes;
|
||||
delete fec_owned;
|
||||
fec_owned = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ParComplexGridFunction::Update()
|
||||
{
|
||||
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
|
||||
if (pfes->GetSequence() == fes_sequence)
|
||||
{
|
||||
return; // space and grid function are in sync, no-op
|
||||
}
|
||||
fes_sequence = pfes->GetSequence();
|
||||
|
||||
const int vsize = pfes->GetVSize();
|
||||
|
||||
const Operator *T = pfes->GetUpdateOperator();
|
||||
@@ -719,6 +929,17 @@ ParComplexGridFunction::Update()
|
||||
}
|
||||
}
|
||||
|
||||
int ParComplexGridFunction::VectorDim() const
|
||||
{
|
||||
const FiniteElement *fe = pfes->GetTypicalFE();
|
||||
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
|
||||
{
|
||||
return pfes->GetVDim();
|
||||
}
|
||||
return pfes->GetVDim()*std::max(pfes->GetMesh()->SpaceDimension(),
|
||||
fe->GetRangeDim());
|
||||
}
|
||||
|
||||
void
|
||||
ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
|
||||
Coefficient &imag_coeff)
|
||||
@@ -789,7 +1010,6 @@ ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
|
||||
void
|
||||
ParComplexGridFunction::Distribute(const Vector *tv)
|
||||
{
|
||||
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
|
||||
const int tvsize = pfes->GetTrueVSize();
|
||||
|
||||
tv->Read();
|
||||
@@ -807,7 +1027,6 @@ ParComplexGridFunction::Distribute(const Vector *tv)
|
||||
void
|
||||
ParComplexGridFunction::ParallelProject(Vector &tv) const
|
||||
{
|
||||
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
|
||||
const int tvsize = pfes->GetTrueVSize();
|
||||
|
||||
tv.Write();
|
||||
@@ -825,6 +1044,60 @@ ParComplexGridFunction::ParallelProject(Vector &tv) const
|
||||
tvi.SyncAliasMemory(tv);
|
||||
}
|
||||
|
||||
void ParComplexGridFunction::Save(std::ostream &os) const
|
||||
{
|
||||
os << "ParComplexGridFunction\n";
|
||||
pfes->Save(os);
|
||||
os << '\n';
|
||||
|
||||
int vsize = pfes->GetVSize();
|
||||
real_t *data_ = const_cast<real_t*>(HostRead());
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
|
||||
if (pfes->GetOrdering() == Ordering::byNODES)
|
||||
{
|
||||
Vector::Print(os, 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
Vector::Print(os, pfes->GetVDim());
|
||||
}
|
||||
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
|
||||
os.flush();
|
||||
}
|
||||
|
||||
void ParComplexGridFunction::Save(const char *fname, int precision) const
|
||||
{
|
||||
int rank = pfes->GetMyRank();
|
||||
ostringstream fname_with_suffix;
|
||||
fname_with_suffix << fname << "." << setfill('0') << setw(6) << rank;
|
||||
ofstream ofs(fname_with_suffix.str().c_str());
|
||||
ofs.precision(precision);
|
||||
Save(ofs);
|
||||
}
|
||||
|
||||
std::ostream &operator<<(std::ostream &os, const ParComplexGridFunction &sol)
|
||||
{
|
||||
sol.Save(os);
|
||||
return os;
|
||||
}
|
||||
|
||||
|
||||
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
ComplexOperator::Convention
|
||||
|
||||
+159
-16
@@ -35,15 +35,53 @@ private:
|
||||
GridFunction * gfi;
|
||||
|
||||
protected:
|
||||
void Destroy() { delete gfr; delete gfi; }
|
||||
/// FE space on which the grid function lives. Owned if #fec_owned
|
||||
/// is not NULL.
|
||||
FiniteElementSpace *fes;
|
||||
|
||||
/** @brief Used when the grid function is read from a file. It can also be
|
||||
set explicitly, see MakeOwner().
|
||||
|
||||
If not NULL, this pointer is owned by the ComplexGridFunction. */
|
||||
FiniteElementCollection *fec_owned;
|
||||
|
||||
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
|
||||
|
||||
void Destroy();
|
||||
|
||||
public:
|
||||
/** @brief Construct a ComplexGridFunction associated with the
|
||||
FiniteElementSpace @a *f. */
|
||||
ComplexGridFunction(FiniteElementSpace *f);
|
||||
|
||||
/** @brief Construct a ComplexGridFunction on the given Mesh, using the data
|
||||
from @a input.
|
||||
|
||||
The content of @a input should be in the format created by the method
|
||||
Save(). The reconstructed FiniteElementSpace and FiniteElementCollection
|
||||
are owned by the ComplexGridFunction. */
|
||||
ComplexGridFunction(Mesh *m, std::istream &input);
|
||||
|
||||
void Update();
|
||||
|
||||
/** Return update counter, similar to Mesh::GetSequence(). Used to
|
||||
check if it is up to date with the space. */
|
||||
long GetSequence() const { return fes_sequence; }
|
||||
|
||||
/// Make the ComplexGridFunction the owner of #fec_owned and #fes.
|
||||
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
|
||||
of #fec_owned and #fes is taken away. */
|
||||
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
|
||||
|
||||
/// Returns a pointer to the FiniteElementCollection used to
|
||||
/// construct this ComplexGridFunction if this class owns that
|
||||
/// object. Otherwise this function will return NULL.
|
||||
FiniteElementCollection *OwnFEC() { return fec_owned; }
|
||||
|
||||
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
|
||||
/// underlying #fes
|
||||
int VectorDim() const;
|
||||
|
||||
/// Assign constant values to the ComplexGridFunction data.
|
||||
ComplexGridFunction &operator=(const std::complex<real_t> & value)
|
||||
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
|
||||
@@ -63,8 +101,8 @@ public:
|
||||
VectorCoefficient &imag_coeff,
|
||||
Array<int> &attr);
|
||||
|
||||
FiniteElementSpace *FESpace() { return gfr->FESpace(); }
|
||||
const FiniteElementSpace *FESpace() const { return gfr->FESpace(); }
|
||||
FiniteElementSpace *FESpace() { return fes; }
|
||||
const FiniteElementSpace *FESpace() const { return fes; }
|
||||
|
||||
GridFunction & real() { return *gfr; }
|
||||
GridFunction & imag() { return *gfi; }
|
||||
@@ -79,11 +117,52 @@ public:
|
||||
/// @a gfr and @a gfi to match the ComplexGridFunction.
|
||||
void SyncAlias() { gfr->SyncAliasMemory(*this); gfi->SyncAliasMemory(*this); }
|
||||
|
||||
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued scalar fields
|
||||
///
|
||||
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
|
||||
/// const IntegrationRule *irs[],
|
||||
/// const Array<int> *elems) const
|
||||
/// for more detailed documentation.
|
||||
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
|
||||
const IntegrationRule *irs[] = NULL) const
|
||||
{
|
||||
real_t err_r = gfr->ComputeL2Error(exsolr, irs);
|
||||
real_t err_i = gfi->ComputeL2Error(exsoli, irs);
|
||||
return sqrt(err_r * err_r + err_i * err_i);
|
||||
}
|
||||
|
||||
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued vector fields
|
||||
///
|
||||
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
|
||||
/// const IntegrationRule *irs[],
|
||||
/// const Array<int> *elems) const
|
||||
/// for more detailed documentation.
|
||||
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
|
||||
VectorCoefficient &exsoli,
|
||||
const IntegrationRule *irs[] = NULL,
|
||||
Array<int> *elems = NULL) const
|
||||
{
|
||||
real_t err_r = gfr->ComputeL2Error(exsolr, irs, elems);
|
||||
real_t err_i = gfi->ComputeL2Error(exsoli, irs, elems);
|
||||
return sqrt(err_r * err_r + err_i * err_i);
|
||||
}
|
||||
|
||||
/// Save the ComplexGridFunction to an output stream.
|
||||
virtual void Save(std::ostream &out) const;
|
||||
|
||||
/// Save the ComplexGridFunction to a file
|
||||
/** The given @a precision will be used for ASCII output. */
|
||||
virtual void Save(const char *fname, int precision=16) const;
|
||||
|
||||
/// Destroys the grid function.
|
||||
virtual ~ComplexGridFunction() { Destroy(); }
|
||||
|
||||
};
|
||||
|
||||
/** Overload operator<< for std::ostream and ComplexGridFunction; not valid
|
||||
for the class ParComplexGridFunction */
|
||||
std::ostream &operator<<(std::ostream &out, const ComplexGridFunction &sol);
|
||||
|
||||
/** Class for a complex-valued linear form
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
@@ -345,12 +424,23 @@ public:
|
||||
class ParComplexGridFunction : public Vector
|
||||
{
|
||||
private:
|
||||
|
||||
ParGridFunction * pgfr;
|
||||
ParGridFunction * pgfi;
|
||||
|
||||
protected:
|
||||
void Destroy() { delete pgfr; delete pgfi; }
|
||||
/// FE space on which the grid function lives. Owned if #fec_owned
|
||||
/// is not NULL.
|
||||
ParFiniteElementSpace *pfes;
|
||||
|
||||
/** @brief Used when the grid function is read from a file. It can also be
|
||||
set explicitly, see MakeOwner().
|
||||
|
||||
If not NULL, this pointer is owned by the ParComplexGridFunction. */
|
||||
FiniteElementCollection *fec_owned;
|
||||
|
||||
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
|
||||
|
||||
void Destroy();
|
||||
|
||||
public:
|
||||
|
||||
@@ -358,8 +448,33 @@ public:
|
||||
ParFiniteElementSpace @a *pf. */
|
||||
ParComplexGridFunction(ParFiniteElementSpace *pf);
|
||||
|
||||
/** @brief Construct a ParComplexGridFunction on a given ParMesh,
|
||||
@a pmesh, reading from an std::istream.
|
||||
|
||||
In the process, a ParFiniteElementSpace and a FiniteElementCollection are
|
||||
constructed. The new ParComplexGridFunction assumes ownership of both. */
|
||||
ParComplexGridFunction(ParMesh *pmesh, std::istream &input);
|
||||
|
||||
void Update();
|
||||
|
||||
/** Return update counter, similar to Mesh::GetSequence(). Used to
|
||||
check if it is up to date with the space. */
|
||||
long GetSequence() const { return fes_sequence; }
|
||||
|
||||
/// Make the ParComplexGridFunction the owner of #fec_owned and #pfes.
|
||||
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
|
||||
of #fec_owned and #pfes is taken away. */
|
||||
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
|
||||
|
||||
/// Returns a pointer to the FiniteElementCollection used to
|
||||
/// construct this ParComplexGridFunction if this class owns that
|
||||
/// object. Otherwise this function will return NULL.
|
||||
FiniteElementCollection *OwnFEC() { return fec_owned; }
|
||||
|
||||
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
|
||||
/// underlying #pfes
|
||||
int VectorDim() const;
|
||||
|
||||
/// Assign constant values to the ParComplexGridFunction data.
|
||||
ParComplexGridFunction &operator=(const std::complex<real_t> & value)
|
||||
{ *pgfr = value.real(); *pgfi = value.imag(); return *this; }
|
||||
@@ -385,11 +500,11 @@ public:
|
||||
/// Returns the vector restricted to the true dofs.
|
||||
void ParallelProject(Vector &tv) const;
|
||||
|
||||
FiniteElementSpace *FESpace() { return pgfr->FESpace(); }
|
||||
const FiniteElementSpace *FESpace() const { return pgfr->FESpace(); }
|
||||
FiniteElementSpace *FESpace() { return pfes; }
|
||||
const FiniteElementSpace *FESpace() const { return pfes; }
|
||||
|
||||
ParFiniteElementSpace *ParFESpace() { return pgfr->ParFESpace(); }
|
||||
const ParFiniteElementSpace *ParFESpace() const { return pgfr->ParFESpace(); }
|
||||
ParFiniteElementSpace *ParFESpace() { return pfes; }
|
||||
const ParFiniteElementSpace *ParFESpace() const { return pfes; }
|
||||
|
||||
ParGridFunction & real() { return *pgfr; }
|
||||
ParGridFunction & imag() { return *pgfi; }
|
||||
@@ -402,17 +517,32 @@ public:
|
||||
|
||||
/// Update the alias memory location of the real and imaginary
|
||||
/// ParGridFunction @a pgfr and @a pgfi to match the ParComplexGridFunction.
|
||||
void SyncAlias() { pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
|
||||
|
||||
void SyncAlias()
|
||||
{ pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
|
||||
|
||||
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
|
||||
/// scalar fields
|
||||
///
|
||||
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
|
||||
/// const IntegrationRule *irs[],
|
||||
/// const Array<int> *elems) const
|
||||
/// for more detailed documentation.
|
||||
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
|
||||
const IntegrationRule *irs[] = NULL) const
|
||||
const IntegrationRule *irs[] = NULL,
|
||||
Array<int> *elems = NULL) const
|
||||
{
|
||||
real_t err_r = pgfr->ComputeL2Error(exsolr, irs);
|
||||
real_t err_i = pgfi->ComputeL2Error(exsoli, irs);
|
||||
return sqrt(err_r * err_r + err_i * err_i);
|
||||
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
|
||||
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
|
||||
return hypot(err_r, err_i);
|
||||
}
|
||||
|
||||
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
|
||||
/// vector fields
|
||||
///
|
||||
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
|
||||
/// const IntegrationRule *irs[],
|
||||
/// const Array<int> *elems) const
|
||||
/// for more detailed documentation.
|
||||
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
|
||||
VectorCoefficient &exsoli,
|
||||
const IntegrationRule *irs[] = NULL,
|
||||
@@ -420,15 +550,28 @@ public:
|
||||
{
|
||||
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
|
||||
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
|
||||
return sqrt(err_r * err_r + err_i * err_i);
|
||||
return hypot(err_r, err_i);
|
||||
}
|
||||
|
||||
/// Save the local portion of the ParComplexGridFunction
|
||||
/** This differs from the serial ComplexGridFunction::Save in that it
|
||||
takes into account the signs of the local dofs. */
|
||||
void Save(std::ostream &out) const;
|
||||
|
||||
/// Save the ParComplexGridFunction to files
|
||||
/** Saves one file for each MPI rank. The files will be given suffixes
|
||||
according to the MPI rank. The given @a precision will be used for ASCII
|
||||
output. */
|
||||
void Save(const char *fname, int precision=16) const;
|
||||
|
||||
/// Destroys grid function.
|
||||
virtual ~ParComplexGridFunction() { Destroy(); }
|
||||
|
||||
};
|
||||
|
||||
/** Overload operator<< for std::ostream and ParComplexGridFunction */
|
||||
std::ostream &operator<<(std::ostream &out, const ParComplexGridFunction &sol);
|
||||
|
||||
/** Class for a complex-valued, parallel linear form
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
|
||||
+169
-11
@@ -310,9 +310,9 @@ void DataCollection::SaveField(const std::string &field_name)
|
||||
}
|
||||
}
|
||||
|
||||
void DataCollection::SaveQField(const std::string &q_field_name)
|
||||
void DataCollection::SaveQField(const std::string &field_name)
|
||||
{
|
||||
QFieldMapIterator it = q_field_map.find(q_field_name);
|
||||
QFieldMapIterator it = q_field_map.find(field_name);
|
||||
if (it != q_field_map.end())
|
||||
{
|
||||
SaveOneQField(it);
|
||||
@@ -780,6 +780,11 @@ void ParaViewDataCollectionBase::SetHighOrderOutput(bool high_order_output_)
|
||||
high_order_output = high_order_output_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetBoundaryOutput(bool bdr_output_)
|
||||
{
|
||||
bdr_output = bdr_output_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
|
||||
{
|
||||
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
|
||||
@@ -935,16 +940,19 @@ void ParaViewDataCollection::Save()
|
||||
std::string vtu_prefix = col_path + "/" + GenerateVTUPath() + "/";
|
||||
|
||||
// Save the local part of the mesh and grid functions fields to the local
|
||||
// VTU file
|
||||
// VTU file. Also save coefficient fields.
|
||||
{
|
||||
std::ofstream os(vtu_prefix + GenerateVTUFileName("proc", myid));
|
||||
os.precision(precision);
|
||||
SaveDataVTU(os, levels_of_detail);
|
||||
}
|
||||
|
||||
// Save the local part of the quadrature function fields
|
||||
// Save the local part of the quadrature function fields.
|
||||
for (const auto &qfield : q_field_map)
|
||||
{
|
||||
MFEM_VERIFY(!bdr_output,
|
||||
"QuadratureFunction output is not supported for "
|
||||
"ParaViewDataCollection on domain boundary!");
|
||||
const std::string &field_name = qfield.first;
|
||||
std::ofstream os(vtu_prefix + GenerateVTUFileName(field_name, myid));
|
||||
qfield.second->SaveVTU(os, pv_data_format, GetCompressionLevel(), field_name);
|
||||
@@ -960,7 +968,7 @@ void ParaViewDataCollection::Save()
|
||||
std::ofstream pvtu_out(vtu_prefix + GeneratePVTUFileName("data"));
|
||||
WritePVTUHeader(pvtu_out);
|
||||
|
||||
// Grid function fields
|
||||
// Grid function fields and coefficient fields
|
||||
pvtu_out << "<PPointData>\n";
|
||||
for (auto &field_it : field_map)
|
||||
{
|
||||
@@ -971,7 +979,24 @@ void ParaViewDataCollection::Save()
|
||||
<< VTKComponentLabels(vec_dim) << " "
|
||||
<< "format=\"" << GetDataFormatString() << "\" />\n";
|
||||
}
|
||||
for (auto &field_it : coeff_field_map)
|
||||
{
|
||||
int vec_dim = 1;
|
||||
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
|
||||
<< "\" Name=\"" << field_it.first
|
||||
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
|
||||
<< "format=\"" << GetDataFormatString() << "\" />\n";
|
||||
}
|
||||
for (auto &field_it : vcoeff_field_map)
|
||||
{
|
||||
int vec_dim = field_it.second->GetVDim();
|
||||
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
|
||||
<< "\" Name=\"" << field_it.first
|
||||
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
|
||||
<< "format=\"" << GetDataFormatString() << "\" />\n";
|
||||
}
|
||||
pvtu_out << "</PPointData>\n";
|
||||
|
||||
// Element attributes
|
||||
pvtu_out << "<PCellData>\n";
|
||||
pvtu_out << "\t<PDataArray type=\"Int32\" Name=\"" << "attribute"
|
||||
@@ -1069,7 +1094,8 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
|
||||
}
|
||||
os << " version=\"2.2\" byte_order=\"" << VTKByteOrder() << "\">\n";
|
||||
os << "<UnstructuredGrid>\n";
|
||||
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel());
|
||||
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel(),
|
||||
bdr_output);
|
||||
|
||||
// dump out the grid functions as point data
|
||||
os << "<PointData >\n";
|
||||
@@ -1077,8 +1103,21 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
|
||||
// iterate over all grid functions
|
||||
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
|
||||
{
|
||||
MFEM_VERIFY(!bdr_output,
|
||||
"GridFunction output is not supported for "
|
||||
"ParaViewDataCollection on domain boundary!");
|
||||
SaveGFieldVTU(os,ref,it);
|
||||
}
|
||||
// save the coefficient functions
|
||||
// iterate over all Coefficient and VectorCoefficient functions
|
||||
for (const auto &kv : coeff_field_map)
|
||||
{
|
||||
SaveCoeffFieldVTU(os, ref, kv.first, *kv.second);
|
||||
}
|
||||
for (const auto &kv : vcoeff_field_map)
|
||||
{
|
||||
SaveVCoeffFieldVTU(os, ref, kv.first, *kv.second);
|
||||
}
|
||||
os << "</PointData>\n";
|
||||
// close the mesh
|
||||
os << "</Piece>\n"; // close the piece open in the PrintVTU method
|
||||
@@ -1101,7 +1140,6 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
|
||||
<< "format=\"" << GetDataFormatString() << "\" >" << '\n';
|
||||
if (vec_dim == 1)
|
||||
{
|
||||
// scalar data
|
||||
for (int i = 0; i < mesh->GetNE(); i++)
|
||||
{
|
||||
RefG = GlobGeometryRefiner.Refine(
|
||||
@@ -1131,11 +1169,131 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (IsBinaryFormat())
|
||||
if (pv_data_format != VTKFormat::ASCII)
|
||||
{
|
||||
WriteVTKEncodedCompressed(os,buf.data(),buf.size(),GetCompressionLevel());
|
||||
os << '\n';
|
||||
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
|
||||
}
|
||||
os << "</DataArray>" << std::endl;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SaveCoeffFieldVTU(std::ostream &os, int ref_,
|
||||
const std::string &name, Coefficient &coeff)
|
||||
{
|
||||
RefinedGeometry *RefG;
|
||||
real_t val;
|
||||
std::vector<char> buf;
|
||||
int vec_dim = 1;
|
||||
os << "<DataArray type=\"" << GetDataTypeString()
|
||||
<< "\" Name=\"" << name
|
||||
<< "\" NumberOfComponents=\"" << vec_dim << "\""
|
||||
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
|
||||
{
|
||||
// scalar data
|
||||
if (!bdr_output)
|
||||
{
|
||||
for (int i = 0; i < mesh->GetNE(); i++)
|
||||
{
|
||||
RefG = GlobGeometryRefiner.Refine(
|
||||
mesh->GetElementBaseGeometry(i), ref_, 1);
|
||||
|
||||
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
|
||||
const IntegrationRule *ir = &RefG->RefPts;
|
||||
for (int j = 0; j < ir->GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(j);
|
||||
eltrans->SetIntPoint(&ip);
|
||||
val = coeff.Eval(*eltrans, ip);
|
||||
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < mesh->GetNBE(); i++)
|
||||
{
|
||||
RefG = GlobGeometryRefiner.Refine(
|
||||
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
|
||||
|
||||
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
|
||||
const IntegrationRule *ir = &RefG->RefPts;
|
||||
for (int j = 0; j < ir->GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(j);
|
||||
eltrans->SetIntPoint(&ip);
|
||||
val = coeff.Eval(*eltrans, ip);
|
||||
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (pv_data_format != VTKFormat::ASCII)
|
||||
{
|
||||
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
|
||||
}
|
||||
os << "</DataArray>" << std::endl;
|
||||
}
|
||||
|
||||
void ParaViewDataCollection::SaveVCoeffFieldVTU(std::ostream &os, int ref_,
|
||||
const std::string &name, VectorCoefficient &coeff)
|
||||
{
|
||||
RefinedGeometry *RefG;
|
||||
Vector val;
|
||||
std::vector<char> buf;
|
||||
int vec_dim = coeff.GetVDim();
|
||||
os << "<DataArray type=\"" << GetDataTypeString()
|
||||
<< "\" Name=\"" << name
|
||||
<< "\" NumberOfComponents=\"" << vec_dim << "\""
|
||||
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
|
||||
{
|
||||
// vector data
|
||||
if (!bdr_output)
|
||||
{
|
||||
for (int i = 0; i < mesh->GetNE(); i++)
|
||||
{
|
||||
RefG = GlobGeometryRefiner.Refine(
|
||||
mesh->GetElementBaseGeometry(i), ref_, 1);
|
||||
|
||||
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
|
||||
const IntegrationRule *ir = &RefG->RefPts;
|
||||
for (int j = 0; j < ir->GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(j);
|
||||
eltrans->SetIntPoint(&ip);
|
||||
coeff.Eval(val, *eltrans, ip);
|
||||
for (int jj = 0; jj < val.Size(); jj++)
|
||||
{
|
||||
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
|
||||
}
|
||||
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < mesh->GetNBE(); i++)
|
||||
{
|
||||
RefG = GlobGeometryRefiner.Refine(
|
||||
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
|
||||
|
||||
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
|
||||
const IntegrationRule *ir = &RefG->RefPts;
|
||||
for (int j = 0; j < ir->GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(j);
|
||||
eltrans->SetIntPoint(&ip);
|
||||
coeff.Eval(val, *eltrans, ip);
|
||||
for (int jj = 0; jj < val.Size(); jj++)
|
||||
{
|
||||
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
|
||||
}
|
||||
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (pv_data_format != VTKFormat::ASCII)
|
||||
{
|
||||
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
|
||||
}
|
||||
os << "</DataArray>" << std::endl;
|
||||
}
|
||||
|
||||
+47
-11
@@ -133,6 +133,7 @@ private:
|
||||
|
||||
/// A collection of named QuadratureFunctions
|
||||
typedef NamedFieldsMap<QuadratureFunction> QFieldMap;
|
||||
|
||||
public:
|
||||
typedef GFieldMap::MapType FieldMapType;
|
||||
typedef GFieldMap::iterator FieldMapIterator;
|
||||
@@ -249,10 +250,9 @@ public:
|
||||
{ field_map.Deregister(field_name, own_data); }
|
||||
|
||||
/// Add a QuadratureFunction to the collection.
|
||||
virtual void RegisterQField(const std::string& q_field_name,
|
||||
virtual void RegisterQField(const std::string& field_name,
|
||||
QuadratureFunction *qf)
|
||||
{ q_field_map.Register(q_field_name, qf, own_data); }
|
||||
|
||||
{ q_field_map.Register(field_name, qf, own_data); }
|
||||
|
||||
/// Remove a QuadratureFunction from the collection
|
||||
virtual void DeregisterQField(const std::string& field_name)
|
||||
@@ -280,13 +280,13 @@ public:
|
||||
#endif
|
||||
|
||||
/// Check if a QuadratureFunction with the given name is in the collection.
|
||||
bool HasQField(const std::string& q_field_name) const
|
||||
{ return q_field_map.Has(q_field_name); }
|
||||
bool HasQField(const std::string& field_name) const
|
||||
{ return q_field_map.Has(field_name); }
|
||||
|
||||
/// Get a pointer to a QuadratureFunction in the collection.
|
||||
/** Returns NULL if @a field_name is not in the collection. */
|
||||
QuadratureFunction *GetQField(const std::string& q_field_name)
|
||||
{ return q_field_map.Get(q_field_name); }
|
||||
QuadratureFunction *GetQField(const std::string& field_name)
|
||||
{ return q_field_map.Get(field_name); }
|
||||
|
||||
/// Get a const reference to the internal field map.
|
||||
/** The keys in the map are the field names and the values are pointers to
|
||||
@@ -302,11 +302,13 @@ public:
|
||||
|
||||
/// Get a pointer to the mesh in the collection
|
||||
Mesh *GetMesh() { return mesh; }
|
||||
|
||||
/// Set/change the mesh associated with the collection
|
||||
/** When passed a Mesh, assumes the serial case: MPI rank id is set to 0 and
|
||||
MPI num_procs is set to 1. When passed a ParMesh, MPI info from the
|
||||
ParMesh is used to set the DataCollection's MPI rank and num_procs. */
|
||||
virtual void SetMesh(Mesh *new_mesh);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Set/change the mesh associated with the collection.
|
||||
/** For this case, @a comm is used to set the DataCollection's MPI rank id
|
||||
@@ -369,8 +371,7 @@ public:
|
||||
/// Save one field, assuming the collection directory already exists.
|
||||
virtual void SaveField(const std::string &field_name);
|
||||
/// Save one q-field, assuming the collection directory already exists.
|
||||
virtual void SaveQField(const std::string &q_field_name);
|
||||
|
||||
virtual void SaveQField(const std::string &field_name);
|
||||
/// Load the collection. Not implemented in the base class DataCollection.
|
||||
virtual void Load(int cycle_ = 0);
|
||||
|
||||
@@ -510,7 +511,9 @@ protected:
|
||||
int compression_level = -1;
|
||||
bool high_order_output = false;
|
||||
bool restart_mode = false;
|
||||
bool bdr_output = false;
|
||||
VTKFormat pv_data_format = VTKFormat::BINARY;
|
||||
|
||||
public:
|
||||
ParaViewDataCollectionBase(const std::string &name, Mesh *mesh);
|
||||
|
||||
@@ -543,6 +546,10 @@ public:
|
||||
/// Reading high-order data requires ParaView 5.5 or later.
|
||||
void SetHighOrderOutput(bool high_order_output_);
|
||||
|
||||
/// @brief Configures collection to save only fields evaluated on boundaries of
|
||||
/// the mesh.
|
||||
void SetBoundaryOutput(bool bdr_output_);
|
||||
|
||||
/// If compression is enabled, return the compression level, else return 0.
|
||||
int GetCompressionLevel() const;
|
||||
|
||||
@@ -564,8 +571,6 @@ public:
|
||||
///
|
||||
/// If restart is enabled, new writes will preserve timestep metadata for any
|
||||
/// solutions prior to the currently defined time.
|
||||
///
|
||||
/// Initially, restart mode is disabled.
|
||||
void UseRestartMode(bool restart_mode_);
|
||||
};
|
||||
|
||||
@@ -575,11 +580,23 @@ class ParaViewDataCollection : public ParaViewDataCollectionBase
|
||||
private:
|
||||
std::fstream pvd_stream;
|
||||
|
||||
/// A collection of named Coefficients and VectorCoefficients
|
||||
using CoeffFieldMap = NamedFieldsMap<Coefficient>;
|
||||
using VCoeffFieldMap = NamedFieldsMap<VectorCoefficient>;
|
||||
|
||||
/** A FieldMap mapping registered names to Coefficient and VectorCoefficient
|
||||
pointers. */
|
||||
CoeffFieldMap coeff_field_map;
|
||||
VCoeffFieldMap vcoeff_field_map;
|
||||
protected:
|
||||
void WritePVTUHeader(std::ostream &out);
|
||||
void WritePVTUFooter(std::ostream &out, const std::string &vtu_prefix);
|
||||
void SaveDataVTU(std::ostream &out, int ref);
|
||||
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
|
||||
void SaveCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
|
||||
Coefficient &coeff);
|
||||
void SaveVCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
|
||||
VectorCoefficient& coeff);
|
||||
const char *GetDataFormatString() const;
|
||||
const char *GetDataTypeString() const;
|
||||
|
||||
@@ -598,6 +615,25 @@ public:
|
||||
ParaViewDataCollection(const std::string& collection_name,
|
||||
Mesh *mesh_ = nullptr);
|
||||
|
||||
/// Get a const reference to the internal coefficient-field map.
|
||||
const typename CoeffFieldMap::MapType &GetCoeffFieldMap() const
|
||||
{ return coeff_field_map.GetMap(); }
|
||||
const typename VCoeffFieldMap::MapType &GetVCoeffFieldMap() const
|
||||
{ return vcoeff_field_map.GetMap(); }
|
||||
|
||||
/// Add a Coefficient or VectorCoefficient to the collection.
|
||||
void RegisterCoeffField(const std::string& field_name, Coefficient *coeff)
|
||||
{ coeff_field_map.Register(field_name, coeff, own_data); }
|
||||
void RegisterVCoeffField(const std::string& field_name,
|
||||
VectorCoefficient *vcoeff)
|
||||
{ vcoeff_field_map.Register(field_name, vcoeff, own_data); }
|
||||
|
||||
/// Remove a Coefficient or VectorCoefficient from the collection
|
||||
void DeregisterCoeffField(const std::string& field_name)
|
||||
{ coeff_field_map.Deregister(field_name, own_data); }
|
||||
void DeregisterVCoeffField(const std::string& field_name)
|
||||
{ vcoeff_field_map.Deregister(field_name, own_data); }
|
||||
|
||||
/// Save the collection - the directory name is constructed based on the
|
||||
/// cycle value
|
||||
void Save() override;
|
||||
|
||||
+145
-32
@@ -211,8 +211,8 @@ private:
|
||||
///
|
||||
/// The operator is constructed with solution fields that it will act on and
|
||||
/// parameter fields that define coefficients. Quadrature functions are added by
|
||||
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates f
|
||||
/// those functionas and parameters at quadrature points.
|
||||
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates
|
||||
/// those functions and parameters at quadrature points.
|
||||
///
|
||||
/// Derivatives can be computed by obtaining a DerivativeOperator using
|
||||
/// GetDerivative().
|
||||
@@ -280,6 +280,22 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief Add an integrator to the operator.
|
||||
/// Called only from AddDomainIntegrator() and AddBoundaryIntegrator().
|
||||
template <
|
||||
typename entity_t,
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t>
|
||||
void AddIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &attributes,
|
||||
derivative_ids_t derivative_ids);
|
||||
|
||||
/// @brief Add a domain integrator to the operator.
|
||||
///
|
||||
/// @param qfunc The quadrature function to be added.
|
||||
@@ -305,6 +321,31 @@ public:
|
||||
const Array<int> &domain_attributes,
|
||||
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
|
||||
|
||||
/// @brief Add a boundary integrator to the operator.
|
||||
///
|
||||
/// @param qfunc The quadrature function to be added.
|
||||
/// @param inputs Tuple of FieldOperators for the inputs of the quadrature
|
||||
/// function.
|
||||
/// @param outputs Tuple of FieldOperators for the outputs of the quadrature
|
||||
/// function.
|
||||
/// @param integration_rule IntegrationRule to use with this integrator.
|
||||
/// @param boundary_attributes Boundary attributes marker array indicating over
|
||||
/// which attributes this integrator will integrate over.
|
||||
/// @param derivative_ids Derivatives to be made available for this
|
||||
/// integrator.
|
||||
template <
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t = decltype(std::make_index_sequence<0> {})>
|
||||
void AddBoundaryIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &boundary_attributes,
|
||||
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
|
||||
|
||||
/// @brief Set the parameters for the operator.
|
||||
///
|
||||
/// This has to be called before using Mult() or MultTranspose().
|
||||
@@ -423,7 +464,52 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
const Array<int> &domain_attributes,
|
||||
derivative_ids_t derivative_ids)
|
||||
{
|
||||
using entity_t = Entity::Element;
|
||||
AddIntegrator<Entity::Element>(
|
||||
qfunc, inputs, outputs, integration_rule, domain_attributes, derivative_ids);
|
||||
}
|
||||
|
||||
template <
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t>
|
||||
void DifferentiableOperator::AddBoundaryIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &boundary_attributes,
|
||||
derivative_ids_t derivative_ids)
|
||||
{
|
||||
|
||||
if (mesh.GetNFbyType(FaceType::Boundary) != mesh.GetNBE())
|
||||
{
|
||||
MFEM_ABORT("AddBoundaryIntegrator on meshes with interior boundaries is not supported.");
|
||||
}
|
||||
AddIntegrator<Entity::BoundaryElement>(
|
||||
qfunc, inputs, outputs, integration_rule, boundary_attributes, derivative_ids);
|
||||
}
|
||||
|
||||
template <
|
||||
typename entity_t,
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_t,
|
||||
typename derivative_ids_t>
|
||||
void DifferentiableOperator::AddIntegrator(
|
||||
qfunc_t &qfunc,
|
||||
input_t inputs,
|
||||
output_t outputs,
|
||||
const IntegrationRule &integration_rule,
|
||||
const Array<int> &attributes,
|
||||
derivative_ids_t derivative_ids)
|
||||
{
|
||||
if constexpr (!(std::is_same_v<entity_t, Entity::Element> ||
|
||||
std::is_same_v<entity_t, Entity::BoundaryElement>))
|
||||
{
|
||||
static_assert(dfem::always_false<entity_t>,
|
||||
"entity type not supported in AddIntegrator");
|
||||
}
|
||||
|
||||
static constexpr size_t num_inputs =
|
||||
tuple_size<decltype(inputs)>::value;
|
||||
@@ -477,32 +563,44 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
auto output_to_field =
|
||||
create_descriptors_to_fields_map<entity_t>(fields, outputs);
|
||||
|
||||
// TODO: factor out
|
||||
std::vector<int> inputs_vdim(num_inputs);
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
const Array<int> *elem_attributes = nullptr;
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
inputs_vdim[i] = get<i>(inputs).vdim;
|
||||
});
|
||||
|
||||
|
||||
Array<int> elem_attributes;
|
||||
elem_attributes.SetSize(mesh.GetNE());
|
||||
for (int i = 0; i < mesh.GetNE(); ++i)
|
||||
elem_attributes = &mesh.GetElementAttributes();
|
||||
}
|
||||
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
|
||||
{
|
||||
elem_attributes[i] = mesh.GetAttribute(i);
|
||||
elem_attributes = &mesh.GetBdrFaceAttributes();
|
||||
}
|
||||
|
||||
const auto output_fop = get<0>(outputs);
|
||||
test_space_field_idx = FindIdx(output_fop.GetFieldId(), fields);
|
||||
|
||||
bool use_sum_factorization = false;
|
||||
auto entity_element_type =
|
||||
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
|
||||
if ((entity_element_type == Element::QUADRILATERAL ||
|
||||
entity_element_type == Element::HEXAHEDRON) &&
|
||||
use_tensor_product_structure == true)
|
||||
Element::Type entity_element_type;
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
use_sum_factorization = true;
|
||||
entity_element_type =
|
||||
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
|
||||
|
||||
if ((entity_element_type == Element::QUADRILATERAL ||
|
||||
entity_element_type == Element::HEXAHEDRON) &&
|
||||
use_tensor_product_structure == true)
|
||||
{
|
||||
use_sum_factorization = true;
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
|
||||
{
|
||||
entity_element_type =
|
||||
Element::TypeFromGeometry(mesh.GetTypicalFaceGeometry());
|
||||
|
||||
if ((entity_element_type == Element::SEGMENT ||
|
||||
entity_element_type == Element::QUADRILATERAL) &&
|
||||
use_tensor_product_structure == true)
|
||||
{
|
||||
use_sum_factorization = true;
|
||||
}
|
||||
}
|
||||
|
||||
ElementDofOrdering element_dof_ordering = ElementDofOrdering::NATIVE;
|
||||
@@ -540,8 +638,17 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
prolongation_transpose = get_prolongation_transpose(
|
||||
fields[test_space_field_idx], output_fop, mesh.GetComm());
|
||||
|
||||
const int dimension = mesh.Dimension();
|
||||
[[maybe_unused]] const int num_elements = GetNumEntities<Entity::Element>(mesh);
|
||||
int dimension;
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
dimension = mesh.Dimension();
|
||||
}
|
||||
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
|
||||
{
|
||||
dimension = mesh.Dimension() - 1;
|
||||
}
|
||||
|
||||
[[maybe_unused]] const int num_elements = GetNumEntities<entity_t>(mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(mesh);
|
||||
const int num_qp = integration_rule.GetNPoints();
|
||||
|
||||
@@ -615,6 +722,12 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
thread_blocks.z = 1;
|
||||
}
|
||||
}
|
||||
else if (dimension == 1)
|
||||
{
|
||||
thread_blocks.x = q1d;
|
||||
thread_blocks.y = 1;
|
||||
thread_blocks.z = 1;
|
||||
}
|
||||
|
||||
action_callbacks.push_back(
|
||||
// Explicitly capture everything we need, so we can make explicit choice
|
||||
@@ -630,7 +743,7 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
test_vdim, // int (= output_fop.vdim)
|
||||
test_op_dim, // int (derived from output_fop)
|
||||
inputs, // mfem::future::tuple
|
||||
domain_attributes, // Array<int>
|
||||
attributes, // Array<int>
|
||||
ir_weights, // DeviceTensor
|
||||
use_sum_factorization, // bool
|
||||
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
@@ -663,13 +776,13 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
action_shmem_info.field_sizes,
|
||||
num_entities);
|
||||
|
||||
const bool has_attr = domain_attributes.Size() > 0;
|
||||
const auto d_domain_attr = domain_attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes.Read();
|
||||
const bool has_attr = attributes.Size() > 0;
|
||||
const auto d_attr = attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes->Read();
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
|
||||
{
|
||||
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
|
||||
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, input_shmem,
|
||||
residual_shmem, scratch_shmem] =
|
||||
@@ -739,7 +852,7 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
test_vdim, // int (= output_fop.vdim)
|
||||
test_op_dim, // int (derived from output_fop)
|
||||
inputs, // mfem::future::tuple
|
||||
domain_attributes, // Array<int>
|
||||
attributes, // Array<int>
|
||||
ir_weights, // DeviceTensor
|
||||
use_sum_factorization, // bool
|
||||
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
|
||||
@@ -777,14 +890,14 @@ void DifferentiableOperator::AddDomainIntegrator(
|
||||
shmem_info.direction_size,
|
||||
num_entities);
|
||||
|
||||
const auto d_elem_attr = elem_attributes.Read();
|
||||
const bool has_attr = domain_attributes.Size() > 0;
|
||||
const auto d_domain_attr = domain_attributes.Read();
|
||||
const bool has_attr = attributes.Size() > 0;
|
||||
const auto d_attr = attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes->Read();
|
||||
|
||||
derivative_action_e = 0.0;
|
||||
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
|
||||
{
|
||||
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
|
||||
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
|
||||
direction_shmem, input_shmem,
|
||||
|
||||
+84
-1
@@ -95,6 +95,85 @@ void map_quadrature_data_to_fields_impl(
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_tensor_impl_1d(
|
||||
DeviceTensor<2, real_t> &y,
|
||||
const DeviceTensor<3, real_t> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, vdim);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
acc += fqp(vd, 0, qx) * B(qx, 0, dx);
|
||||
}
|
||||
yd(dx, vd) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, vdim);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
acc += fqp(vd, 0, qx) * G(qx, 0, dx);
|
||||
}
|
||||
yd(dx, vd) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d);
|
||||
|
||||
for (int sq = 0; sq < output.size_on_qp; sq++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
yqp(sq, qx) = fqp(sq, qx);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor with sum factorization on tensor product elements");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_tensor_impl_2d(
|
||||
@@ -431,7 +510,11 @@ void map_quadrature_data_to_fields(
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
map_quadrature_data_to_fields_tensor_impl_1d(y, f, output, dtq, scratch_mem);
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
map_quadrature_data_to_fields_tensor_impl_2d(y, f, output, dtq, scratch_mem);
|
||||
}
|
||||
|
||||
+109
-5
@@ -338,6 +338,92 @@ void map_field_to_quadrature_data_tensor_product_2d(
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_field_to_quadrature_data_tensor_product_1d(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const field_operator_t &input,
|
||||
const DeviceTensor<1, const real_t> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
[[maybe_unused]] auto B = dtq.B;
|
||||
[[maybe_unused]] auto G = dtq.G;
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
acc += B(qx, 0, dx) * field(dx, vd);
|
||||
}
|
||||
fqp(vd, qx) = acc;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&field_e[0], d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t acc = 0.0;
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
acc += G(qx, 0, dx) * field(dx, vd);
|
||||
}
|
||||
fqp(vd, 0, qx) = acc;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
auto w = Reshape(&integration_weights[0], q1d);
|
||||
auto f = Reshape(&field_qp[0], q1d);
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
f(qx) = w(qx);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (is_identity_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const int q1d = B.GetShape()[0];
|
||||
auto field = Reshape(&field_e[0], input.size_on_qp, q1d);
|
||||
field_qp = field;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<std::decay_t<field_operator_t>>,
|
||||
"can't map field to quadrature data");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_field_to_quadrature_data(
|
||||
@@ -444,7 +530,13 @@ void map_fields_to_quadrature_data(
|
||||
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_1d(
|
||||
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
|
||||
@@ -489,14 +581,20 @@ void map_field_to_quadrature_data_conditional(
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
map_field_to_quadrature_data_tensor_product_1d(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
|
||||
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
|
||||
}
|
||||
}
|
||||
@@ -547,7 +645,13 @@ void map_direction_to_quadrature_data_conditional(
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_1d(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_2d(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
|
||||
@@ -44,7 +44,16 @@ void call_qfunction(
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q, x, q1d)
|
||||
{
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), rs_qp);
|
||||
apply_kernel(r, qfunc, qf_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
@@ -123,7 +132,22 @@ void call_qfunction_derivative_action(
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
{
|
||||
if (dimension == 2)
|
||||
if (dimension == 1)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q, x, q1d)
|
||||
{
|
||||
auto r = Reshape(&residual_shmem(0, q), das_qp);
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
|
||||
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
|
||||
shadow_shmem, q);
|
||||
#else
|
||||
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
else if (dimension == 2)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
@@ -164,7 +188,10 @@ void call_qfunction_derivative_action(
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
else
|
||||
{
|
||||
MFEM_ABORT_KERNEL("unsupported dimension");
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -180,8 +207,8 @@ void call_qfunction_derivative_action(
|
||||
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
|
||||
#endif
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename args_ts, size_t num_args>
|
||||
|
||||
@@ -44,7 +44,7 @@ void process_qf_arg(
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i).value = u((i * m) + j);
|
||||
arg(j, i).value = u((i * n) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -94,8 +94,8 @@ void process_qf_arg(
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i).value = u((i * m) + j);
|
||||
arg(j, i).gradient = v((i * m) + j);
|
||||
arg(j, i).value = u((i * n) + j);
|
||||
arg(j, i).gradient = v((i * n) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -181,6 +181,14 @@ void process_derivative_from_native_dual(
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_derivative_from_native_dual(
|
||||
DeviceTensor<1, T> &r,
|
||||
const dual<T, T> &x)
|
||||
{
|
||||
r(0) = x.gradient;
|
||||
}
|
||||
|
||||
template <typename T0, typename T1>
|
||||
MFEM_HOST_DEVICE inline
|
||||
@@ -230,7 +238,7 @@ void process_qf_arg(
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
arg(j, i) = u((i * n) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -330,7 +338,7 @@ void process_qf_arg(
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
arg(j, i) = u((i * n) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+3
-3
@@ -454,7 +454,7 @@ MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
|
||||
*
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @tparam i integer sequence used to index the tuples
|
||||
* @param x tuple of values to be subracted from
|
||||
* @param x tuple of values to be subtracted from
|
||||
* @param y tuple of values to subtract from x
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
@@ -596,7 +596,7 @@ MFEM_HOST_DEVICE constexpr auto div_helper(const real_t a,
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a the constant denomenator
|
||||
* @param a the constant denominator
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
@@ -726,7 +726,7 @@ MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const real_t a)
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @tparam i a list of indices used to acces each element of the tuple
|
||||
* @tparam i a list of indices used to access each element of the tuple
|
||||
* @param out the ostream to write the output to
|
||||
* @param A the tuple of values
|
||||
* @brief helper used to implement printing a tuple of values
|
||||
|
||||
+45
-3
@@ -944,7 +944,44 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>, "can't use GetElementRestriction on type");
|
||||
static_assert(dfem::always_false<T>,
|
||||
"can't use get_element_restriction on type");
|
||||
}
|
||||
return nullptr; // Unreachable, but avoids compiler warning
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
/// @brief Get the face restriction operator for a field descriptor.
|
||||
///
|
||||
/// @param f the field descriptor.
|
||||
/// @param o the face dof ordering.
|
||||
/// @param ft the face type
|
||||
/// @param m indicator if single or double valued
|
||||
/// @returns the face restriction operator for the field descriptor in
|
||||
/// specified ordering.
|
||||
inline
|
||||
const Operator *get_face_restriction(const FieldDescriptor &f,
|
||||
ElementDofOrdering o,
|
||||
FaceType ft,
|
||||
L2FaceValues m)
|
||||
{
|
||||
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
|
||||
std::is_same_v<T, const ParFiniteElementSpace *>)
|
||||
{
|
||||
return arg->GetFaceRestriction(o, ft, m);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
// ParameterSpace does not support face restrictions
|
||||
MFEM_ABORT("internal error");
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>,
|
||||
"can't use get_face_restriction on type");
|
||||
}
|
||||
return nullptr; // Unreachable, but avoids compiler warning
|
||||
}, f.data);
|
||||
@@ -965,6 +1002,11 @@ const Operator *get_restriction(const FieldDescriptor &f,
|
||||
{
|
||||
return get_element_restriction(f, o);
|
||||
}
|
||||
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
|
||||
{
|
||||
return get_face_restriction(f, o, FaceType::Boundary,
|
||||
L2FaceValues::SingleValued);
|
||||
}
|
||||
MFEM_ABORT("restriction not implemented for Entity");
|
||||
return nullptr;
|
||||
}
|
||||
@@ -974,7 +1016,7 @@ const Operator *get_restriction(const FieldDescriptor &f,
|
||||
/// @param f the field descriptor.
|
||||
/// @param o the element dof ordering.
|
||||
/// @param fop the field operator.
|
||||
/// @returns a tuple containting a std::function with the transpose
|
||||
/// @returns a tuple containing a std::function with the transpose
|
||||
/// restriction callback and it's height.
|
||||
template <typename entity_t, typename fop_t>
|
||||
inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
|
||||
@@ -1389,7 +1431,7 @@ create_descriptors_to_fields_map(
|
||||
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
|
||||
{
|
||||
// TODO-bug: stealing dimension from the first field
|
||||
fop.dim = GetDimension<Entity::Element>(fields[0]);
|
||||
fop.dim = GetDimension<entity_t>(fields[0]);
|
||||
fop.vdim = 1;
|
||||
fop.size_on_qp = 1;
|
||||
map = -1;
|
||||
|
||||
+59
-42
@@ -661,65 +661,78 @@ void ScalarFiniteElement::ScalarLocalL2Restriction(
|
||||
void NodalFiniteElement::CreateLexicographicFullMap(const IntegrationRule &ir)
|
||||
const
|
||||
{
|
||||
// Get the FULL version of the map.
|
||||
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
|
||||
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
|
||||
auto *d2q_new = new DofToQuad(d2q);
|
||||
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
|
||||
const int nqpt = ir.GetNPoints();
|
||||
|
||||
const int b_dim = (range_type == VECTOR) ? dim : 1;
|
||||
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int d = 0; d < b_dim; d++)
|
||||
// Get the FULL version of the map.
|
||||
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
|
||||
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
|
||||
auto *d2q_new = new DofToQuad(d2q);
|
||||
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
|
||||
const int nqpt = ir.GetNPoints();
|
||||
|
||||
const int b_dim = (range_type == VECTOR) ? dim : 1;
|
||||
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
for (int d = 0; d < b_dim; d++)
|
||||
{
|
||||
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
|
||||
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
|
||||
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
|
||||
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
|
||||
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const int g_dim = [this]()
|
||||
{
|
||||
switch (deriv_type)
|
||||
const int g_dim = [this]()
|
||||
{
|
||||
case GRAD: return dim;
|
||||
case DIV: return 1;
|
||||
case CURL: return cdim;
|
||||
default: return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
for (int d = 0; d < g_dim; d++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
switch (deriv_type)
|
||||
{
|
||||
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
|
||||
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
|
||||
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
|
||||
case GRAD: return dim;
|
||||
case DIV: return 1;
|
||||
case CURL: return cdim;
|
||||
default: return 0;
|
||||
}
|
||||
}();
|
||||
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
for (int d = 0; d < g_dim; d++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
|
||||
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
|
||||
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dof2quad_array.Append(d2q_new);
|
||||
dof2quad_array.Append(d2q_new);
|
||||
}
|
||||
}
|
||||
|
||||
const DofToQuad &NodalFiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const
|
||||
{
|
||||
//Should make this loop a function of FiniteElement
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
DofToQuad *d2q = nullptr;
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
const DofToQuad &d2q = *dof2quad_array[i];
|
||||
if (d2q.IntRule == &ir && d2q.mode == mode) { return d2q; }
|
||||
//Should make this loop a function of FiniteElement
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule == &ir && d2q->mode == mode) { break; }
|
||||
d2q = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
if (d2q) { return *d2q; }
|
||||
if (mode != DofToQuad::LEXICOGRAPHIC_FULL)
|
||||
{
|
||||
return FiniteElement::GetDofToQuad(ir, mode);
|
||||
@@ -2620,8 +2633,12 @@ const DofToQuad &TensorBasisElement::GetTensorDofToQuad(
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
auto* d2q_ = dof2quad_array[i];
|
||||
if (d2q_->IntRule == &ir && d2q_->mode == mode)
|
||||
{
|
||||
d2q = d2q_;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!d2q)
|
||||
{
|
||||
|
||||
+32
-8
@@ -308,13 +308,25 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
|
||||
FiniteElement::INTEGRAL,
|
||||
BasisType::GetType(name[12]));
|
||||
}
|
||||
else if (!strncmp(name, "RT_R1D",6))
|
||||
else if (!strncmp(name, "RT_R1D_", 7))
|
||||
{
|
||||
fec = new RT_R1D_FECollection(atoi(name+11),atoi(name + 7));
|
||||
fec = new RT_R1D_FECollection(atoi(name + 11), atoi(name + 7));
|
||||
}
|
||||
else if (!strncmp(name, "RT_R2D",6))
|
||||
else if (!strncmp(name, "RT_R1D@", 7))
|
||||
{
|
||||
fec = new RT_R2D_FECollection(atoi(name+11),atoi(name + 7));
|
||||
fec = new RT_R1D_FECollection(atoi(name + 14), atoi(name + 10),
|
||||
BasisType::GetType(name[7]),
|
||||
BasisType::GetType(name[8]));
|
||||
}
|
||||
else if (!strncmp(name, "RT_R2D_", 7))
|
||||
{
|
||||
fec = new RT_R2D_FECollection(atoi(name + 11), atoi(name + 7));
|
||||
}
|
||||
else if (!strncmp(name, "RT_R2D@", 7))
|
||||
{
|
||||
fec = new RT_R2D_FECollection(atoi(name + 14), atoi(name + 10),
|
||||
BasisType::GetType(name[7]),
|
||||
BasisType::GetType(name[8]));
|
||||
}
|
||||
else if (!strncmp(name, "RT_", 3))
|
||||
{
|
||||
@@ -336,13 +348,25 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
|
||||
BasisType::GetType(name[9]),
|
||||
BasisType::GetType(name[10]));
|
||||
}
|
||||
else if (!strncmp(name, "ND_R1D",6))
|
||||
else if (!strncmp(name, "ND_R1D_", 7))
|
||||
{
|
||||
fec = new ND_R1D_FECollection(atoi(name+11),atoi(name + 7));
|
||||
fec = new ND_R1D_FECollection(atoi(name + 11), atoi(name + 7));
|
||||
}
|
||||
else if (!strncmp(name, "ND_R2D",6))
|
||||
else if (!strncmp(name, "ND_R1D@", 7))
|
||||
{
|
||||
fec = new ND_R2D_FECollection(atoi(name+11),atoi(name + 7));
|
||||
fec = new ND_R1D_FECollection(atoi(name + 14), atoi(name + 10),
|
||||
BasisType::GetType(name[7]),
|
||||
BasisType::GetType(name[8]));
|
||||
}
|
||||
else if (!strncmp(name, "ND_R2D_", 7))
|
||||
{
|
||||
fec = new ND_R2D_FECollection(atoi(name + 11), atoi(name + 7));
|
||||
}
|
||||
else if (!strncmp(name, "ND_R2D@", 7))
|
||||
{
|
||||
fec = new ND_R2D_FECollection(atoi(name + 14), atoi(name + 10),
|
||||
BasisType::GetType(name[7]),
|
||||
BasisType::GetType(name[8]));
|
||||
}
|
||||
else if (!strncmp(name, "ND_", 3))
|
||||
{
|
||||
|
||||
@@ -120,7 +120,9 @@ public:
|
||||
| ND_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
|
||||
| ND_Trace@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
|
||||
| ND_R1D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
|
||||
| ND_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
|
||||
| ND_R2D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
|
||||
| ND_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
|
||||
| RT_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | Raviart-Thomas vector elements |
|
||||
| RT@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | Raviart-Thomas vector elements |
|
||||
| RT_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
|
||||
@@ -128,7 +130,9 @@ public:
|
||||
| RT_Trace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
|
||||
| RT_ValTrace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | VALUE | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
|
||||
| RT_R1D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
|
||||
| RT_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
|
||||
| RT_R2D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
|
||||
| RT_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
|
||||
| L2_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
|
||||
| L2_T[BTYPE]_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
|
||||
| L2Int_[DIM]_[ORDER] | L2 | * | 0 | INTEGRAL | Discontinuous L2 elements |
|
||||
|
||||
@@ -769,8 +769,8 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
u += Gt[dx][qx] * QQ0[qy][qx];
|
||||
v += Bt[dx][qx] * QQ1[qy][qx];
|
||||
}
|
||||
DQ0[qy][dx] = u;
|
||||
DQ1[qy][dx] = v;
|
||||
DQ0[dx][qy] = u;
|
||||
DQ1[dx][qy] = v;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
@@ -782,8 +782,8 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
real_t v = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += DQ0[qy][dx] * Bt[dy][qy];
|
||||
v += DQ1[qy][dx] * Gt[dy][qy];
|
||||
u += DQ0[dx][qy] * Bt[dy][qy];
|
||||
v += DQ1[dx][qy] * Gt[dy][qy];
|
||||
}
|
||||
Y(dx,dy,e) += (u + v);
|
||||
}
|
||||
|
||||
@@ -17,11 +17,9 @@
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
namespace mfem::internal
|
||||
{
|
||||
|
||||
// PA Diffusion Apply 2D kernel
|
||||
@@ -333,23 +331,8 @@ PAVectorDiffusionApply3D(const int NE, const Array<real_t> &b,
|
||||
}
|
||||
});
|
||||
}
|
||||
} // namespace internal
|
||||
} // namespace mfem::internal
|
||||
|
||||
template <int DIM, int VDIM, int T_D1D, int T_Q1D>
|
||||
VectorDiffusionIntegrator::ApplyKernelType
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2)
|
||||
{
|
||||
return internal::PAVectorDiffusionApply2D<T_D1D, T_Q1D, VDIM>;
|
||||
}
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
return internal::PAVectorDiffusionApply3D;
|
||||
}
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
#endif
|
||||
|
||||
#endif // MFEM_BILININTEG_VECDIFFUSION_KERNELS_HPP
|
||||
|
||||
@@ -9,13 +9,14 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
#include "bilininteg_vecdiffusion_kernels.hpp"
|
||||
#include "./bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
|
||||
|
||||
// #include "bilininteg_vecdiffusion_kernels.hpp"
|
||||
// #include "bilininteg_vecdiffusion_pa.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -23,7 +24,7 @@ namespace mfem
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(const IntegrationRule *ir)
|
||||
: BilinearFormIntegrator(ir)
|
||||
{
|
||||
static Kernels kernels;
|
||||
// static Kernels kernels;
|
||||
}
|
||||
|
||||
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q)
|
||||
@@ -67,210 +68,263 @@ VectorDiffusionIntegrator::VectorDiffusionIntegrator(MatrixCoefficient &mq)
|
||||
vdim = mq.GetVDim();
|
||||
}
|
||||
|
||||
// PA Diffusion Assemble 2D kernel
|
||||
static void PAVectorDiffusionSetup2D(const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &op)
|
||||
{
|
||||
const int NQ = Q1D*Q1D;
|
||||
auto W = w.Read();
|
||||
|
||||
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
|
||||
auto y = Reshape(op.Write(), NQ, 3, NE);
|
||||
|
||||
const bool const_c = c.Size() == 1;
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1) :
|
||||
Reshape(c.Read(), NQ, NE);
|
||||
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
const real_t J11 = J(q,0,0,e);
|
||||
const real_t J21 = J(q,1,0,e);
|
||||
const real_t J12 = J(q,0,1,e);
|
||||
const real_t J22 = J(q,1,1,e);
|
||||
|
||||
const real_t C1 = const_c ? C(0,0) : C(q,e);
|
||||
const real_t c_detJ = W[q] * C1 / ((J11*J22)-(J21*J12));
|
||||
y(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
|
||||
y(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
|
||||
y(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Assemble 3D kernel
|
||||
static void PAVectorDiffusionSetup3D(const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &op)
|
||||
{
|
||||
const int NQ = Q1D*Q1D*Q1D;
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
auto y = Reshape(op.Write(), NQ, 6, NE);
|
||||
|
||||
const bool const_c = c.Size() == 1;
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1) :
|
||||
Reshape(c.Read(), NQ,NE);
|
||||
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
const real_t J11 = J(q,0,0,e);
|
||||
const real_t J21 = J(q,1,0,e);
|
||||
const real_t J31 = J(q,2,0,e);
|
||||
const real_t J12 = J(q,0,1,e);
|
||||
const real_t J22 = J(q,1,1,e);
|
||||
const real_t J32 = J(q,2,1,e);
|
||||
const real_t J13 = J(q,0,2,e);
|
||||
const real_t J23 = J(q,1,2,e);
|
||||
const real_t J33 = J(q,2,2,e);
|
||||
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
|
||||
const real_t C1 = const_c ? C(0,0) : C(q,e);
|
||||
|
||||
const real_t c_detJ = W[q] * C1 / detJ;
|
||||
// adj(J)
|
||||
const real_t A11 = (J22 * J33) - (J23 * J32);
|
||||
const real_t A12 = (J32 * J13) - (J12 * J33);
|
||||
const real_t A13 = (J12 * J23) - (J22 * J13);
|
||||
const real_t A21 = (J31 * J23) - (J21 * J33);
|
||||
const real_t A22 = (J11 * J33) - (J13 * J31);
|
||||
const real_t A23 = (J21 * J13) - (J11 * J23);
|
||||
const real_t A31 = (J21 * J32) - (J31 * J22);
|
||||
const real_t A32 = (J31 * J12) - (J11 * J32);
|
||||
const real_t A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
y(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
y(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
y(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
y(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
y(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
y(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorDiffusionSetup(const int dim,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &op)
|
||||
{
|
||||
if (!(dim == 2 || dim == 3))
|
||||
{
|
||||
MFEM_ABORT("Dimension not supported.");
|
||||
}
|
||||
if (dim == 2)
|
||||
{
|
||||
PAVectorDiffusionSetup2D(Q1D, NE, W, J, C, op);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
PAVectorDiffusionSetup3D(Q1D, NE, W, J, C, op);
|
||||
}
|
||||
}
|
||||
|
||||
void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assumes tensor-product elements
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
const IntegrationRule *ir
|
||||
= IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
|
||||
const auto *ir = IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
|
||||
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q);
|
||||
}
|
||||
const bool mixed =
|
||||
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
|
||||
if (mixed) { ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q); }
|
||||
else { ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q); }
|
||||
return;
|
||||
}
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
const int nq = ir->GetNPoints();
|
||||
|
||||
// If vdim is not set, set it to the space dimension
|
||||
vdim = (vdim == -1) ? fes.GetVDim() : vdim;
|
||||
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
|
||||
|
||||
const MemoryType mt = pa_mt == MemoryType::DEFAULT
|
||||
? Device::GetDeviceMemoryType()
|
||||
: pa_mt;
|
||||
|
||||
ne = fes.GetNE();
|
||||
dim = mesh->Dimension();
|
||||
sdim = mesh->SpaceDimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
|
||||
const int nq = ir->GetNPoints();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
|
||||
const int q1d = quad1D;
|
||||
|
||||
MFEM_VERIFY(!VQ && !MQ,
|
||||
"Only scalar coefficient supported for partial assembly for VectorDiffusionIntegrator");
|
||||
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
|
||||
CoefficientVector coeff(qs, CoefficientStorage::FULL);
|
||||
|
||||
if (Q)
|
||||
{
|
||||
coeff.Project(*Q);
|
||||
}
|
||||
else if (VQ)
|
||||
{
|
||||
coeff.Project(*VQ);
|
||||
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
|
||||
}
|
||||
else if (MQ)
|
||||
{
|
||||
coeff.ProjectTranspose(*MQ);
|
||||
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
|
||||
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
|
||||
}
|
||||
else { coeff.SetConstant(1.0); }
|
||||
|
||||
coeff_vdim = coeff.GetVDim();
|
||||
const bool scalar_coeff = coeff_vdim == 1;
|
||||
const bool vector_coeff = coeff_vdim == vdim;
|
||||
const bool matrix_coeff = coeff_vdim == vdim * vdim;
|
||||
MFEM_VERIFY(scalar_coeff + vector_coeff + matrix_coeff == 1, "");
|
||||
|
||||
const int pa_size = dim * dim;
|
||||
pa_data.SetSize(nq * pa_size * vdim * (matrix_coeff ? dim : 1) * ne, mt);
|
||||
|
||||
const Array<real_t> &w = ir->GetWeights();
|
||||
const Vector &j = geom->J;
|
||||
Vector &d = pa_data;
|
||||
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PAVectorDiffusionSetup"); }
|
||||
if (dim == 2 && sdim == 3)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int SDIM = 3;
|
||||
const int NQ = quad1D*quad1D;
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, SDIM, DIM, ne);
|
||||
auto D = Reshape(d.Write(), NQ, SDIM, ne);
|
||||
MFEM_VERIFY(scalar_coeff, "");
|
||||
const int nc = vdim;
|
||||
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
|
||||
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
|
||||
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
|
||||
auto D = Reshape(pa_data.Write(), q1d, q1d, pa_size,
|
||||
vdim * (matrix_coeff ? dim : 1), ne);
|
||||
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
const auto C = const_c ? Reshape(coeff.Read(), 1,1) :
|
||||
Reshape(coeff.Read(), NQ,ne);
|
||||
|
||||
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
|
||||
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
const real_t wq = W[q];
|
||||
const real_t J11 = J(q,0,0,e);
|
||||
const real_t J21 = J(q,1,0,e);
|
||||
const real_t J31 = J(q,2,0,e);
|
||||
const real_t J12 = J(q,0,1,e);
|
||||
const real_t J22 = J(q,1,1,e);
|
||||
const real_t J32 = J(q,2,1,e);
|
||||
const real_t E = J11*J11 + J21*J21 + J31*J31;
|
||||
const real_t G = J12*J12 + J22*J22 + J32*J32;
|
||||
const real_t F = J11*J12 + J21*J22 + J31*J32;
|
||||
const real_t iw = 1.0 / sqrt(E*G - F*F);
|
||||
const real_t C1 = const_c ? C(0,0) : C(q,e);
|
||||
const real_t alpha = wq * C1 * iw;
|
||||
D(q,0,e) = alpha * G; // 1,1
|
||||
D(q,1,e) = -alpha * F; // 1,2
|
||||
D(q,2,e) = alpha * E; // 2,2
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
for (int i = 0; i < nc; ++i)
|
||||
{
|
||||
const real_t wq = W(qx, qy);
|
||||
const real_t J11 = J(qx, qy, 0, 0, e);
|
||||
const real_t J21 = J(qx, qy, 1, 0, e);
|
||||
const real_t J31 = J(qx, qy, 2, 0, e);
|
||||
const real_t J12 = J(qx, qy, 0, 1, e);
|
||||
const real_t J22 = J(qx, qy, 1, 1, e);
|
||||
const real_t J32 = J(qx, qy, 2, 1, e);
|
||||
const real_t E = J11*J11 + J21*J21 + J31*J31;
|
||||
const real_t G = J12*J12 + J22*J22 + J32*J32;
|
||||
const real_t F = J11*J12 + J21*J22 + J31*J32;
|
||||
const real_t iw = 1.0 / sqrt(E*G - F*F);
|
||||
const auto C0 = C(0, qx, qy, e);
|
||||
const real_t alpha = wq * C0 * iw;
|
||||
D(qx, qy, 0, i, e) = alpha * G; // 1,1
|
||||
D(qx, qy, 1, i, e) = -alpha * F; // 1,2
|
||||
D(qx, qy, 2, i, e) = -alpha * F; // 2,1 == 1,2
|
||||
D(qx, qy, 3, i, e) = alpha * E; // 2,2
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else if (dim == 2 && sdim == 2)
|
||||
{
|
||||
const int nc = vdim, cvdim = coeff_vdim;
|
||||
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
|
||||
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
|
||||
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
|
||||
auto DE = Reshape(pa_data.Write(), q1d, q1d, pa_size,
|
||||
vdim * (matrix_coeff ? dim : 1), ne);
|
||||
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
const real_t J11 = J(qx, qy, 0, 0, e);
|
||||
const real_t J21 = J(qx, qy, 1, 0, e);
|
||||
const real_t J12 = J(qx, qy, 0, 1, e);
|
||||
const real_t J22 = J(qx, qy, 1, 1, e);
|
||||
const real_t w_detJ = W(qx, qy) / ((J11*J22)-(J21*J12));
|
||||
const real_t D0 = w_detJ * (J12*J12 + J22*J22);
|
||||
const real_t D1 = -w_detJ * (J12*J11 + J22*J21);
|
||||
const real_t D2 = w_detJ * (J11*J11 + J21*J21);
|
||||
const int map[4] = {0, 2, 1, 3};
|
||||
|
||||
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
|
||||
{
|
||||
const auto k = matrix_coeff ? map[i] : (vector_coeff ? i : 0);
|
||||
const auto Cc = C(k, qx, qy, e);
|
||||
DE(qx, qy, 0, i, e) = D0 * Cc;
|
||||
DE(qx, qy, 1, i, e) = D1 * Cc;
|
||||
DE(qx, qy, 2, i, e) = D1 * Cc;
|
||||
DE(qx, qy, 3, i, e) = D2 * Cc;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else if (dim == 3 && sdim == 3)
|
||||
{
|
||||
const int nc = vdim, cvdim = coeff_vdim;
|
||||
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d, q1d);
|
||||
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
|
||||
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
|
||||
auto DE = Reshape(pa_data.Write(), q1d, q1d, q1d, pa_size,
|
||||
vdim * (matrix_coeff ? dim : 1), ne);
|
||||
|
||||
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
const real_t J11 = J(qx, qy, qz, 0, 0, e);
|
||||
const real_t J21 = J(qx, qy, qz, 1, 0, e);
|
||||
const real_t J31 = J(qx, qy, qz, 2, 0, e);
|
||||
const real_t J12 = J(qx, qy, qz, 0, 1, e);
|
||||
const real_t J22 = J(qx, qy, qz, 1, 1, e);
|
||||
const real_t J32 = J(qx, qy, qz, 2, 1, e);
|
||||
const real_t J13 = J(qx, qy, qz, 0, 2, e);
|
||||
const real_t J23 = J(qx, qy, qz, 1, 2, e);
|
||||
const real_t J33 = J(qx, qy, qz, 2, 2, e);
|
||||
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
const real_t c_detJ = W(qx, qy, qz) / detJ;
|
||||
// adj(J)
|
||||
const real_t A11 = (J22 * J33) - (J23 * J32);
|
||||
const real_t A12 = (J32 * J13) - (J12 * J33);
|
||||
const real_t A13 = (J12 * J23) - (J22 * J13);
|
||||
const real_t A21 = (J31 * J23) - (J21 * J33);
|
||||
const real_t A22 = (J11 * J33) - (J13 * J31);
|
||||
const real_t A23 = (J21 * J13) - (J11 * J23);
|
||||
const real_t A31 = (J21 * J32) - (J31 * J22);
|
||||
const real_t A32 = (J31 * J12) - (J11 * J32);
|
||||
const real_t A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
const real_t D11 = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
const real_t D21 = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
const real_t D31 = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
const real_t D22 = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
const real_t D32 = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
const real_t D33 = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
const int map[9] = {0, 3, 6, 1, 4, 7, 2, 5, 8};
|
||||
|
||||
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
|
||||
{
|
||||
const auto k = matrix_coeff ? map[i] : vector_coeff ? i : 0;
|
||||
const auto Ck = C(k, qx, qy, qz, e);
|
||||
DE(qx, qy, qz, 0, i, e) = D11 * Ck;
|
||||
DE(qx, qy, qz, 1, i, e) = D21 * Ck;
|
||||
DE(qx, qy, qz, 2, i, e) = D31 * Ck;
|
||||
DE(qx, qy, qz, 3, i, e) = D22 * Ck;
|
||||
DE(qx, qy, qz, 4, i, e) = D32 * Ck;
|
||||
DE(qx, qy, qz, 5, i, e) = D33 * Ck;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorDiffusionSetup(dim, quad1D, ne, w, j, coeff, d);
|
||||
MFEM_ABORT("Unknown VectorDiffusionIntegrator::AssemblePA kernel for"
|
||||
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
|
||||
}
|
||||
}
|
||||
|
||||
// PA Diffusion Apply kernel
|
||||
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
// Use CEED backend if available
|
||||
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
|
||||
|
||||
// Add the VectorDiffusionAddMultPA specializations
|
||||
static const auto vector_diffusion_kernel_specializations =
|
||||
(
|
||||
// 2D, SDIM = 2
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 2,2>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 3,3>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 4,4>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 5,5>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 6,6>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 7,7>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 8,8>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 9,9>::Add(),
|
||||
// 2D, SDIM = 3
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 2,2>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 3,3>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 4,4>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 5,5>::Add(),
|
||||
// 3D, SDIM = 3
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,2>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,3>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 3,4>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,5>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,6>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,6>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,8>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 6,7>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 7,8>::Add(),
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 8,9>::Add(),
|
||||
true);
|
||||
MFEM_CONTRACT_VAR(vector_diffusion_kernel_specializations);
|
||||
|
||||
ApplyPAKernels::Run(dim, sdim, dofs1D, quad1D,
|
||||
ne, coeff_vdim, maps->B, maps->G, pa_data, x, y,
|
||||
sdim, dofs1D, quad1D);
|
||||
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAVectorDiffusionDiagonal2D(const int NE,
|
||||
const Array<real_t> &b,
|
||||
@@ -284,12 +338,15 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
// note the different shape for D, this is a (symmetric) matrix so we only
|
||||
// store necessary entries
|
||||
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
|
||||
MFEM_VERIFY(d.Size() == Q1D*Q1D*4*2*NE, "");
|
||||
const auto D = Reshape(d.Read(), Q1D*Q1D, /*3*/4, 2, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, 2, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -310,9 +367,9 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const int q = qx + qy * Q1D;
|
||||
const real_t D0 = D(q,0,e);
|
||||
const real_t D1 = D(q,1,e);
|
||||
const real_t D2 = D(q,2,e);
|
||||
const real_t D0 = D(q,0,0,e);
|
||||
const real_t D1 = D(q,1,0,e);
|
||||
const real_t D2 = D(q,3/*2*/,0,e); // size from 3 (symmetric) to 4 (dims x dims)
|
||||
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
|
||||
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
|
||||
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
|
||||
@@ -356,7 +413,8 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= max_q1d, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
MFEM_VERIFY(d.Size() == Q1D*Q1D*Q1D*9*3*NE, "");
|
||||
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 9/*PA_SIZE:dims*dims*/, 3/*VDIM*/, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, 3, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
@@ -384,7 +442,8 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
|
||||
const int k = j >= i ?
|
||||
3 - (3-i)*(2-i)/2 + j:
|
||||
3 - (3-j)*(2-j)/2 + i;
|
||||
const real_t O = Q(q,k,e);
|
||||
// using 6 symmetric values
|
||||
const real_t O = Q(q,k,0,e);
|
||||
const real_t Bz = B(qz,dz);
|
||||
const real_t Gz = G(qz,dz);
|
||||
const real_t L = i==2 ? Gz : Bz;
|
||||
@@ -468,12 +527,14 @@ void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported.");
|
||||
PAVectorDiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
// PA Diffusion Apply kernel
|
||||
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
@@ -514,5 +575,6 @@ VectorDiffusionIntegrator::Kernels::Kernels()
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
*/
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -0,0 +1,202 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../kernels.hpp"
|
||||
|
||||
using mfem::kernels::internal::SetMaxOf;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
|
||||
void SmemPAVectorDiffusionApply2D(const int NE,
|
||||
const int coeff_vdim,
|
||||
const Array<real_t> &b,
|
||||
const Array<real_t> &g,
|
||||
const Vector &d,
|
||||
const Vector &x,
|
||||
Vector &y,
|
||||
const int sdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static constexpr int DIM = 2;
|
||||
const int SDIM = T_SDIM ? T_SDIM : sdim;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const int PA_SIZE = DIM*DIM;
|
||||
const bool matrix_coeff = coeff_vdim == DIM*DIM;
|
||||
|
||||
const auto B = b.Read(), G = g.Read();
|
||||
const auto DE = Reshape(d.Read(), Q1D, Q1D, PA_SIZE,
|
||||
SDIM * (matrix_coeff ? SDIM : 1), NE);
|
||||
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
|
||||
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
|
||||
kernels::internal::vd_regs2d_t<3, DIM, MQ1> r0, r1;
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
|
||||
|
||||
for (int i = 0; i < SDIM; i++)
|
||||
{
|
||||
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
|
||||
{
|
||||
kernels::internal::LoadDofs2d(e, D1D, i, XE, r0);
|
||||
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t gradX = r1[i][0][qy][qx];
|
||||
const real_t gradY = r1[i][1][qy][qx];
|
||||
const int k = matrix_coeff ? (j + i * SDIM) : i;
|
||||
const real_t O11 = DE(qx,qy,0,k,e), O12 = DE(qx,qy,1,k,e);
|
||||
const real_t O21 = DE(qx,qy,2,k,e), O22 = DE(qx,qy,3,k,e);
|
||||
r0[i][0][qy][qx] = (O11 * gradX) + (O12 * gradY);
|
||||
r0[i][1][qy][qx] = (O21 * gradX) + (O22 * gradY);
|
||||
} // qx
|
||||
} // qy
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
|
||||
const int ij = matrix_coeff ? j : i;
|
||||
kernels::internal::WriteDofs2d(e, D1D, i, ij, r1, YE);
|
||||
} // j
|
||||
} // i
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
|
||||
void SmemPAVectorDiffusionApply3D(const int NE,
|
||||
const int coeff_vdim,
|
||||
const Array<real_t> &b,
|
||||
const Array<real_t> &g,
|
||||
const Vector &d,
|
||||
const Vector &x,
|
||||
Vector &y,
|
||||
const int sdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
|
||||
static constexpr int DIM = 3;
|
||||
const int SDIM = T_SDIM ? T_SDIM : sdim;
|
||||
MFEM_VERIFY(SDIM == 3, "SDIM must be 3");
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const int PA_SIZE = DIM*DIM;
|
||||
const bool matrix_coeff = coeff_vdim == DIM*DIM;
|
||||
|
||||
const auto B = b.Read(), G = g.Read();
|
||||
const auto DE = Reshape(d.Read(), Q1D, Q1D, Q1D, PA_SIZE,
|
||||
SDIM * (matrix_coeff ? SDIM : 1), NE);
|
||||
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
|
||||
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
|
||||
kernels::internal::vd_regs3d_t<3, DIM, MQ1> r0, r1;
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
|
||||
|
||||
for (int i = 0; i < SDIM; i++)
|
||||
{
|
||||
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
|
||||
{
|
||||
kernels::internal::LoadDofs3d(e, D1D, i, XE, r0);
|
||||
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t gradX = r1[i][0][qz][qy][qx];
|
||||
const real_t gradY = r1[i][1][qz][qy][qx];
|
||||
const real_t gradZ = r1[i][2][qz][qy][qx];
|
||||
const int k = matrix_coeff ? (j + i * SDIM) : i;
|
||||
const real_t O11 = DE(qx,qy,qz,0,k,e), O12 = DE(qx,qy,qz,1,k,e),
|
||||
O13 = DE(qx,qy,qz,2,k,e);
|
||||
const real_t O22 = DE(qx,qy,qz,3,k,e), O23 = DE(qx,qy,qz,4,k,e);
|
||||
const real_t O33 = DE(qx,qy,qz,5,k,e);
|
||||
r0[i][0][qz][qy][qx] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
|
||||
r0[i][1][qz][qy][qx] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
|
||||
r0[i][2][qz][qy][qx] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
|
||||
} // qx
|
||||
} // qy
|
||||
} // qz
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
|
||||
const int ij = matrix_coeff ? j : i;
|
||||
kernels::internal::WriteDofs3d(e, D1D, i, ij, r1, YE);
|
||||
} // j
|
||||
} // i
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, int T_SDIM, int T_D1D, int T_Q1D>
|
||||
VectorDiffusionIntegrator::ApplyKernelType
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if (DIM == 2)
|
||||
{
|
||||
return internal::SmemPAVectorDiffusionApply2D<T_SDIM, T_D1D, T_Q1D>;
|
||||
}
|
||||
else if (DIM == 3)
|
||||
{
|
||||
return internal::SmemPAVectorDiffusionApply3D<T_SDIM, T_D1D, T_Q1D>;
|
||||
}
|
||||
else { MFEM_ABORT("Unsupported kernel"); }
|
||||
}
|
||||
|
||||
inline VectorDiffusionIntegrator::ApplyKernelType
|
||||
VectorDiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int sdim,
|
||||
int d1d, int q1d)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return internal::SmemPAVectorDiffusionApply2D;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return internal::SmemPAVectorDiffusionApply3D;
|
||||
}
|
||||
else { MFEM_ABORT("Unsupported kernel"); }
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
+199
-382
@@ -9,120 +9,218 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../ceed/integrators/mass/mass.hpp"
|
||||
|
||||
#include "./bilininteg_vecmass_pa.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assuming the same element type
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
ElementTransformation *T = mesh->GetTypicalElementTransformation();
|
||||
const IntegrationRule *ir
|
||||
= IntRule ? IntRule : &MassIntegrator::GetRule(el, el, *T);
|
||||
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
|
||||
const auto *ir = IntRule ? IntRule : &MassIntegrator::GetRule(el, el, Trans);
|
||||
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q);
|
||||
}
|
||||
const bool mixed =
|
||||
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
|
||||
if (mixed) { ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q); }
|
||||
else { ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q); }
|
||||
return;
|
||||
}
|
||||
|
||||
// If vdim is not set, set it to the space dimension
|
||||
vdim = (vdim == -1) ? Trans.GetSpaceDim() : vdim;
|
||||
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
|
||||
MFEM_VERIFY(vdim == mesh->Dimension(), "vdim != dim");
|
||||
|
||||
const MemoryType mt = pa_mt == MemoryType::DEFAULT
|
||||
? Device::GetDeviceMemoryType()
|
||||
: pa_mt;
|
||||
|
||||
ne = mesh->GetNE();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetMesh()->GetNE();
|
||||
nq = ir->GetNPoints();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
|
||||
GeometricFactors::JACOBIANS);
|
||||
const int nq = ir->GetNPoints();
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
real_t coeff = 1.0;
|
||||
const int q1d = quad1D;
|
||||
|
||||
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(qs);
|
||||
|
||||
if (Q)
|
||||
{
|
||||
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
|
||||
MFEM_VERIFY(cQ != NULL, "Only ConstantCoefficient is supported.");
|
||||
coeff = cQ->constant;
|
||||
coeff.Project(*Q);
|
||||
}
|
||||
if (!(dim == 2 || dim == 3))
|
||||
else if (VQ)
|
||||
{
|
||||
MFEM_ABORT("Dimension not supported.");
|
||||
coeff.Project(*VQ);
|
||||
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
|
||||
}
|
||||
else if (MQ)
|
||||
{
|
||||
coeff.ProjectTranspose(*MQ);
|
||||
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
|
||||
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
|
||||
}
|
||||
else { coeff.SetConstant(1.0); }
|
||||
|
||||
coeff_vdim = coeff.GetVDim();
|
||||
const bool const_coeff = coeff_vdim == 1;
|
||||
const bool vector_coeff = coeff_vdim == vdim;
|
||||
const bool matrix_coeff = coeff_vdim == vdim * vdim;
|
||||
MFEM_VERIFY(const_coeff + vector_coeff + matrix_coeff == 1, "");
|
||||
|
||||
pa_data.SetSize(coeff_vdim * nq * ne, mt);
|
||||
|
||||
const auto w_r = ir->GetWeights().Read();
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
const real_t constant = coeff;
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
auto w = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
const auto W = Reshape(w_r, q1d, q1d);
|
||||
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
|
||||
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
|
||||
auto D = Reshape(pa_data.Write(), q1d, q1d, coeff_vdim, ne);
|
||||
|
||||
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
const real_t J11 = J(q,0,0,e);
|
||||
const real_t J12 = J(q,1,0,e);
|
||||
const real_t J21 = J(q,0,1,e);
|
||||
const real_t J22 = J(q,1,1,e);
|
||||
const real_t detJ = (J11*J22)-(J21*J12);
|
||||
v(q,e) = w[q] * constant * detJ;
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
const real_t J11 = J(qx, qy, 0, 0, e), J12 = J(qx, qy, 1, 0, e);
|
||||
const real_t J21 = J(qx, qy, 0, 1, e), J22 = J(qx, qy, 1, 1, e);
|
||||
const real_t detJ = (J11 * J22) - (J21 * J12);
|
||||
const real_t w_det = W(qx, qy) * detJ;
|
||||
D(qx, qy, 0, e) = C(0, qx, qy, e) * w_det;
|
||||
if (const_coeff) { continue; }
|
||||
D(qx, qy, 1, e) = C(1, qx, qy, e) * w_det;
|
||||
if (vector_coeff) { continue; }
|
||||
assert(matrix_coeff);
|
||||
D(qx, qy, 2, e) = C(2, qx, qy, e) * w_det;
|
||||
D(qx, qy, 3, e) = C(3, qx, qy, e) * w_det;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
if (dim == 3)
|
||||
else if (dim == 3)
|
||||
{
|
||||
const real_t constant = coeff;
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
auto W = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ,NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
const auto W = Reshape(w_r, q1d, q1d, q1d);
|
||||
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
|
||||
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
|
||||
auto D = Reshape(pa_data.Write(), q1d, q1d, q1d, coeff_vdim, ne);
|
||||
|
||||
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const real_t J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
|
||||
const real_t J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
|
||||
const real_t J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
|
||||
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
v(q,e) = W[q] * constant * detJ;
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
const real_t J11 = J(qx, qy, qz, 0, 0, e),
|
||||
J12 = J(qx, qy, qz, 0, 1, e),
|
||||
J13 = J(qx, qy, qz, 0, 2, e);
|
||||
const real_t J21 = J(qx, qy, qz, 1, 0, e),
|
||||
J22 = J(qx, qy, qz, 1, 1, e),
|
||||
J23 = J(qx, qy, qz, 1, 2, e);
|
||||
const real_t J31 = J(qx, qy, qz, 2, 0, e),
|
||||
J32 = J(qx, qy, qz, 2, 1, e),
|
||||
J33 = J(qx, qy, qz, 2, 2, e);
|
||||
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
const real_t w_det = W(qx, qy, qz) * detJ;
|
||||
D(qx, qy, qz, 0, e) = C(0, qx, qy, qz, e) * w_det;
|
||||
if (const_coeff) { continue; }
|
||||
D(qx, qy, qz, 1, e) = C(1, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 2, e) = C(2, qx, qy, qz, e) * w_det;
|
||||
if (vector_coeff) { continue; }
|
||||
D(qx, qy, qz, 3, e) = C(3, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 4, e) = C(4, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 5, e) = C(5, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 6, e) = C(6, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 7, e) = C(7, qx, qy, qz, e) * w_det;
|
||||
D(qx, qy, qz, 8, e) = C(8, qx, qy, qz, e) * w_det;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Unknown VectorMassIntegrator::AssemblePA kernel for"
|
||||
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
|
||||
}
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
const Array<real_t> &B_,
|
||||
const Array<real_t> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
// Use CEED backend if available
|
||||
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
|
||||
|
||||
// Add the VectorMassAddMultPA specializations
|
||||
static const auto vector_mass_kernel_specializations =
|
||||
( // 2D
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 2,2>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,3>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,4>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,4>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,6>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 5,5>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 6,6>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 7,7>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 8,8>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 9,9>::Add(),
|
||||
// 3D
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,2>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,3>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,4>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,5>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,5>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,6>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,8>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,6>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,8>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 6,7>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 7,8>::Add(),
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 8,9>::Add(),
|
||||
true);
|
||||
MFEM_CONTRACT_VAR(vector_mass_kernel_specializations);
|
||||
|
||||
VectorMassAddMultPA::Run(dim, dofs1D, quad1D,
|
||||
ne, coeff_vdim, maps->B, pa_data, x, y,
|
||||
dofs1D, quad1D);
|
||||
|
||||
}
|
||||
|
||||
template <const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
const Array<real_t> &b,
|
||||
const Vector &pa_data, Vector &diag,
|
||||
const int d1d = 0, const int q1d = 0)
|
||||
{
|
||||
constexpr int VDIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 2;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, NE);
|
||||
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -137,7 +235,7 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
temp[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp[qx][dy] += B(qy, dy) * B(qy, dy) * op(qx, qy, e);
|
||||
temp[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -150,33 +248,31 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
{
|
||||
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
|
||||
}
|
||||
y(dx, dy, 0, e) = temp1;
|
||||
y(dx, dy, 1, e) = temp1;
|
||||
Y(dx, dy, 0, e) = temp1;
|
||||
Y(dx, dy, 1, e) = temp1;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
template <const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
const Array<real_t> &B_,
|
||||
const Array<real_t> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
const Vector &pa_data, Vector &diag,
|
||||
const int d1d = 0, const int q1d = 0)
|
||||
{
|
||||
constexpr int VDIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
const auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
MFEM_VERIFY(pa_data.Size() == Q1D * Q1D * Q1D * NE, "pa_data size error");
|
||||
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
@@ -192,7 +288,8 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
temp[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
temp[qx][qy][dz] += B(qz, dz) * B(qz, dz) * op(qx, qy, qz, e);
|
||||
temp[qx][qy][dz] +=
|
||||
B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -207,7 +304,8 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
temp2[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp2[qx][dy][dz] += B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
|
||||
temp2[qx][dy][dz] +=
|
||||
B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -221,323 +319,42 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
real_t temp3 = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp3 += B(qx, dx) * B(qx, dx)
|
||||
* temp2[qx][dy][dz];
|
||||
temp3 += B(qx, dx) * B(qx, dx) * temp2[qx][dy][dz];
|
||||
}
|
||||
y(dx, dy, dz, 0, e) = temp3;
|
||||
y(dx, dy, dz, 1, e) = temp3;
|
||||
y(dx, dy, dz, 2, e) = temp3;
|
||||
Y(dx, dy, dz, 0, e) = temp3;
|
||||
Y(dx, dy, dz, 1, e) = temp3;
|
||||
Y(dx, dy, dz, 2, e) = temp3;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorMassAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
static void PAVectorMassAssembleDiagonal(const int dim, const int D1D,
|
||||
const int Q1D, const int NE,
|
||||
const Array<real_t> &B,
|
||||
const Array<real_t> &Bt,
|
||||
const Vector &op,
|
||||
Vector &y)
|
||||
const Vector &pa_data,
|
||||
Vector &diag)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal2D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
return PAVectorMassAssembleDiagonal2D(NE, B, pa_data, diag, D1D, Q1D);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal3D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
return PAVectorMassAssembleDiagonal3D(NE, B, pa_data, diag, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Dimension not implemented.");
|
||||
}
|
||||
|
||||
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
if (DeviceCanUseCeed()) { ceedOp->GetDiagonal(diag); }
|
||||
else
|
||||
{
|
||||
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->Bt,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassApply2D(const int NE,
|
||||
const Array<real_t> &B_,
|
||||
const Array<real_t> &Bt_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 2;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t sol_xy[max_Q1D][max_Q1D];
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xy[qy][qx] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t sol_x[max_Q1D];
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
sol_x[qy] = 0.0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx,dy,c,e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_x[qx] += B(qx,dx)* s;
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t d2q = B(qy,dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xy[qy][qx] += d2q * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xy[qy][qx] *= op(qx,qy,e);
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t sol_x[max_D1D];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_x[dx] = 0.0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t s = sol_xy[qy][qx];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_x[dx] += Bt(dx,qx) * s;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t q2d = Bt(dy,qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx,dy,c,e) += q2d * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassApply3D(const int NE,
|
||||
const Array<real_t> &B_,
|
||||
const Array<real_t> &Bt_,
|
||||
const Vector &op_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
real_t sol_xyz[max_Q1D][max_Q1D][max_Q1D];
|
||||
for (int c = 0; c < VDIM; ++ c)
|
||||
{
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xyz[qz][qy][qx] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
real_t sol_xy[max_Q1D][max_Q1D];
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xy[qy][qx] = 0.0;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
real_t sol_x[max_Q1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_x[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const real_t s = x(dx,dy,dz,c,e);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_x[qx] += B(qx,dx) * s;
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t wy = B(qy,dy);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xy[qy][qx] += wy * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const real_t wz = B(qz,dz);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
real_t sol_xy[max_D1D][max_D1D];
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_xy[dy][dx] = 0;
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
real_t sol_x[max_D1D];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_x[dx] = 0;
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t s = sol_xyz[qz][qy][qx];
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_x[dx] += Bt(dx,qx) * s;
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const real_t wy = Bt(dy,qy);
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
sol_xy[dy][dx] += wy * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const real_t wz = Bt(dz,qz);
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
y(dx,dy,dz,c,e) += wz * sol_xy[dy][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorMassApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<real_t> &B,
|
||||
const Array<real_t> &Bt,
|
||||
const Vector &op,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
return PAVectorMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->AddMult(x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
|
||||
MFEM_VERIFY(coeff_vdim == 1, "coeff_vdim != 1");
|
||||
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported");
|
||||
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,212 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../kernels.hpp"
|
||||
|
||||
using mfem::kernels::internal::SetMaxOf;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// \cond DO_NOT_DOCUMENT
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
void SmemPAVectorMassApply2D(const int NE,
|
||||
const int coeff_vdim,
|
||||
const Array<real_t> &b,
|
||||
const Vector &d,
|
||||
const Vector &x,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static constexpr int DIM = 2, VDIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const bool const_coeff = coeff_vdim == 1;
|
||||
const bool vector_coeff = coeff_vdim == DIM;
|
||||
const bool matrix_coeff = coeff_vdim == DIM*DIM;
|
||||
|
||||
const auto B = b.Read();
|
||||
const auto D = Reshape(d.Read(), Q1D, Q1D, coeff_vdim, NE);
|
||||
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
|
||||
kernels::internal::v_regs2d_t<VDIM, MQ1> r0, r1;
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
|
||||
kernels::internal::LoadDofs2d(e, D1D, X, r0);
|
||||
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1);
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t Qx = r1[0][qy][qx];
|
||||
const real_t Qy = r1[1][qy][qx];
|
||||
const real_t D0 = D(qx, qy, 0, e);
|
||||
|
||||
if (const_coeff)
|
||||
{
|
||||
r0[0][qy][qx] = D0 * Qx;
|
||||
r0[1][qy][qx] = D0 * Qy;
|
||||
}
|
||||
if (vector_coeff)
|
||||
{
|
||||
const real_t D1 = D(qx, qy, 1, e);
|
||||
r0[0][qy][qx] = D0 * Qx;
|
||||
r0[1][qy][qx] = D1 * Qy;
|
||||
}
|
||||
if (matrix_coeff)
|
||||
{
|
||||
const real_t D1 = D(qx, qy, 1, e);
|
||||
const real_t D2 = D(qx, qy, 2, e);
|
||||
const real_t D3 = D(qx, qy, 3, e);
|
||||
r0[0][qy][qx] = D0 * Qx + D1 * Qy;
|
||||
r0[1][qy][qx] = D2 * Qx + D3 * Qy;
|
||||
}
|
||||
}
|
||||
}
|
||||
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
|
||||
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
void SmemPAVectorMassApply3D(const int NE,
|
||||
const int coeff_vdim,
|
||||
const Array<real_t> &b,
|
||||
const Vector &d,
|
||||
const Vector &x,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static constexpr int VDIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const bool const_coeff = coeff_vdim == 1;
|
||||
const bool vector_coeff = coeff_vdim == VDIM;
|
||||
const bool matrix_coeff = coeff_vdim == VDIM*VDIM;
|
||||
|
||||
const auto B = b.Read();
|
||||
const auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, coeff_vdim, NE);
|
||||
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
|
||||
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
|
||||
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1;
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
|
||||
kernels::internal::LoadDofs3d(e, D1D, X, r0);
|
||||
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1);
|
||||
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t Qx = r1[0][qz][qy][qx];
|
||||
const real_t Qy = r1[1][qz][qy][qx];
|
||||
const real_t Qz = r1[2][qz][qy][qx];
|
||||
const real_t D0 = D(qx, qy, qz, 0, e);
|
||||
if (const_coeff)
|
||||
{
|
||||
r0[0][qz][qy][qx] = D0 * Qx;
|
||||
r0[1][qz][qy][qx] = D0 * Qy;
|
||||
r0[2][qz][qy][qx] = D0 * Qz;
|
||||
}
|
||||
if (vector_coeff)
|
||||
{
|
||||
const real_t D1 = D(qx, qy, qz, 1, e);
|
||||
const real_t D2 = D(qx, qy, qz, 2, e);
|
||||
r0[0][qz][qy][qx] = D0 * Qx;
|
||||
r0[1][qz][qy][qx] = D1 * Qy;
|
||||
r0[2][qz][qy][qx] = D2 * Qz;
|
||||
}
|
||||
if (matrix_coeff)
|
||||
{
|
||||
const real_t D1 = D(qx, qy, qz, 1, e);
|
||||
const real_t D2 = D(qx, qy, qz, 2, e);
|
||||
const real_t D3 = D(qx, qy, qz, 3, e);
|
||||
const real_t D4 = D(qx, qy, qz, 4, e);
|
||||
const real_t D5 = D(qx, qy, qz, 5, e);
|
||||
const real_t D6 = D(qx, qy, qz, 6, e);
|
||||
const real_t D7 = D(qx, qy, qz, 7, e);
|
||||
const real_t D8 = D(qx, qy, qz, 8, e);
|
||||
r0[0][qz][qy][qx] = D0 * Qx + D1 * Qy + D2 * Qz;
|
||||
r0[1][qz][qy][qx] = D3 * Qx + D4 * Qy + D5 * Qz;
|
||||
r0[2][qz][qy][qx] = D6 * Qx + D7 * Qy + D8 * Qz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
|
||||
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
VectorMassIntegrator::VectorMassAddMultPAType
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Kernel()
|
||||
{
|
||||
if (DIM == 2)
|
||||
{
|
||||
return internal::SmemPAVectorMassApply2D<T_D1D,T_Q1D>;
|
||||
}
|
||||
else if (DIM == 3)
|
||||
{
|
||||
return internal::SmemPAVectorMassApply3D<T_D1D, T_Q1D>;
|
||||
}
|
||||
else { MFEM_ABORT("Unsupported kernel"); }
|
||||
}
|
||||
|
||||
inline VectorMassIntegrator::VectorMassAddMultPAType
|
||||
VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int d1d, int q1d)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return internal::SmemPAVectorMassApply2D;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return internal::SmemPAVectorMassApply3D;
|
||||
}
|
||||
else { MFEM_ABORT("Unsupported kernel"); }
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
+710
-3
@@ -14,6 +14,7 @@
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/tensor.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -26,7 +27,713 @@ namespace kernels
|
||||
namespace internal
|
||||
{
|
||||
|
||||
/// Load B1d matrice into shared memory
|
||||
// Types for tensors mapped to registers
|
||||
// - N is the number of threads in each of the x and y dimensions
|
||||
// - N should not be greater than 32, to have a maximum of 1024 threads
|
||||
// On GPU, the last two dimensions are set to 0 to match a 2D tile of threads
|
||||
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
|
||||
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
|
||||
template <int N = 0>
|
||||
using s_regs2d_t = mfem::future::tensor<real_t, 0, 0>;
|
||||
|
||||
template <int VDIM, int N>
|
||||
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, 0, 0>;
|
||||
|
||||
template <int VDIM, int DIM, int N = 0>
|
||||
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, 0, 0>;
|
||||
|
||||
template <int N>
|
||||
using s_regs3d_t = mfem::future::tensor<real_t, N, 0, 0>;
|
||||
|
||||
template <int VDIM, int N>
|
||||
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, 0, 0>;
|
||||
|
||||
template <int VDIM, int DIM, int N>
|
||||
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, 0, 0>;
|
||||
|
||||
// on GPU, SetMaxOf is a no-op, for minimal register usage
|
||||
constexpr int SetMaxOf(int n) { return n; }
|
||||
#else
|
||||
template <int N>
|
||||
using s_regs2d_t = mfem::future::tensor<real_t, N, N>;
|
||||
|
||||
template <int VDIM, int N>
|
||||
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, N, N>;
|
||||
|
||||
template <int VDIM, int DIM, int N>
|
||||
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N>;
|
||||
|
||||
template <int N>
|
||||
using s_regs3d_t = mfem::future::tensor<real_t, N, N, N>;
|
||||
|
||||
template <int VDIM, int N>
|
||||
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, N, N>;
|
||||
|
||||
template <int VDIM, int DIM, int N>
|
||||
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N, N>;
|
||||
|
||||
// on CPU, get next multiple of 4, allowing better alignments
|
||||
template <int N>
|
||||
constexpr int NextMultipleOf(int n)
|
||||
{
|
||||
static_assert(N > 0 && (N & (N - 1)) == 0, "N must be a power of 2");
|
||||
return (n + (N - 1)) & ~(N - 1);
|
||||
}
|
||||
constexpr int SetMaxOf(int n) { return NextMultipleOf<4>(n); }
|
||||
#endif // CUDA/HIP && DEVICE_COMPILE
|
||||
|
||||
/// Load 2D matrix into shared memory
|
||||
template <int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
|
||||
const real_t *M, real_t (*N)[MQ1])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
|
||||
{
|
||||
N[dy][qx] = M[dy * q1d + qx];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 2D input VDIM*DIM vector into given register tensor, specific component
|
||||
template <int VDIM, int DIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d, const int c,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][d][dy][dx] = X(dx, dy, c, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 2D input VDIM*DIM vector into given register tensor
|
||||
template <int VDIM, int DIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c) { LoadDofs2d(e, d1d, c, X, Y); }
|
||||
}
|
||||
|
||||
/// Load 2D input VDIM vector into given register tensor
|
||||
template <int VDIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
v_regs2d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][dy][dx] = X(dx, dy, c, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 2D input scalar into given register tensor
|
||||
template <int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
|
||||
const DeviceTensor<3, const real_t> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[dy][dx] = X(dx, dy, e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 2D vector into given device tensor, with read (i) write (j) indices
|
||||
template <int VDIM, int DIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
|
||||
const int i, const int j,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
const DeviceTensor<4, real_t> &Y)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
real_t y = 0.0;
|
||||
for (int d = 0; d < DIM; d++) { y += X(i, d, dy, dx); }
|
||||
Y(dx, dy, j, e) += y;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 2D VDIM*DIM vector into given device tensor
|
||||
template <int VDIM, int DIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
const DeviceTensor<4, real_t> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c) { WriteDofs2d(e, d1d, c, c, X, Y); }
|
||||
}
|
||||
|
||||
/// Write 2D VDIM vector into given device tensor
|
||||
template <int VDIM, int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
|
||||
v_regs2d_t<VDIM, MQ1> &X,
|
||||
const DeviceTensor<4, real_t> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y(dx, dy, c, e) += X(c, dy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input VDIM*DIM vector into given register tensor, specific component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d, const int c,
|
||||
const DeviceTensor<5, const real_t> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][d][dz][dy][dx] = X(dx, dy, dz, c, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input VDIM*DIM vector into given register tensor
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
|
||||
const DeviceTensor<5, const real_t> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c) { LoadDofs3d(e, d1d, c, X, Y); }
|
||||
}
|
||||
|
||||
/// Load 3D input VDIM vector into given register tensor
|
||||
template <int VDIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
|
||||
const DeviceTensor<5, const real_t> &X,
|
||||
v_regs3d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][dz][dy][dx] = X(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input scalar into given register tensor
|
||||
template <int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[dz][dy][dx] = X(dx,dy,dz,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 3D scalar into given device tensor, with read (i) write (j) indices
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
|
||||
const int i, const int j,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
const DeviceTensor<5, real_t> &Y)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
real_t value = 0.0;
|
||||
for (int d = 0; d < DIM; d++) { value += X(i, d, dz, dy, dx); }
|
||||
Y(dx, dy, dz, j, e) += value;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 3D VDIM*DIM vector into given device tensor
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
const DeviceTensor<5, real_t> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c) { WriteDofs3d(e, d1d, c, c, X, Y); }
|
||||
}
|
||||
|
||||
/// Write 3D VDIM vector into given device tensor
|
||||
template <int VDIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
|
||||
v_regs3d_t<VDIM, MQ1> &X,
|
||||
const DeviceTensor<5, real_t> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y(dx, dy, dz, c, e) += X(c, dz, dy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar contraction, X direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractX2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const s_regs2d_t<MQ1> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
|
||||
{
|
||||
smem[y][x] = X[y][x];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
|
||||
{
|
||||
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
|
||||
}
|
||||
Y[y][x] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar contraction, Y direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractY2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const s_regs2d_t<MQ1> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[y][x]; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
|
||||
{
|
||||
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
|
||||
}
|
||||
Y[y][x] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar copy
|
||||
template <int MQ1 = 0>
|
||||
inline MFEM_HOST_DEVICE void Copy2d(const int q1d,
|
||||
s_regs2d_t<MQ1> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { Y[y][x] = X[y][x]; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar contraction: X & Y directions, with additional copy
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void Contract2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*Bx)[MQ1],
|
||||
const real_t (*By)[MQ1],
|
||||
s_regs2d_t<MQ1> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
if (!Transpose)
|
||||
{
|
||||
ContractX2d<false>(d1d, q1d, smem, Bx, X, Y);
|
||||
ContractY2d<false>(d1d, q1d, smem, By, Y, X);
|
||||
Copy2d(q1d, X, Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
Copy2d(q1d, X, Y);
|
||||
ContractY2d<true>(d1d, q1d, smem, By, Y, X);
|
||||
ContractX2d<true>(d1d, q1d, smem, Bx, X, Y);
|
||||
}
|
||||
}
|
||||
|
||||
/// 2D scalar evaluation
|
||||
template <int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
s_regs2d_t<MQ1> &X,
|
||||
s_regs2d_t<MQ1> &Y)
|
||||
{
|
||||
Contract2d<Transpose, MQ1>(d1d, q1d, smem, B, B, X, Y);
|
||||
}
|
||||
|
||||
/// 2D vector evaluation
|
||||
template <int VDIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
v_regs2d_t<VDIM, MQ1> &X,
|
||||
v_regs2d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Eval2d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
|
||||
}
|
||||
}
|
||||
|
||||
/// 2D vector transposed evaluation
|
||||
template <int VDIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void EvalTranspose2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
v_regs2d_t<VDIM, MQ1> &X,
|
||||
v_regs2d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
Eval2d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
|
||||
}
|
||||
|
||||
/// 2D vector gradient, with component
|
||||
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
|
||||
const int c)
|
||||
{
|
||||
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
|
||||
const real_t (*By)[MQ1] = (d == 1) ? G : B;
|
||||
Contract2d<Transpose>(d1d, q1d, smem, Bx, By, X[c][d], Y[c][d]);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
/// 2D vector gradient
|
||||
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
|
||||
}
|
||||
}
|
||||
|
||||
/// 2D vector transposed gradient
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
constexpr bool Transpose = true;
|
||||
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y);
|
||||
}
|
||||
|
||||
/// 2D scalar contraction, with component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
|
||||
const int c)
|
||||
{
|
||||
constexpr bool Transpose = true;
|
||||
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
|
||||
}
|
||||
|
||||
/// 3D scalar contraction, X direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractX3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const s_regs3d_t<MQ1> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
for (int z = 0; z < d1d; ++z)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
|
||||
{
|
||||
smem[y][x] = X[z][y][x];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
|
||||
{
|
||||
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
|
||||
}
|
||||
Y[z][y][x] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D scalar contraction, Y direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractY3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const s_regs3d_t<MQ1> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
for (int z = 0; z < d1d; ++z)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[z][y][x]; }
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
|
||||
{
|
||||
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
|
||||
}
|
||||
Y[z][y][x] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D scalar contraction, Z direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractZ3d(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const s_regs3d_t<MQ1> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
for (int z = 0; z < (Transpose ? d1d : q1d); ++z)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
|
||||
{
|
||||
u += (Transpose ? B[z][k] : B[k][z]) * X[k][y][x];
|
||||
}
|
||||
Y[z][y][x] = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D scalar contraction: X, Y & Z directions
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void Contract3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*Bx)[MQ1],
|
||||
const real_t (*By)[MQ1],
|
||||
const real_t (*Bz)[MQ1],
|
||||
s_regs3d_t<MQ1> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
if (!Transpose)
|
||||
{
|
||||
ContractX3d<false>(d1d, q1d, smem, Bx, X, Y);
|
||||
ContractY3d<false>(d1d, q1d, smem, By, Y, X);
|
||||
ContractZ3d<false>(d1d, q1d, Bz, X, Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
ContractZ3d<true>(d1d, q1d, Bz, X, Y);
|
||||
ContractY3d<true>(d1d, q1d, smem, By, Y, X);
|
||||
ContractX3d<true>(d1d, q1d, smem, Bx, X, Y);
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D scalar evaluation
|
||||
template <int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
s_regs3d_t<MQ1> &X,
|
||||
s_regs3d_t<MQ1> &Y)
|
||||
{
|
||||
Contract3d<Transpose>(d1d, q1d, smem, B, B, B, X, Y);
|
||||
}
|
||||
|
||||
/// 3D vector evaluation
|
||||
template <int VDIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
v_regs3d_t<VDIM, MQ1> &X,
|
||||
v_regs3d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Eval3d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D vector transposed evaluation
|
||||
template <int VDIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void EvalTranspose3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
v_regs3d_t<VDIM, MQ1> &X,
|
||||
v_regs3d_t<VDIM, MQ1> &Y)
|
||||
{
|
||||
Eval3d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
|
||||
}
|
||||
|
||||
/// 3D vector gradient, with component
|
||||
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
|
||||
const int c)
|
||||
{
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
|
||||
const real_t (*By)[MQ1] = (d == 1) ? G : B;
|
||||
const real_t (*Bz)[MQ1] = (d == 2) ? G : B;
|
||||
Contract3d<Transpose>(d1d, q1d, smem, Bx, By, Bz, X[c][d], Y[c][d]);
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D vector gradient
|
||||
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
|
||||
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
Grad3d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
|
||||
}
|
||||
}
|
||||
|
||||
/// 3D vector transposed gradient
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y);
|
||||
}
|
||||
|
||||
/// 3D vector transposed gradient, with component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
|
||||
real_t (&smem)[MQ1][MQ1],
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
|
||||
const int c)
|
||||
{
|
||||
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y, c);
|
||||
}
|
||||
|
||||
/// Load B1d matrix into shared memory
|
||||
template<int MD1, int MQ1>
|
||||
MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
|
||||
const ConstDeviceMatrix &b,
|
||||
@@ -48,7 +755,7 @@ MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load Bt1d matrices into shared memory
|
||||
/// Load Bt1d matrix into shared memory
|
||||
template<int MD1, int MQ1>
|
||||
MFEM_HOST_DEVICE inline void LoadBt(const int D1D, const int Q1D,
|
||||
const ConstDeviceMatrix &b,
|
||||
@@ -1551,7 +2258,7 @@ MFEM_HOST_DEVICE inline void GradXt(const int D1D, const int Q1D,
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace kernels::internal
|
||||
} // namespace internal
|
||||
|
||||
} // namespace kernels
|
||||
|
||||
|
||||
+4
-4
@@ -5220,10 +5220,10 @@ void ConformingProlongationOperator::Mult(const Vector &x, Vector &y) const
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
const int end = external_ldofs[i];
|
||||
std::copy(xdata+j-i, xdata+end-i, ydata+j);
|
||||
if (end > j) { std::copy(xdata+j-i, xdata+end-i, ydata+j); }
|
||||
j = end+1;
|
||||
}
|
||||
std::copy(xdata+j-m, xdata+Width(), ydata+j);
|
||||
if (Width() > (j-m)) { std::copy(xdata+j-m, xdata+Width(), ydata+j); }
|
||||
|
||||
const int out_layout = 0; // 0 - output is ldofs array
|
||||
if (!local)
|
||||
@@ -5251,10 +5251,10 @@ void ConformingProlongationOperator::MultTranspose(
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
const int end = external_ldofs[i];
|
||||
std::copy(xdata+j, xdata+end, ydata+j-i);
|
||||
if (end > j) { std::copy(xdata+j, xdata+end, ydata+j-i); }
|
||||
j = end+1;
|
||||
}
|
||||
std::copy(xdata+j, xdata+Height(), ydata+j-m);
|
||||
if (Height() > j) { std::copy(xdata+j, xdata+Height(), ydata+j-m); }
|
||||
|
||||
const int out_layout = 2; // 2 - output is an array on all ltdofs
|
||||
if (!local)
|
||||
|
||||
@@ -41,6 +41,12 @@ public:
|
||||
qspace(&qspace_), own_qspace(false), vdim(vdim_)
|
||||
{ UseDevice(true); }
|
||||
|
||||
/// Same as above but specify the device memory type
|
||||
QuadratureFunction(QuadratureSpaceBase &qspace_, MemoryType mt, int vdim_ = 1)
|
||||
: Vector(vdim_*qspace_.GetSize(), mt),
|
||||
qspace(&qspace_), own_qspace(false), vdim(vdim_)
|
||||
{ UseDevice(true); }
|
||||
|
||||
/// Create a QuadratureFunction based on the given QuadratureSpaceBase.
|
||||
/** The QuadratureFunction does not assume ownership of the
|
||||
QuadratureSpaceBase.
|
||||
@@ -48,6 +54,10 @@ public:
|
||||
QuadratureFunction(QuadratureSpaceBase *qspace_, int vdim_ = 1)
|
||||
: QuadratureFunction(*qspace_, vdim_) { }
|
||||
|
||||
/// Same as above but specify the device memory type
|
||||
QuadratureFunction(QuadratureSpaceBase *qspace_, MemoryType mt, int vdim_ = 1)
|
||||
: QuadratureFunction(*qspace_, mt, vdim_) { }
|
||||
|
||||
/** @brief Create a QuadratureFunction based on the given QuadratureSpaceBase,
|
||||
using the external (host) data, @a qf_data. */
|
||||
/** The QuadratureFunction does not assume ownership of the
|
||||
|
||||
@@ -5368,6 +5368,15 @@ void TMOP_Integrator::ParEnableNormalization(const ParGridFunction &x)
|
||||
}
|
||||
#endif
|
||||
|
||||
void TMOP_Integrator::GetNormalizationFactors(real_t &m_normal,
|
||||
real_t &l_normal,
|
||||
real_t &s_normal)
|
||||
{
|
||||
m_normal = this->metric_normal;
|
||||
l_normal = this->lim_normal;
|
||||
s_normal = this->surf_fit_normal;
|
||||
}
|
||||
|
||||
void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
|
||||
real_t &metric_energy,
|
||||
real_t &lim_energy)
|
||||
|
||||
@@ -452,6 +452,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 9; }
|
||||
};
|
||||
|
||||
/// 2D non-barrier Shape+Size+Orientation (VOS) metric (polyconvex).
|
||||
@@ -502,6 +504,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 22; }
|
||||
};
|
||||
|
||||
/// 2D barrier shape metric (polyconvex).
|
||||
@@ -522,6 +526,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 50; }
|
||||
};
|
||||
|
||||
/// 2D non-barrier size (V) metric (not polyconvex).
|
||||
@@ -593,6 +599,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 58; }
|
||||
};
|
||||
|
||||
/// 2D non-barrier Shape+Size (VS) metric.
|
||||
@@ -675,6 +683,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 85; }
|
||||
};
|
||||
|
||||
/// 2D compound barrier Shape+Size (VS) metric (balanced).
|
||||
@@ -732,6 +742,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 98; }
|
||||
};
|
||||
|
||||
/// 2D untangling metric.
|
||||
@@ -751,6 +763,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 211; }
|
||||
};
|
||||
|
||||
/// Shifted barrier form of metric 56 (area, ideal barrier metric), 2D
|
||||
@@ -771,6 +785,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 252; }
|
||||
};
|
||||
|
||||
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
|
||||
@@ -790,6 +806,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 301; }
|
||||
};
|
||||
|
||||
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
|
||||
@@ -872,6 +890,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 311; }
|
||||
};
|
||||
|
||||
/// 3D Shape (S) metric, untangling version of 303.
|
||||
@@ -930,6 +950,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 316; }
|
||||
};
|
||||
|
||||
/// 3D Size (V) metric.
|
||||
@@ -1074,6 +1096,7 @@ public:
|
||||
AddQualityMetric(sz_metric, gamma);
|
||||
}
|
||||
|
||||
int Id() const override { return 333; }
|
||||
virtual ~TMOP_Metric_333() { delete sh_metric; delete sz_metric; }
|
||||
};
|
||||
|
||||
@@ -1136,6 +1159,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 342; }
|
||||
};
|
||||
|
||||
/// 3D barrier Shape+Size (VS) metric, well-posed (polyconvex).
|
||||
@@ -1177,6 +1202,8 @@ public:
|
||||
|
||||
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const real_t weight, DenseMatrix &A) const override;
|
||||
|
||||
int Id() const override { return 352; }
|
||||
};
|
||||
|
||||
/// 3D non-barrier Shape (S) metric.
|
||||
@@ -1963,6 +1990,12 @@ class TMOP_Integrator : public NonlinearFormIntegrator
|
||||
protected:
|
||||
friend class TMOPNewtonSolver;
|
||||
friend class TMOPComboIntegrator;
|
||||
friend class TMOPEnergyPA2D;
|
||||
friend class TMOPEnergyPA3D;
|
||||
friend class TMOPAssembleGradPA2D;
|
||||
friend class TMOPAssembleGradPA3D;
|
||||
friend class TMOPAddMultPA2D;
|
||||
friend class TMOPAddMultPA3D;
|
||||
|
||||
// Initial positions of the mesh nodes. Not owned. The pointer is set at the
|
||||
// start of the solve by TMOPNewtonSolver::Mult(), and unset at the end.
|
||||
@@ -2483,6 +2516,11 @@ public:
|
||||
void ParEnableNormalization(const ParGridFunction &x);
|
||||
#endif
|
||||
|
||||
/** @brief Get the normalization factors of the metric */
|
||||
void GetNormalizationFactors(real_t &metric_normal,
|
||||
real_t &lim_normal,
|
||||
real_t &surf_fit_normal);
|
||||
|
||||
/** @brief Enables FD-based approximation and computes dx. */
|
||||
void EnableFiniteDifferences(const GridFunction &x);
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
@@ -0,0 +1,180 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/* // Original i-j assembly (old invariants code).
|
||||
for (int e = 0; e < NE; e++)
|
||||
{
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
el.CalcDShape(ip, DSh);
|
||||
Mult(DSh, Jrt, DS);
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
for (int r = 0; r < dim; r++)
|
||||
{
|
||||
for (int c = 0; c < dim; c++)
|
||||
{
|
||||
for (int rr = 0; rr < dim; rr++)
|
||||
{
|
||||
for (int cc = 0; cc < dim; cc++)
|
||||
{
|
||||
const real_t H = h(r, c, rr, cc);
|
||||
A(e, i + r*dof, j + rr*dof) +=
|
||||
weight_q * DS(i, c) * DS(j, cc) * H;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}*/
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleDiagPA_2D(const int NE,
|
||||
const ConstDeviceMatrix &B,
|
||||
const ConstDeviceMatrix &G,
|
||||
const DeviceTensor<5, const real_t> &J,
|
||||
const DeviceTensor<7, const real_t> &H,
|
||||
DeviceTensor<4> &D,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
// Takes into account Jtr by replacing H with Href at all quad points.
|
||||
MFEM_SHARED real_t Href_data[2 * 2 * 2 * MQ1 * MQ1];
|
||||
DeviceTensor<5, real_t> Href(Href_data, 2, 2, 2, MQ1, MQ1);
|
||||
for (int v = 0; v < 2; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, e);
|
||||
real_t Jrt_data[4];
|
||||
ConstDeviceMatrix Jrt(Jrt_data, 2, 2);
|
||||
kernels::CalcInverse<2>(Jtr, Jrt_data);
|
||||
|
||||
for (int m = 0; m < 2; m++)
|
||||
{
|
||||
for (int n = 0; n < 2; n++)
|
||||
{
|
||||
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
|
||||
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
|
||||
Href(v, m, n, qx, qy) = 0.0;
|
||||
for (int s = 0; s < 2; s++)
|
||||
{
|
||||
for (int t = 0; t < 2; t++)
|
||||
{
|
||||
Href(v, m, n, qx, qy) +=
|
||||
Jrt(m, s) * H(v, s, v, t, qx, qy, e) * Jrt(n, t);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_SHARED real_t qd[2 * 2 * MQ1 * MD1];
|
||||
DeviceTensor<4, real_t> QD(qd, 2, 2, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < 2; v++)
|
||||
{
|
||||
// Contract in y.
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
for (int m = 0; m < 2; m++)
|
||||
{
|
||||
for (int n = 0; n < 2; n++) { QD(m, n, qx, dy) = 0.0; }
|
||||
}
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t By = B(qy, dy);
|
||||
const real_t Gy = G(qy, dy);
|
||||
for (int m = 0; m < 2; m++)
|
||||
{
|
||||
for (int n = 0; n < 2; n++)
|
||||
{
|
||||
const real_t L = (m == 1 ? Gy : By);
|
||||
const real_t R = (n == 1 ? Gy : By);
|
||||
QD(m, n, qx, dy) += L * Href(v, m, n, qx, qy) * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Contract in x.
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
|
||||
{
|
||||
real_t d = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t Bx = B(qx, dx);
|
||||
const real_t Gx = G(qx, dx);
|
||||
|
||||
for (int m = 0; m < 2; m++)
|
||||
{
|
||||
for (int n = 0; n < 2; n++)
|
||||
{
|
||||
const real_t L = (m == 0 ? Gx : Bx);
|
||||
const real_t R = (n == 0 ? Gx : Bx);
|
||||
d += L * QD(m, n, qx, dy) * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
D(dx, dy, v, e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag2D, TMOP_AssembleDiagPA_2D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag2D);
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &diagonal) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto B = Reshape(PA.maps->B.Read(), q, d);
|
||||
const auto G = Reshape(PA.maps->G.Read(), q, d);
|
||||
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
|
||||
const auto H = Reshape(PA.H.Read(), 2, 2, 2, 2, q, q, NE);
|
||||
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
|
||||
|
||||
TMOPAssembleDiag2D::Run(d, q, NE, B, G, J, H, D, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,83 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleDiagPA_C0_2D(const int NE,
|
||||
const ConstDeviceMatrix &B,
|
||||
const DeviceTensor<5, const real_t> &H0,
|
||||
DeviceTensor<4> &D,
|
||||
const int d1d, const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t qd[MQ1 * MD1];
|
||||
DeviceTensor<2, real_t> QD(qd, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < 2; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
QD(qx, dy) = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t bb = B(qy, dy) * B(qy, dy);
|
||||
QD(qx, dy) += bb * H0(v, v, qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
|
||||
{
|
||||
real_t d = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t bb = B(qx, dx) * B(qx, dx);
|
||||
d += bb * QD(qx, dy);
|
||||
}
|
||||
D(dx, dy, v, e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef2D, TMOP_AssembleDiagPA_C0_2D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef2D);
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &diagonal) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto B = Reshape(PA.maps->B.Read(), q, d);
|
||||
const auto H0 = Reshape(PA.H0.Read(), 2, 2, q, q, NE);
|
||||
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
|
||||
|
||||
TMOPAssembleDiagCoef2D::Run(d, q, NE, B, H0, D, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,229 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleDiagPA_3D(const int NE,
|
||||
const ConstDeviceMatrix &B,
|
||||
const ConstDeviceMatrix &G,
|
||||
const DeviceTensor<6, const real_t> &J,
|
||||
const DeviceTensor<8, const real_t> &H,
|
||||
DeviceTensor<5> &D,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[3][3][MQ1][MQ1];
|
||||
kernels::internal::vd_regs3d_t<3, 3, MQ1> rH, r0, r1;
|
||||
|
||||
for (int v = 0; v < 3; ++v)
|
||||
{
|
||||
// Takes into account Jtr by replacing H with Href at all quad points.
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
|
||||
real_t Jrt_data[9];
|
||||
ConstDeviceMatrix Jrt(Jrt_data, 3, 3);
|
||||
kernels::CalcInverse<3>(Jtr, Jrt_data);
|
||||
|
||||
real_t h[3][3];
|
||||
for (int s = 0; s < 3; s++)
|
||||
{
|
||||
for (int t = 0; t < 3; t++)
|
||||
{
|
||||
h[s][t] = H(v, s, v, t, qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
|
||||
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
|
||||
rH(m, n, qz, qy, qx) = 0.0;
|
||||
for (int s = 0; s < 3; s++)
|
||||
{
|
||||
for (int t = 0; t < 3; t++)
|
||||
{
|
||||
rH(m, n, qz, qy, qx) += Jrt(m, s) * h[s][t] * Jrt(n, t);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// Contract in z.
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
r0(m, n, dz, qy, qx) = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const real_t Bz = B(qz, dz), Gz = G(qz, dz);
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
const real_t L = (m == 2 ? Gz : Bz);
|
||||
const real_t R = (n == 2 ? Gz : Bz);
|
||||
r0(m, n, dz, qy, qx) += L * rH(m, n, qz, qy, qx) * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// Contract in y.
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
smem[m][n][qy][qx] = r0(m, n, dz, qy, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
r1(m, n, dz, dy, qx) = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t By = B(qy, dy);
|
||||
const real_t Gy = G(qy, dy);
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
const real_t L = (m == 1 ? Gy : By);
|
||||
const real_t R = (n == 1 ? Gy : By);
|
||||
r1(m, n, dz, dy, qx) += L * smem[m][n][qy][qx] * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// Contract in x.
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
smem[m][n][dy][qx] = r1(m, n, dz, dy, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
|
||||
{
|
||||
real_t d = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t Bx = B(qx, dx);
|
||||
const real_t Gx = G(qx, dx);
|
||||
for (int m = 0; m < 3; m++)
|
||||
{
|
||||
for (int n = 0; n < 3; n++)
|
||||
{
|
||||
const real_t L = (m == 0 ? Gx : Bx);
|
||||
const real_t R = (n == 0 ? Gx : Bx);
|
||||
d += L * smem[m][n][dy][qx] * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
D(dx, dy, dz, v, e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag3D, TMOP_AssembleDiagPA_3D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag3D);
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &diagonal) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto B = Reshape(PA.maps->B.Read(), q, d);
|
||||
const auto G = Reshape(PA.maps->G.Read(), q, d);
|
||||
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
|
||||
const auto H = Reshape(PA.H.Read(), 3, 3, 3, 3, q, q, q, NE);
|
||||
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
|
||||
|
||||
TMOPAssembleDiag3D::Run(d, q, NE, B, G, J, H, D, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,131 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleDiagPA_C0_3D(const int NE,
|
||||
const ConstDeviceMatrix &B,
|
||||
const DeviceTensor<6, const real_t> &H0,
|
||||
DeviceTensor<5> &D,
|
||||
const int d1d, const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
kernels::internal::s_regs3d_t<MQ1> r0, r1;
|
||||
|
||||
for (int v = 0; v < 3; ++v)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const real_t Bz = B(qz, dz);
|
||||
u += Bz * H0(v, v, qx, qy, qz, e) * Bz;
|
||||
}
|
||||
r0[dz][qy][qx] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// second tensor contraction, along y direction
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
smem[qy][qx] = r0[dz][qy][qx];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const real_t By = B(qy, dy);
|
||||
u += By * smem[qy][qx] * By;
|
||||
}
|
||||
r1[dz][dy][qx] = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// third tensor contraction, along x direction
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
smem[dy][qx] = r1[dz][dy][qx];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const real_t Bx = B(qx, dx);
|
||||
u += Bx * smem[dy][qx] * Bx;
|
||||
}
|
||||
D(dx, dy, dz, v, e) += u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef3D, TMOP_AssembleDiagPA_C0_3D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef3D);
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &diagonal) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto B = Reshape(PA.maps->B.Read(), q, d);
|
||||
const auto H0 = Reshape(PA.H0.Read(), 3, 3, q, q, q, NE);
|
||||
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
|
||||
|
||||
TMOPAssembleDiagCoef3D::Run(d, q, NE, B, H0, D, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,35 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_2D(const Vector &x) const
|
||||
{
|
||||
const int mid = metric->Id();
|
||||
|
||||
// Calls TMOPAssembleGradPA2D::Mult for the given mid.
|
||||
TMOPAssembleGradPA2D ker(this, x);
|
||||
if (mid == 1) { return tmop::Kernel<1>(ker); }
|
||||
if (mid == 2) { return tmop::Kernel<2>(ker); }
|
||||
if (mid == 7) { return tmop::Kernel<7>(ker); }
|
||||
if (mid == 56) { return tmop::Kernel<56>(ker); }
|
||||
if (mid == 77) { return tmop::Kernel<77>(ker); }
|
||||
if (mid == 80) { return tmop::Kernel<80>(ker); }
|
||||
if (mid == 94) { return tmop::Kernel<94>(ker); }
|
||||
|
||||
MFEM_ABORT("Unsupported TMOP metric " << mid);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,106 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class TMOPAssembleGradPA2D
|
||||
{
|
||||
const mfem::TMOP_Integrator *ti; // not owned
|
||||
const Vector &x;
|
||||
|
||||
public:
|
||||
TMOPAssembleGradPA2D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
|
||||
x(x) {}
|
||||
|
||||
int Ndof() const { return ti->PA.maps->ndof; }
|
||||
int Nqpt() const { return ti->PA.maps->nqpt; }
|
||||
|
||||
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
|
||||
static void Mult(TMOPAssembleGradPA2D &ker)
|
||||
{
|
||||
const mfem::TMOP_Integrator *ti = ker.ti;
|
||||
const real_t metric_normal = ti->metric_normal;
|
||||
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
Array<real_t> mp;
|
||||
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
|
||||
{
|
||||
m->GetWeights(mp);
|
||||
}
|
||||
const real_t *w = mp.Read();
|
||||
|
||||
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
|
||||
const auto X = Reshape(ker.x.Read(), D1D, D1D, 2, NE);
|
||||
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D);
|
||||
const auto J = Reshape(ti->PA.Jtr.Read(), 2, 2, Q1D, Q1D, NE);
|
||||
auto H = Reshape(ti->PA.H.Write(), 2, 2, 2, 2, Q1D, Q1D, NE);
|
||||
|
||||
const Vector &mc = ti->PA.MC;
|
||||
const bool const_m0 = mc.Size() == 1;
|
||||
const auto MC = const_m0
|
||||
? Reshape(mc.Read(), 1, 1, 1)
|
||||
: Reshape(mc.Read(), Q1D, Q1D, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
|
||||
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
|
||||
|
||||
kernels::internal::LoadDofs2d(e, D1D, X, r0);
|
||||
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, e);
|
||||
const real_t detJtr = kernels::Det<2>(Jtr);
|
||||
const real_t m_coef = const_m0 ? MC(0, 0, 0) : MC(qx, qy, e);
|
||||
const real_t weight = metric_normal * m_coef * W(qx, qy) * detJtr;
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
real_t Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^t.DSh
|
||||
const real_t Jpr[4] =
|
||||
{
|
||||
r1[0][0][qy][qx], r1[1][0][qy][qx],
|
||||
r1[0][1][qy][qx], r1[1][1][qy][qx]
|
||||
};
|
||||
|
||||
// Jpt = Jpr.Jrt
|
||||
real_t Jpt[4];
|
||||
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
|
||||
|
||||
METRIC{}.AssembleH(qx, qy, e, weight, Jpt, w, H);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,145 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleGradPA_C0_2D(const real_t lim_normal,
|
||||
const ConstDeviceCube &LD,
|
||||
const bool const_c0,
|
||||
const DeviceTensor<3, const real_t> &C0,
|
||||
const int NE,
|
||||
const DeviceTensor<5, const real_t> &J,
|
||||
const ConstDeviceMatrix &W,
|
||||
const real_t *b,
|
||||
const real_t *bld,
|
||||
const DeviceTensor<4, const real_t> &X0,
|
||||
const DeviceTensor<4, const real_t> &X1,
|
||||
DeviceTensor<5> &H0,
|
||||
const bool exp_lim,
|
||||
const int d1d, const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t sB[MD1][MQ1];
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
|
||||
|
||||
kernels::internal::s_regs2d_t<MQ1> rm0, rm1; // scalar LD
|
||||
kernels::internal::LoadDofs2d(e, D1D, LD, rm0);
|
||||
kernels::internal::Eval2d(D1D, Q1D, smem, sB, rm0, rm1);
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
kernels::internal::v_regs2d_t<2,MQ1> r00, r01; // vector X0
|
||||
kernels::internal::LoadDofs2d(e, D1D, X0, r00);
|
||||
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r00, r01);
|
||||
|
||||
kernels::internal::v_regs2d_t<2,MQ1> r10, r11; // vector X1
|
||||
kernels::internal::LoadDofs2d(e, D1D, X1, r10);
|
||||
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r10, r11);
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, e);
|
||||
const real_t detJtr = kernels::Det<2>(Jtr);
|
||||
const real_t weight = W(qx, qy) * detJtr;
|
||||
const real_t coeff0 = const_c0 ? C0(0, 0, 0) : C0(qx, qy, e);
|
||||
const real_t weight_m = weight * lim_normal * coeff0;
|
||||
|
||||
const real_t D = rm1(qy, qx);
|
||||
const real_t p0[2] = { r01(0, qy, qx), r01(1, qy, qx) };
|
||||
const real_t p1[2] = { r11(0, qy, qx), r11(1, qy, qx) };
|
||||
|
||||
const real_t dist = D; // GetValues, default comp set to 0
|
||||
|
||||
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
|
||||
real_t grad_grad[4];
|
||||
|
||||
if (!exp_lim)
|
||||
{
|
||||
// d2.Diag(1.0 / (dist * dist), x.Size());
|
||||
const real_t c = 1.0 / (dist * dist);
|
||||
kernels::Diag<2>(c, grad_grad);
|
||||
}
|
||||
else
|
||||
{
|
||||
real_t tmp[2];
|
||||
kernels::Subtract<2>(1.0, p1, p0, tmp);
|
||||
real_t dsq = kernels::DistanceSquared<2>(p1, p0);
|
||||
real_t dist_squared = dist * dist;
|
||||
real_t dist_squared_squared = dist_squared * dist_squared;
|
||||
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
|
||||
grad_grad[0] =
|
||||
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
|
||||
(20.0 * f / dist_squared);
|
||||
grad_grad[1] =
|
||||
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
|
||||
grad_grad[2] = grad_grad[1];
|
||||
grad_grad[3] =
|
||||
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
|
||||
(20.0 * f / dist_squared);
|
||||
}
|
||||
ConstDeviceMatrix gg(grad_grad, 2, 2);
|
||||
|
||||
for (int i = 0; i < 2; i++)
|
||||
{
|
||||
for (int j = 0; j < 2; j++)
|
||||
{
|
||||
H0(i, j, qx, qy, e) = weight_m * gg(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef2D, TMOP_AssembleGradPA_C0_2D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef2D);
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &x) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const real_t ln = lim_normal;
|
||||
const bool const_c0 = PA.C0.Size() == 1;
|
||||
const auto C0 = PA.C0.Size() == 1
|
||||
? Reshape(PA.C0.Read(), 1, 1, 1)
|
||||
: Reshape(PA.C0.Read(), q, q, NE);
|
||||
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
|
||||
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
|
||||
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
|
||||
const auto LD = Reshape(PA.LD.Read(), d, d, NE);
|
||||
const auto XL = Reshape(PA.XL.Read(), d, d, 2, NE);
|
||||
const auto X = Reshape(x.Read(), d, d, 2, NE);
|
||||
auto H0 = Reshape(PA.H0.Write(), 2, 2, q, q, NE);
|
||||
|
||||
const auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
|
||||
const bool exp_lim = el ? true : false;
|
||||
|
||||
TMOPAssembleGradCoef2D::Run(d, q, ln, LD, const_c0, C0, NE,
|
||||
J, W, b, bld, XL, X, H0, exp_lim, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,35 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_3D(const Vector &x) const
|
||||
{
|
||||
const int mid = metric->Id();
|
||||
|
||||
// Calls TMOPAssembleGradPA3D::Mult for the given mid.
|
||||
TMOPAssembleGradPA3D ker(this, x);
|
||||
if (mid == 302) { return tmop::Kernel<302>(ker); }
|
||||
if (mid == 303) { return tmop::Kernel<303>(ker); }
|
||||
if (mid == 315) { return tmop::Kernel<315>(ker); }
|
||||
if (mid == 318) { return tmop::Kernel<318>(ker); }
|
||||
if (mid == 321) { return tmop::Kernel<321>(ker); }
|
||||
if (mid == 332) { return tmop::Kernel<332>(ker); }
|
||||
if (mid == 338) { return tmop::Kernel<338>(ker); }
|
||||
|
||||
MFEM_ABORT("Unsupported TMOP metric " << mid);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,111 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class TMOPAssembleGradPA3D
|
||||
{
|
||||
const TMOP_Integrator *ti; // not owned
|
||||
const Vector &x;
|
||||
|
||||
public:
|
||||
TMOPAssembleGradPA3D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
|
||||
x(x) {}
|
||||
|
||||
int Ndof() const { return ti->PA.maps->ndof; }
|
||||
int Nqpt() const { return ti->PA.maps->nqpt; }
|
||||
|
||||
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
|
||||
static void Mult(TMOPAssembleGradPA3D &ker)
|
||||
{
|
||||
const TMOP_Integrator *ti = ker.ti;
|
||||
const real_t metric_normal = ti->metric_normal;
|
||||
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
|
||||
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
Array<real_t> mp;
|
||||
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
|
||||
{
|
||||
m->GetWeights(mp);
|
||||
}
|
||||
const real_t *w = mp.Read();
|
||||
|
||||
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
|
||||
const auto X = Reshape(ker.x.Read(), D1D, D1D, D1D, 3, NE);
|
||||
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D, Q1D);
|
||||
const auto J = Reshape(ti->PA.Jtr.Read(), 3, 3, Q1D, Q1D, Q1D, NE);
|
||||
auto H = Reshape(ti->PA.H.Write(), 3, 3, 3, 3, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
const Vector &mc = ti->PA.MC;
|
||||
const bool const_m0 = mc.Size() == 1;
|
||||
const auto MC = const_m0
|
||||
? Reshape(mc.Read(), 1, 1, 1, 1)
|
||||
: Reshape(mc.Read(), Q1D, Q1D, Q1D, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
|
||||
kernels::internal::vd_regs3d_t<3, 3, MQ1> r0, r1;
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
|
||||
|
||||
kernels::internal::LoadDofs3d(e, D1D, X, r0);
|
||||
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
|
||||
const real_t detJtr = kernels::Det<3>(Jtr);
|
||||
const real_t m_coef = const_m0 ?
|
||||
MC(0, 0, 0, 0) :
|
||||
MC(qx, qy, qz, e);
|
||||
const real_t weight = metric_normal * m_coef * W(qx, qy, qz) * detJtr;
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
real_t Jrt[9];
|
||||
kernels::CalcInverse<3>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^T.DSh
|
||||
real_t Jpr[9] =
|
||||
{
|
||||
r1(0, 0, qz, qy, qx), r1(1, 0, qz, qy, qx), r1(2, 0, qz, qy, qx),
|
||||
r1(0, 1, qz, qy, qx), r1(1, 1, qz, qy, qx), r1(2, 1, qz, qy, qx),
|
||||
r1(0, 2, qz, qy, qx), r1(1, 2, qz, qy, qx), r1(2, 2, qz, qy, qx)
|
||||
};
|
||||
|
||||
// Jpt = X^T . DS = (X^T.DSh) . Jrt = Jpr . Jrt
|
||||
real_t Jpt[9];
|
||||
kernels::Mult(3, 3, 3, Jpr, Jrt, Jpt);
|
||||
|
||||
METRIC{}.AssembleH(qx, qy, qz, e, weight, Jrt, Jpr, Jpt, w, H);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,167 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AssembleGradPA_C0_3D(const real_t lim_normal,
|
||||
const DeviceTensor<4, const real_t> &LD,
|
||||
const bool const_c0,
|
||||
const DeviceTensor<4, const real_t> &C0,
|
||||
const int NE,
|
||||
const DeviceTensor<6, const real_t> &J,
|
||||
const ConstDeviceCube &W,
|
||||
const real_t *b,
|
||||
const real_t *bld,
|
||||
const DeviceTensor<5, const real_t> &X0,
|
||||
const DeviceTensor<5, const real_t> &X1,
|
||||
DeviceTensor<6> &H0,
|
||||
const bool exp_lim,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t sB[MD1][MQ1];
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
|
||||
|
||||
kernels::internal::s_regs3d_t<MQ1> rm0, rm1; // scalar LD
|
||||
kernels::internal::LoadDofs3d(e, D1D, LD, rm0);
|
||||
kernels::internal::Eval3d(D1D, Q1D, smem, sB, rm0, rm1);
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
|
||||
kernels::internal::v_regs3d_t<3,MQ1> r00, r01; // vector X0
|
||||
kernels::internal::LoadDofs3d(e, D1D, X0, r00);
|
||||
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r00, r01);
|
||||
|
||||
kernels::internal::v_regs3d_t<3,MQ1> r10, r11; // vector X1
|
||||
kernels::internal::LoadDofs3d(e, D1D, X1, r10);
|
||||
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r10, r11);
|
||||
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
|
||||
const real_t detJtr = kernels::Det<3>(Jtr);
|
||||
const real_t weight = W(qx, qy, qz) * detJtr;
|
||||
const real_t coeff0 = const_c0
|
||||
? C0(0, 0, 0, 0)
|
||||
: C0(qx, qy, qz, e);
|
||||
const real_t weight_m = weight * lim_normal * coeff0;
|
||||
const real_t D = rm1(qz, qy, qx);
|
||||
const real_t p0[3] = { r01(0, qz, qy, qx),
|
||||
r01(1, qz, qy, qx),
|
||||
r01(2, qz, qy, qx)
|
||||
};
|
||||
const real_t p1[3] = { r11(0, qz, qy, qx),
|
||||
r11(1, qz, qy, qx),
|
||||
r11(2, qz, qy, qx)
|
||||
};
|
||||
|
||||
const real_t dist = D; // GetValues, default comp set to 0
|
||||
|
||||
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
|
||||
|
||||
real_t grad_grad[9];
|
||||
|
||||
if (!exp_lim)
|
||||
{
|
||||
// d2.Diag(1.0 / (dist * dist), x.Size());
|
||||
const real_t c = 1.0 / (dist * dist);
|
||||
kernels::Diag<3>(c, grad_grad);
|
||||
}
|
||||
else
|
||||
{
|
||||
real_t tmp[3];
|
||||
kernels::Subtract<3>(1.0, p1, p0, tmp);
|
||||
real_t dsq = kernels::DistanceSquared<3>(p1, p0);
|
||||
real_t dist_squared = dist * dist;
|
||||
real_t dist_squared_squared = dist_squared * dist_squared;
|
||||
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
|
||||
grad_grad[0] =
|
||||
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
|
||||
(20.0 * f / dist_squared);
|
||||
grad_grad[1] =
|
||||
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
|
||||
grad_grad[2] =
|
||||
(400.0 * tmp[0] * tmp[2] * f) / dist_squared_squared;
|
||||
grad_grad[3] = grad_grad[1];
|
||||
grad_grad[4] =
|
||||
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
|
||||
(20.0 * f / dist_squared);
|
||||
grad_grad[5] =
|
||||
(400.0 * tmp[1] * tmp[2] * f) / dist_squared_squared;
|
||||
grad_grad[6] = grad_grad[2];
|
||||
grad_grad[7] = grad_grad[5];
|
||||
grad_grad[8] =
|
||||
((400.0 * tmp[2] * tmp[2] * f) / dist_squared_squared) +
|
||||
(20.0 * f / dist_squared);
|
||||
}
|
||||
ConstDeviceMatrix gg(grad_grad, 3, 3);
|
||||
|
||||
for (int i = 0; i < 3; i++)
|
||||
{
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
H0(i, j, qx, qy, qz, e) = weight_m * gg(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef3D, TMOP_AssembleGradPA_C0_3D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef3D);
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_C0_3D(const Vector &x) const
|
||||
{
|
||||
const real_t ln = lim_normal;
|
||||
const bool const_c0 = PA.C0.Size() == 1;
|
||||
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto C0 = const_c0
|
||||
? Reshape(PA.C0.Read(), 1, 1, 1, 1)
|
||||
: Reshape(PA.C0.Read(), q, q, q, NE);
|
||||
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
|
||||
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
|
||||
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
|
||||
const auto LD = Reshape(PA.LD.Read(), d, d, d, NE);
|
||||
const auto XL = Reshape(PA.XL.Read(), d, d, d, 3, NE);
|
||||
const auto X = Reshape(x.Read(), d, d, d, 3, NE);
|
||||
auto H0 = Reshape(PA.H0.Write(), 3, 3, q, q, q, NE);
|
||||
|
||||
auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
|
||||
const bool exp_lim = (el) ? true : false;
|
||||
|
||||
TMOPAssembleGradCoef3D::Run(d, q, ln, LD, const_c0, C0, NE,
|
||||
J, W, b, bld, XL, X, H0, exp_lim, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,76 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_001 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
return ie.Get_I1();
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI1[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1(dI1));
|
||||
kernels::Set(2, 2, 1.0, ie.Get_dI1(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// weight * ddI1
|
||||
real_t ddI1[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t h = ddi1(r, c);
|
||||
H(r, c, i, j, qx, qy, e) = weight * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_001;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 1);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,74 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_002 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
return 0.5 * ie.Get_I1b() - 1.0;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI1b[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1b(dI1b).dI2b(dI2b));
|
||||
kernels::Set(2, 2, 1. / 2., ie.Get_dI1b(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx, const int qy, const int e, const real_t weight,
|
||||
const real_t (&Jpt)[4], const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// 0.5 * weight * dI1b
|
||||
real_t ddI1[4], ddI1b[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b));
|
||||
const real_t half_weight = 0.5 * weight;
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t h = ddi1b(r, c);
|
||||
H(r, c, i, j, qx, qy, e) = half_weight * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_002;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 2);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,88 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_007 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
return ie.Get_I1() * (1.0 + 1.0 / ie.Get_I2()) - 4.0;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI1[4], dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).dI1(dI1).dI2(dI2).dI2b(dI2b));
|
||||
const real_t I2 = ie.Get_I2();
|
||||
kernels::Add(2, 2, 1.0 + 1.0 / I2, ie.Get_dI1(), -ie.Get_I1() / (I2 * I2),
|
||||
ie.Get_dI2(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).ddI1(ddI1).ddI2(ddI2).dI1(dI1).dI2(dI2).dI2b(dI2b));
|
||||
const real_t c1 = 1. / ie.Get_I2();
|
||||
const real_t c2 = weight * c1 * c1;
|
||||
const real_t c3 = ie.Get_I1() * c2;
|
||||
ConstDeviceMatrix di1(ie.Get_dI1(), DIM, DIM);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r, c, i, j, qx, qy, e) =
|
||||
weight * (1.0 + c1) * ddi1(r, c) - c3 * ddi2(r, c) -
|
||||
c2 * (di1(i, j) * di2(r, c) + di2(i, j) * di1(r, c)) +
|
||||
2.0 * c1 * c3 * di2(r, c) * di2(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_007;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 7);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,82 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_056 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
return 0.5 * (I2b + 1.0 / I2b) - 1.0;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// 0.5*(1 - 1/I2b^2)*dI2b
|
||||
real_t dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b));
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// (0.5 - 0.5/I2b^2)*ddI2b + (1/I2b^3)*(dI2b x dI2b)
|
||||
real_t dI2b[4], ddI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b).ddI2b(ddI2b));
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r, c, i, j, qx, qy, e) =
|
||||
weight * (0.5 - 0.5 / (I2b * I2b)) * ddi2b(r, c) +
|
||||
weight / (I2b * I2b * I2b) * di2b(r, c) * di2b(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_056;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 56);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,81 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_077 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
return 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2(dI2).dI2b(dI2b));
|
||||
const real_t I2 = ie.Get_I2();
|
||||
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI2[4], dI2b[4], ddI2[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).dI2(dI2).dI2b(dI2b).ddI2(ddI2));
|
||||
const real_t I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r, c, i, j, qx, qy, e) =
|
||||
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r, c) +
|
||||
weight * (I2inv_sq / I2) * di2(r, c) * di2(i, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_077;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 77);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,89 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_080 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
const real_t eval_w_77 = 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
|
||||
return w[0] * eval_w_02 + w[1] * eval_w_77;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// w0 P_2 + w1 P_77
|
||||
real_t dI1b[4], dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).dI1b(dI1b).dI2(dI2).dI2b(dI2b));
|
||||
kernels::Set(2, 2, w[0] * 0.5, ie.Get_dI1b(), P);
|
||||
const real_t I2 = ie.Get_I2();
|
||||
kernels::Add(2, 2, w[1] * 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
// w0 H_2 + w1 H_77
|
||||
real_t ddI1[4], ddI1b[4], dI2[4], dI2b[4], ddI2[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).dI2(dI2).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b).ddI2(ddI2));
|
||||
|
||||
const real_t I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r, c, i, j, qx, qy, e) =
|
||||
w[0] * 0.5 * weight * ddi1b(r, c) +
|
||||
w[1] * (weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r, c) +
|
||||
weight * (I2inv_sq / I2) * di2(r, c) * di2(i, j));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_080;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 80);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,88 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult2.hpp"
|
||||
#include "../tools/energy2.hpp"
|
||||
#include "../assemble/grad2.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_094 : TMOP_PA_Metric_2D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
|
||||
{
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
|
||||
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
const real_t eval_w_56 = 0.5 * (I2b + 1.0 / I2b) - 1.0;
|
||||
return w[0] * eval_w_02 + w[1] * eval_w_56;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
|
||||
{
|
||||
// w0 P_2 + w1 P_56
|
||||
real_t dI1b[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1b(dI1b).dI2b(dI2b));
|
||||
kernels::Set(2, 2, w[0] * 0.5, ie.Get_dI1b(), P);
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
kernels::Add(2, 2, w[1] * 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(),
|
||||
P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
const real_t (&Jpt)[4],
|
||||
const real_t *w,
|
||||
const DeviceTensor<7> &H) override
|
||||
{
|
||||
// w0 H_2 + w1 H_56
|
||||
real_t ddI1[4], ddI1b[4], dI2b[4], ddI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(
|
||||
Args().J(Jpt).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b).ddI2b(ddI2b));
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r, c, i, j, qx, qy, e) =
|
||||
w[0] * 0.5 * weight * ddi1b(r, c) +
|
||||
w[1] *
|
||||
(weight * (0.5 - 0.5 / (I2b * I2b)) * ddi2b(r, c) +
|
||||
weight / (I2b * I2b * I2b) * di2b(r, c) * di2b(i, j));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_094;
|
||||
|
||||
using assemble = TMOPAssembleGradPA2D;
|
||||
using energy = TMOPEnergyPA2D;
|
||||
using mult = TMOPAddMultPA2D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 94);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,111 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_302 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
// I1b * I2b / 9 - 1
|
||||
return ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// (I1b/9)*dI2b + (I2b/9)*dI1b
|
||||
real_t B[9];
|
||||
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(
|
||||
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
|
||||
const real_t alpha = ie.Get_I1b() / 9.;
|
||||
const real_t beta = ie.Get_I2b() / 9.;
|
||||
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(Jrt);
|
||||
MFEM_CONTRACT_VAR(Jpr);
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t B[9];
|
||||
real_t dI1b[9], ddI1b[9];
|
||||
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
|
||||
real_t dI3b[9]; // = Jrt;
|
||||
// (dI2b*dI1b + dI1b*dI2b)/9 + (I1b/9)*ddI2b + (I2b/9)*ddI1b
|
||||
kernels::InvariantsEvaluator3D ie(Args()
|
||||
.J(Jpt)
|
||||
.B(B)
|
||||
.dI1b(dI1b)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2)
|
||||
.ddI2b(ddI2b)
|
||||
.dI3b(dI3b));
|
||||
|
||||
const real_t c1 = weight / 9.;
|
||||
const real_t I1b = ie.Get_I1b();
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
|
||||
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp =
|
||||
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
|
||||
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
|
||||
H(r, c, i, j, qx, qy, qz, e) = c1 * dp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_302;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 302);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,103 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_303 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
// mu_303 = I1b/3 - 1
|
||||
return ie.Get_I1b() / 3. - 1.;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// dI1b/3
|
||||
real_t B[9];
|
||||
real_t dI1b[9], dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(
|
||||
Args().J(Jpt).B(B).dI1b(dI1b).dI3b(dI3b));
|
||||
kernels::Set(3, 3, 1. / 3., ie.Get_dI1b(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t B[9];
|
||||
real_t dI1b[9], ddI1[9], ddI1b[9];
|
||||
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
|
||||
// ddI1b/3
|
||||
kernels::InvariantsEvaluator3D ie(Args()
|
||||
.J(Jpt)
|
||||
.B(B)
|
||||
.dI1b(dI1b)
|
||||
.ddI1(ddI1)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2)
|
||||
.ddI2b(ddI2b)
|
||||
.dI3b(dI3b)
|
||||
.ddI3b(ddI3b));
|
||||
|
||||
const real_t c1 = weight / 3.;
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp = ddi1b(r, c);
|
||||
H(r, c, i, j, qx, qy, qz, e) = c1 * dp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_303;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 303);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,91 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_315 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
// (I3b - 1)^2
|
||||
const real_t a = ie.Get_I3b() - 1.0;
|
||||
return a * a;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// 2*(I3b - 1)*dI3b
|
||||
real_t dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
kernels::Set(3, 3, 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
// 2*(dI3b x dI3b) + 2*(I3b - 1)*ddI3b
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b).ddI3b(ddI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp = 2.0 * weight * (I3b - 1.0) * ddi3b(r, c) +
|
||||
2.0 * weight * di3b(r, c) * di3b(i, j);
|
||||
H(r, c, i, j, qx, qy, qz, e) = dp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_315;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 315);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,97 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_318 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
// 0.5 * (I3 + 1/I3) - 1.
|
||||
const real_t I3 = ie.Get_I3();
|
||||
return 0.5 * (I3 + 1.0 / I3) - 1.0;
|
||||
}
|
||||
|
||||
// P_318 = (I3b - 1/I3b^3)*dI3b.
|
||||
// Uses the I3b form, as dI3 and ddI3 were not implemented at the time
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b));
|
||||
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
kernels::Set(3, 3, I3b - 1.0 / (I3b * I3b * I3b), ie.Get_dI3b(sign_detJ),
|
||||
P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
// dP_318 = (I3b - 1/I3b^3)*ddI3b + (1 + 3/I3b^4)*(dI3b x dI3b)
|
||||
// Uses the I3b form, as dI3 and ddI3 were not implemented at the time
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b).ddI3b(ddI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp =
|
||||
weight * (I3b - 1.0 / (I3b * I3b * I3b)) * ddi3b(r, c) +
|
||||
weight * (1.0 + 3.0 / (I3b * I3b * I3b * I3b)) *
|
||||
di3b(r, c) * di3b(i, j);
|
||||
H(r, c, i, j, qx, qy, qz, e) = dp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_318;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 318);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,123 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_321 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
// I1 + I2/I3 - 6
|
||||
return ie.Get_I1() + ie.Get_I2() / ie.Get_I3() - 6.0;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
// dI1 + (1/I3)*dI2 - (2*I2/I3b^3)*dI3b
|
||||
real_t B[9];
|
||||
real_t dI1[9], dI2[9], dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(
|
||||
Args().J(Jpt).B(B).dI1(dI1).dI2(dI2).dI3b(dI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t I3 = ie.Get_I3();
|
||||
const real_t alpha = 1.0 / I3;
|
||||
const real_t beta = -2. * ie.Get_I2() / (I3 * ie.Get_I3b(sign_detJ));
|
||||
kernels::Add(3, 3, alpha, ie.Get_dI2(), beta, ie.Get_dI3b(sign_detJ), P);
|
||||
kernels::Add(3, 3, ie.Get_dI1(), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
real_t B[9];
|
||||
real_t dI1b[9], ddI1[9], ddI1b[9];
|
||||
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
|
||||
// ddI1 + (-2/I3b^3)*(dI2 x dI3b + dI3b x dI2)
|
||||
// + (1/I3)*ddI2
|
||||
// + (6*I2/I3b^4)*(dI3b x dI3b)
|
||||
// + (-2*I2/I3b^3)*ddI3b
|
||||
kernels::InvariantsEvaluator3D ie(Args()
|
||||
.J(Jpt)
|
||||
.B(B)
|
||||
.dI1b(dI1b)
|
||||
.ddI1(ddI1)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2)
|
||||
.ddI2b(ddI2b)
|
||||
.dI3b(dI3b)
|
||||
.ddI3b(ddI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t I2 = ie.Get_I2();
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
|
||||
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
|
||||
const real_t c0 = 1.0 / I3b;
|
||||
const real_t c1 = weight * c0 * c0;
|
||||
const real_t c2 = -2 * c0 * c1;
|
||||
const real_t c3 = c2 * I2;
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp =
|
||||
weight * ddi1(r, c) + c1 * ddi2(r, c) + c3 * ddi3b(r, c) +
|
||||
c2 * ((di2(r, c) * di3b(i, j) + di3b(r, c) * di2(i, j))) -
|
||||
3 * c0 * c3 * di3b(r, c) * di3b(i, j);
|
||||
H(r, c, i, j, qx, qy, qz, e) = dp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_321;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 321);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,120 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_332 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
const real_t eval_w_302 = ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
|
||||
const real_t a = ie.Get_I3b() - 1.0;
|
||||
const real_t eval_w_315 = a * a;
|
||||
return w[0] * eval_w_302 + w[1] * eval_w_315;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
// w0 P_302 + w1 P_315
|
||||
real_t B[9];
|
||||
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(
|
||||
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
|
||||
const real_t alpha = w[0] * ie.Get_I1b() / 9.;
|
||||
const real_t beta = w[0] * ie.Get_I2b() / 9.;
|
||||
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
kernels::Add(3, 3, w[1] * 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
real_t B[9];
|
||||
real_t dI1b[9], /*ddI1[9],*/ ddI1b[9];
|
||||
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
// w0 H_302 + w1 H_315
|
||||
kernels::InvariantsEvaluator3D ie(Args()
|
||||
.J(Jpt)
|
||||
.B(B)
|
||||
.dI1b(dI1b)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2)
|
||||
.ddI2b(ddI2b)
|
||||
.dI3b(dI3b)
|
||||
.ddI3b(ddI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t c1 = weight / 9.0;
|
||||
const real_t I1b = ie.Get_I1b();
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
|
||||
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
|
||||
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp_302 =
|
||||
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
|
||||
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
|
||||
const real_t dp_315 =
|
||||
2.0 * weight * (I3b - 1.0) * ddi3b(r, c) +
|
||||
2.0 * weight * di3b(r, c) * di3b(i, j);
|
||||
H(r, c, i, j, qx, qy, qz, e) =
|
||||
w[0] * c1 * dp_302 + w[1] * dp_315;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_332;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 332);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,122 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../mult/mult3.hpp"
|
||||
#include "../tools/energy3.hpp"
|
||||
#include "../assemble/grad3.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct TMOP_PA_Metric_338 : TMOP_PA_Metric_3D
|
||||
{
|
||||
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
|
||||
const real_t *w) override
|
||||
{
|
||||
real_t B[9];
|
||||
MFEM_CONTRACT_VAR(w);
|
||||
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
|
||||
const real_t eval_w_302 = ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
|
||||
const real_t I3 = ie.Get_I3();
|
||||
const real_t eval_w_318 = 0.5 * (I3 + 1.0 / I3) - 1.0;
|
||||
return w[0] * eval_w_302 + w[1] * eval_w_318;
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
|
||||
{
|
||||
// w0 P_302 + w1 P_318
|
||||
real_t B[9];
|
||||
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
|
||||
kernels::InvariantsEvaluator3D ie(
|
||||
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
|
||||
const real_t alpha = w[0] * ie.Get_I1b() / 9.;
|
||||
const real_t beta = w[0] * ie.Get_I2b() / 9.;
|
||||
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
|
||||
real_t sign_detJ;
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
kernels::Add(3, 3, w[1] * (I3b - 1.0 / (I3b * I3b * I3b)),
|
||||
ie.Get_dI3b(sign_detJ), P);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE
|
||||
void AssembleH(const int qx,
|
||||
const int qy,
|
||||
const int qz,
|
||||
const int e,
|
||||
const real_t weight,
|
||||
real_t *Jrt,
|
||||
real_t *Jpr,
|
||||
const real_t (&Jpt)[9],
|
||||
const real_t *w,
|
||||
const DeviceTensor<8> &H) const override
|
||||
{
|
||||
real_t B[9];
|
||||
real_t dI1b[9], ddI1b[9];
|
||||
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
|
||||
real_t *dI3b = Jrt, *ddI3b = Jpr;
|
||||
// w0 H_302 + w1 H_318
|
||||
kernels::InvariantsEvaluator3D ie(Args()
|
||||
.J(Jpt)
|
||||
.B(B)
|
||||
.dI1b(dI1b)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2)
|
||||
.ddI2b(ddI2b)
|
||||
.dI3b(dI3b)
|
||||
.ddI3b(ddI3b));
|
||||
real_t sign_detJ;
|
||||
const real_t c1 = weight / 9.;
|
||||
const real_t I1b = ie.Get_I1b();
|
||||
const real_t I2b = ie.Get_I2b();
|
||||
const real_t I3b = ie.Get_I3b(sign_detJ);
|
||||
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
|
||||
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
|
||||
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
|
||||
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const real_t dp_302 =
|
||||
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
|
||||
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
|
||||
const real_t dp_318 =
|
||||
weight * (I3b - 1.0 / (I3b * I3b * I3b)) * ddi3b(r, c) +
|
||||
weight * (1.0 + 3.0 / (I3b * I3b * I3b * I3b)) *
|
||||
di3b(r, c) * di3b(i, j);
|
||||
H(r, c, i, j, qx, qy, qz, e) =
|
||||
w[0] * c1 * dp_302 + w[1] * dp_318;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
using metric = TMOP_PA_Metric_338;
|
||||
|
||||
using assemble = TMOPAssembleGradPA3D;
|
||||
using energy = TMOPEnergyPA3D;
|
||||
using mult = TMOPAddMultPA3D;
|
||||
|
||||
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 338);
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,118 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AddMultGradPA_2D(const int NE,
|
||||
const real_t *b,
|
||||
const real_t *g,
|
||||
const DeviceTensor<5, const real_t> &J,
|
||||
const DeviceTensor<7, const real_t> &H,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
DeviceTensor<4> &Y,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
|
||||
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
|
||||
|
||||
kernels::internal::LoadDofs2d(e, D1D, X, r0);
|
||||
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
real_t Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^T.DSh
|
||||
const real_t Jpr[4] =
|
||||
{
|
||||
r1[0][0][qy][qx], r1[1][0][qy][qx],
|
||||
r1[0][1][qy][qx], r1[1][1][qy][qx]
|
||||
};
|
||||
|
||||
// Jpt = Jpr . Jrt
|
||||
real_t Jpt[4];
|
||||
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
|
||||
|
||||
// B = Jpt : H
|
||||
real_t B[4];
|
||||
DeviceMatrix M(B, 2, 2);
|
||||
ConstDeviceMatrix J(Jpt, 2, 2);
|
||||
for (int i = 0; i < 2; i++)
|
||||
{
|
||||
for (int j = 0; j < 2; j++)
|
||||
{
|
||||
M(i, j) = 0.0;
|
||||
for (int r = 0; r < 2; r++)
|
||||
{
|
||||
for (int c = 0; c < 2; c++)
|
||||
{
|
||||
M(i, j) += H(r, c, i, j, qx, qy, e) * J(r, c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// C = Jrt . B
|
||||
real_t C[4];
|
||||
kernels::MultABt(2, 2, 2, Jrt, B, C);
|
||||
r0[0][0][qy][qx] = C[0], r0[0][1][qy][qx] = C[1];
|
||||
r0[1][0][qy][qx] = C[2], r0[1][1][qy][qx] = C[3];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradKernels, TMOP_AddMultGradPA_2D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradKernels);
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_2D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto *b = PA.maps->B.Read(), *g = PA.maps->G.Read();
|
||||
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
|
||||
const auto H = Reshape(PA.H.Read(), 2, 2, 2, 2, q, q, NE);
|
||||
const auto X = Reshape(R.Read(), d, d, 2, NE);
|
||||
auto Y = Reshape(C.ReadWrite(), d, d, 2, NE);
|
||||
|
||||
TMOPMultGradKernels::Run(d, q, NE, b, g, J, H, X, Y, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,88 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AddMultGradPA_C0_2D(const int NE,
|
||||
const real_t *b,
|
||||
const DeviceTensor<5, const real_t> &H0,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
DeviceTensor<4> &Y,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t sB[MD1][MQ1];
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
|
||||
kernels::internal::v_regs2d_t<2,MQ1> r0, r1;
|
||||
kernels::internal::LoadDofs2d(e, D1D, X, r0);
|
||||
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1);
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
// Xh = X^T . Sh
|
||||
const real_t Xh[2] = { r1(0, qy, qx), r1(1, qy, qx) };
|
||||
|
||||
real_t H_data[4];
|
||||
DeviceMatrix H(H_data, 2, 2);
|
||||
for (int i = 0; i < 2; i++)
|
||||
{
|
||||
for (int j = 0; j < 2; j++) { H(i, j) = H0(i, j, qx, qy, e); }
|
||||
}
|
||||
|
||||
// p2 = H . Xh
|
||||
real_t p2[2];
|
||||
kernels::Mult(2, 2, H_data, Xh, p2);
|
||||
r0(0,qy,qx) = p2[0];
|
||||
r0(1,qy,qx) = p2[1];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
|
||||
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradCoefKernels, TMOP_AddMultGradPA_C0_2D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradCoefKernels);
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_C0_2D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto H0 = Reshape(PA.H0.Read(), 2, 2, q, q, NE);
|
||||
const auto *b = PA.maps->B.Read();
|
||||
const auto X = Reshape(R.Read(), d, d, 2, NE);
|
||||
auto Y = Reshape(C.ReadWrite(), d, d, 2, NE);
|
||||
|
||||
TMOPMultGradCoefKernels::Run(d, q, NE, b, H0, X, Y, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,121 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AddMultGradPA_3D(const int NE,
|
||||
const real_t *b,
|
||||
const real_t *g,
|
||||
const DeviceTensor<6, const real_t> &J,
|
||||
const DeviceTensor<8, const real_t> &H,
|
||||
const DeviceTensor<5, const real_t> &X,
|
||||
DeviceTensor<5> &Y, const int d1d, const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
|
||||
kernels::internal::vd_regs3d_t<3, 3, MQ1> r0, r1;
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
|
||||
|
||||
kernels::internal::LoadDofs3d(e, D1D, X, r0);
|
||||
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
real_t Jrt[9];
|
||||
kernels::CalcInverse<3>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^T.DSh
|
||||
const real_t Jpr[9] =
|
||||
{
|
||||
r1(0, 0, qz, qy, qx), r1(1, 0, qz, qy, qx), r1(2, 0, qz, qy, qx),
|
||||
r1(0, 1, qz, qy, qx), r1(1, 1, qz, qy, qx), r1(2, 1, qz, qy, qx),
|
||||
r1(0, 2, qz, qy, qx), r1(1, 2, qz, qy, qx), r1(2, 2, qz, qy, qx)
|
||||
};
|
||||
|
||||
// Jpt = X^T.DS = (X^T.DSh).Jrt = Jpr.Jrt
|
||||
real_t Jpt[9];
|
||||
kernels::Mult(3, 3, 3, Jpr, Jrt, Jpt);
|
||||
|
||||
// B = Jpt : H
|
||||
real_t B[9];
|
||||
DeviceMatrix M(B, 3, 3);
|
||||
ConstDeviceMatrix J(Jpt, 3, 3);
|
||||
for (int i = 0; i < 3; i++)
|
||||
{
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
M(i, j) = 0.0;
|
||||
for (int r = 0; r < 3; r++)
|
||||
{
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
M(i, j) += H(r, c, i, j, qx, qy, qz, e) * J(r, c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Y += DS . M^t += DSh . (Jrt . M^t)
|
||||
real_t A[9];
|
||||
kernels::MultABt(3, 3, 3, Jrt, B, A);
|
||||
r0(0,0, qz,qy,qx) = A[0], r0(0,1, qz,qy,qx) = A[1], r0(0,2, qz,qy,qx) = A[2];
|
||||
r0(1,0, qz,qy,qx) = A[3], r0(1,1, qz,qy,qx) = A[4], r0(1,2, qz,qy,qx) = A[5];
|
||||
r0(2,0, qz,qy,qx) = A[6], r0(2,1, qz,qy,qx) = A[7], r0(2,2, qz,qy,qx) = A[8];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradKernels3D, TMOP_AddMultGradPA_3D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradKernels3D);
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_3D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto *b = PA.maps->B.Read(), *g = PA.maps->G.Read();
|
||||
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
|
||||
const auto X = Reshape(R.Read(), d, d, d, 3, NE);
|
||||
const auto H = Reshape(PA.H.Read(), 3, 3, 3, 3, q, q, q, NE);
|
||||
auto Y = Reshape(C.ReadWrite(), d, d, d, 3, NE);
|
||||
|
||||
TMOPMultGradKernels3D::Run(d, q, NE, b, g, J, H, X, Y, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,101 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../pa.hpp"
|
||||
#include "../../tmop.hpp"
|
||||
#include "../../kernels.hpp"
|
||||
#include "../../../general/forall.hpp"
|
||||
#include "../../../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
|
||||
void TMOP_AddMultGradPA_C0_3D(const int NE,
|
||||
const real_t *b,
|
||||
const DeviceTensor<6, const real_t> &H0,
|
||||
const DeviceTensor<5, const real_t> &X,
|
||||
DeviceTensor<5> &Y,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_SHARED real_t sB[MD1][MQ1];
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
|
||||
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
|
||||
|
||||
kernels::internal::v_regs3d_t<3,MQ1> r0, r1; // vector X
|
||||
kernels::internal::LoadDofs3d(e, D1D, X, r0);
|
||||
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1);
|
||||
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
// Xh = X^T . Sh
|
||||
const real_t Xh[3] =
|
||||
{
|
||||
r1(0, qz, qy, qx),
|
||||
r1(1, qz, qy, qx),
|
||||
r1(2, qz, qy, qx)
|
||||
};
|
||||
|
||||
real_t H_data[9];
|
||||
DeviceMatrix H(H_data, 3, 3);
|
||||
for (int i = 0; i < 3; i++)
|
||||
{
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
H(i, j) = H0(i, j, qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
|
||||
// p2 = H . Xh
|
||||
real_t p2[3];
|
||||
kernels::Mult(3, 3, H_data, Xh, p2);
|
||||
r0(0,qz,qy,qx) = p2[0];
|
||||
r0(1,qz,qy,qx) = p2[1];
|
||||
r0(2,qz,qy,qx) = p2[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
|
||||
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
|
||||
});
|
||||
}
|
||||
|
||||
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradCoefKernels3D, TMOP_AddMultGradPA_C0_3D);
|
||||
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradCoefKernels3D);
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_C0_3D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
|
||||
|
||||
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
|
||||
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
|
||||
|
||||
const auto H0 = Reshape(PA.H0.Read(), 3, 3, q, q, q, NE);
|
||||
const auto *b = PA.maps->B.Read();
|
||||
const auto X = Reshape(R.Read(), d, d, d, 3, NE);
|
||||
auto Y = Reshape(C.ReadWrite(), d, d, d, 3, NE);
|
||||
|
||||
TMOPMultGradCoefKernels3D::Run(d, q, NE, b, H0, X, Y, d, q);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user