Compare commits
749
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5457706a29 | ||
|
|
fd195d53a5 | ||
|
|
ced299b0b3 | ||
|
|
667022093d | ||
|
|
5a3e9a93c2 | ||
|
|
b2bf589c87 | ||
|
|
e52948f9e5 | ||
|
|
c860bf20ea | ||
|
|
ac48dfcfa5 | ||
|
|
7ff0bd3bb0 | ||
|
|
3e31395f85 | ||
|
|
b2de4c4ba1 | ||
|
|
5e98b82b26 | ||
|
|
dfc582149d | ||
|
|
79680d9bc9 | ||
|
|
a713e386c2 | ||
|
|
4155b0bdda | ||
|
|
dbbd425a22 | ||
|
|
156f338e49 | ||
|
|
53581cb5b7 | ||
|
|
f261928799 | ||
|
|
12509fda28 | ||
|
|
a9b36b1e5e | ||
|
|
64ef39bbe6 | ||
|
|
7985a225bb | ||
|
|
f4baeb41ad | ||
|
|
fd73f3895f | ||
|
|
5820d70a50 | ||
|
|
d923432284 | ||
|
|
cffd9c618f | ||
|
|
6ccc490e18 | ||
|
|
80f973d708 | ||
|
|
ef13f6e9d8 | ||
|
|
48cb7bcc2e | ||
|
|
621a9ed1d6 | ||
|
|
5ca23fd56f | ||
|
|
1b9d97ec28 | ||
|
|
4add547b04 | ||
|
|
9c9b5cee2f | ||
|
|
80b5ecfb7b | ||
|
|
074a716634 | ||
|
|
ad9682118d | ||
|
|
25cba3afa5 | ||
|
|
76f061dd9c | ||
|
|
22f25269a5 | ||
|
|
684d8fc64b | ||
|
|
ded0173a92 | ||
|
|
7aca674961 | ||
|
|
72f83edd53 | ||
|
|
be6f6299aa | ||
|
|
6c5f513eaa | ||
|
|
c81fabcc54 | ||
|
|
f963bd897b | ||
|
|
dcd5bee0f6 | ||
|
|
75cc8433e9 | ||
|
|
f700d97549 | ||
|
|
ec39b3509c | ||
|
|
449ec725e2 | ||
|
|
399d8e1e9b | ||
|
|
7330aca4e6 | ||
|
|
9ebfcf05af | ||
|
|
10dbed9658 | ||
|
|
be1db1e4b7 | ||
|
|
37fcdc1816 | ||
|
|
0af98d7ff6 | ||
|
|
cb6192167c | ||
|
|
bdf6aa6369 | ||
|
|
1dd20334e2 | ||
|
|
da40ac4f2d | ||
|
|
f09a062c04 | ||
|
|
8889988956 | ||
|
|
06b4a68c7d | ||
|
|
17e15d08cd | ||
|
|
c05ce7dacd | ||
|
|
c80f7fb1f1 | ||
|
|
0eac62aa3b | ||
|
|
fbe07d97ea | ||
|
|
0c97d6f375 | ||
|
|
bff5d5e0cb | ||
|
|
26a152fb11 | ||
|
|
e4ce8375f3 | ||
|
|
aed9c8ef4a | ||
|
|
e4e85e28ef | ||
|
|
fff973f192 | ||
|
|
da959ef7e9 | ||
|
|
aba87d9e14 | ||
|
|
f0efcf4253 | ||
|
|
7aee5f56ba | ||
|
|
9bd3409458 | ||
|
|
775f06c43b | ||
|
|
18ff1d8289 | ||
|
|
bca03a17af | ||
|
|
5a0962c674 | ||
|
|
cca1678ccb | ||
|
|
97ca2f9ecc | ||
|
|
6479b2607d | ||
|
|
c1de6939f9 | ||
|
|
a00f222761 | ||
|
|
99ebc58be4 | ||
|
|
cf213ea6b6 | ||
|
|
031be712a1 | ||
|
|
134bc32d93 | ||
|
|
e640a3e3fb | ||
|
|
faba224c26 | ||
|
|
0d999709e6 | ||
|
|
9300f47c83 | ||
|
|
75e49b217c | ||
|
|
c2649eb998 | ||
|
|
a1ce49fb57 | ||
|
|
f58cfc8170 | ||
|
|
6c837d2954 | ||
|
|
5b37c3b595 | ||
|
|
b46baa5f5e | ||
|
|
e49f9f7988 | ||
|
|
7c36b55628 | ||
|
|
485121d3ad | ||
|
|
af527e27d1 | ||
|
|
3e680af733 | ||
|
|
4212310405 | ||
|
|
faa73ef554 | ||
|
|
ecb6b06aa0 | ||
|
|
af4649a088 | ||
|
|
a9f58f3982 | ||
|
|
6de6675783 | ||
|
|
085ee02a29 | ||
|
|
9a124335a7 | ||
|
|
91d5e490aa | ||
|
|
610196629e | ||
|
|
8453b4008d | ||
|
|
fab2afd8dc | ||
|
|
dd931b2584 | ||
|
|
8a42ea2834 | ||
|
|
24e5d5fc0a | ||
|
|
6722dd7a70 | ||
|
|
cb862cbfa1 | ||
|
|
f7445844ba | ||
|
|
672e2a442b | ||
|
|
3e1f10daea | ||
|
|
416536eb9d | ||
|
|
35778347d0 | ||
|
|
f557e348da | ||
|
|
881598e5da | ||
|
|
564b7ab4ec | ||
|
|
3f2f925400 | ||
|
|
463e34dc7f | ||
|
|
55e42eeefe | ||
|
|
077954d4b3 | ||
|
|
9a456b908e | ||
|
|
616839388a | ||
|
|
2fda3db982 | ||
|
|
4823a33a6a | ||
|
|
a96319e0be | ||
|
|
5f4283f512 | ||
|
|
8735d28561 | ||
|
|
c9f7a90f81 | ||
|
|
3b35d8210d | ||
|
|
3babbe993b | ||
|
|
5f5421fde2 | ||
|
|
8644c8a8dd | ||
|
|
b863dd186f | ||
|
|
66702d831c | ||
|
|
a10c7a943b | ||
|
|
60ab6ab8f5 | ||
|
|
878df1fef2 | ||
|
|
a1758e51e5 | ||
|
|
ccf84aab7c | ||
|
|
41aed0e916 | ||
|
|
96eff4684f | ||
|
|
b6255fc825 | ||
|
|
18d27f6ffb | ||
|
|
7bfb57ef17 | ||
|
|
ab394d795e | ||
|
|
cad9cc4c82 | ||
|
|
4dc741ca48 | ||
|
|
918eb114d3 | ||
|
|
3341acf0f7 | ||
|
|
287cb24d0a | ||
|
|
70370b6241 | ||
|
|
9fb2327be9 | ||
|
|
9bf156adf2 | ||
|
|
9f0fcd6b10 | ||
|
|
ea291fb157 | ||
|
|
fce4ae7bb0 | ||
|
|
ef44f047aa | ||
|
|
ae002f7369 | ||
|
|
e4cd3f9e18 | ||
|
|
916e0b6acc | ||
|
|
2350a5e9eb | ||
|
|
001c686a19 | ||
|
|
65d36906c7 | ||
|
|
327f104c53 | ||
|
|
4f01b485df | ||
|
|
fc7f3fddfe | ||
|
|
937651e509 | ||
|
|
da9fc85862 | ||
|
|
996553be3d | ||
|
|
ff6715b8b1 | ||
|
|
16d9a2c311 | ||
|
|
c652a269ca | ||
|
|
75bb2016a9 | ||
|
|
60d5a6cb77 | ||
|
|
3ee5f840ce | ||
|
|
abbad56994 | ||
|
|
54acbdd395 | ||
|
|
76d4f1942b | ||
|
|
939310203d | ||
|
|
abdcf82d70 | ||
|
|
ad93d526b7 | ||
|
|
87c1a5cb77 | ||
|
|
979f08b3eb | ||
|
|
89ad250940 | ||
|
|
1ed3b48c2e | ||
|
|
fbd9189e7b | ||
|
|
1dd889cb16 | ||
|
|
2e8fbd661a | ||
|
|
6e424dba6e | ||
|
|
d5dec97d23 | ||
|
|
2d401bcb74 | ||
|
|
4f383f4b19 | ||
|
|
7f35ecb8f5 | ||
|
|
60a04e4e4f | ||
|
|
8d95a6e5ca | ||
|
|
10cb466fb2 | ||
|
|
5be9de7e95 | ||
|
|
9e261aeb36 | ||
|
|
3fe3c00c72 | ||
|
|
a13a4f4d8b | ||
|
|
1d925e5b7b | ||
|
|
e779a5d47e | ||
|
|
7cd35f97f7 | ||
|
|
f69b6204df | ||
|
|
8baa46babd | ||
|
|
bf9b6f4d83 | ||
|
|
494fc00d34 | ||
|
|
4dd3fcf811 | ||
|
|
33d7cd11a2 | ||
|
|
fbd80e7493 | ||
|
|
a4fb0daa8e | ||
|
|
0b36f2adaa | ||
|
|
0288a5f146 | ||
|
|
a1efd7a514 | ||
|
|
e0c69fb83d | ||
|
|
43e88dd04f | ||
|
|
946d4dde84 | ||
|
|
e890e9e6a5 | ||
|
|
7930c675ea | ||
|
|
298b14c82d | ||
|
|
abb68a80e6 | ||
|
|
aec0b75047 | ||
|
|
f4e7c56119 | ||
|
|
eb70410a54 | ||
|
|
9bccf40eb2 | ||
|
|
3cb7465ab7 | ||
|
|
8e78471fdf | ||
|
|
c0f8501950 | ||
|
|
c31510289f | ||
|
|
9b2bc9e57a | ||
|
|
76d2f8fea9 | ||
|
|
63f746b8dc | ||
|
|
18d64b8b93 | ||
|
|
a740225601 | ||
|
|
0d5fc47a73 | ||
|
|
89974e87b6 | ||
|
|
ec071ad4ab | ||
|
|
22c873f097 | ||
|
|
e57ffb8128 | ||
|
|
2d7c578033 | ||
|
|
b503939955 | ||
|
|
8a4a826248 | ||
|
|
8011c106ae | ||
|
|
11d0d6a7be | ||
|
|
2cc4bd7285 | ||
|
|
7ff38189fb | ||
|
|
dc243c6f7c | ||
|
|
b3508002e1 | ||
|
|
06177ea337 | ||
|
|
526d86489a | ||
|
|
e8872fa31f | ||
|
|
d547dfc6bf | ||
|
|
b68a35d611 | ||
|
|
d2e381183e | ||
|
|
d4c37a7c1b | ||
|
|
dee64c36e5 | ||
|
|
846147efc0 | ||
|
|
b621c9c4a2 | ||
|
|
5b1295c955 | ||
|
|
43609b5c35 | ||
|
|
64b7fbdeb2 | ||
|
|
44ed485cf1 | ||
|
|
90d1ed5ae3 | ||
|
|
d7614eeb7e | ||
|
|
c441299f2b | ||
|
|
75526f58cc | ||
|
|
52b8703b78 | ||
|
|
9e4d9799dc | ||
|
|
69e7820d01 | ||
|
|
dbedeecece | ||
|
|
ac4e558164 | ||
|
|
691cd8a687 | ||
|
|
e8847b80a2 | ||
|
|
cdc327a511 | ||
|
|
422eb8710f | ||
|
|
42c47e9225 | ||
|
|
f3dc010bda | ||
|
|
fa34b2dc63 | ||
|
|
23b4cc62e9 | ||
|
|
08c332c1b0 | ||
|
|
b2ad517e03 | ||
|
|
812a907abe | ||
|
|
3c73c50b29 | ||
|
|
0d2e8f93e6 | ||
|
|
1fd8301d38 | ||
|
|
a013a150c1 | ||
|
|
0c9d63ba7f | ||
|
|
a367bcc30d | ||
|
|
d1db3325f2 | ||
|
|
0a8b4ad9af | ||
|
|
2283ea838a | ||
|
|
0fe2aece0b | ||
|
|
dcc3ba856e | ||
|
|
6a4d7db35b | ||
|
|
e1567e2729 | ||
|
|
2ede430196 | ||
|
|
1e7b7403ff | ||
|
|
e33690db45 | ||
|
|
cece1b642b | ||
|
|
daac9192cc | ||
|
|
4699d9c9e1 | ||
|
|
24abcaee7a | ||
|
|
14d59df037 | ||
|
|
5d23e37b83 | ||
|
|
7f5b68dfbd | ||
|
|
ac0454f07f | ||
|
|
9e727d568c | ||
|
|
7fd9af27a5 | ||
|
|
77646c87dd | ||
|
|
8531a43aac | ||
|
|
2b7f4ca792 | ||
|
|
8e41393e14 | ||
|
|
452531e22f | ||
|
|
6b6e5bf4b8 | ||
|
|
274bd5b670 | ||
|
|
b8f3571ba1 | ||
|
|
7f8e9680a6 | ||
|
|
2bebdf7595 | ||
|
|
759dacf996 | ||
|
|
2e76b94e17 | ||
|
|
128b7a092b | ||
|
|
491c558a57 | ||
|
|
45bf80a62e | ||
|
|
fdc885ecd2 | ||
|
|
e9b4630d58 | ||
|
|
5d8442c21c | ||
|
|
47c9ad2e34 | ||
|
|
b31b0e04bd | ||
|
|
838206e6a9 | ||
|
|
c681a74f87 | ||
|
|
b45138e6d7 | ||
|
|
f692d94d08 | ||
|
|
6a0e1a7a89 | ||
|
|
ec8cd31f32 | ||
|
|
ec1ba64dac | ||
|
|
a9590b900a | ||
|
|
e7f2083f0b | ||
|
|
1b93160f5d | ||
|
|
74476c8f89 | ||
|
|
934958771c | ||
|
|
0f827820f6 | ||
|
|
709a8ca7e4 | ||
|
|
fea9d2c4ce | ||
|
|
ea9686bdc0 | ||
|
|
9f03879386 | ||
|
|
43b26e7a5b | ||
|
|
3a1fb995a4 | ||
|
|
87cb7170b2 | ||
|
|
3165f09e0d | ||
|
|
03910bbe86 | ||
|
|
9532220814 | ||
|
|
f5decb7c9e | ||
|
|
4e00bfb158 | ||
|
|
7b79732a28 | ||
|
|
bdf8f6d21b | ||
|
|
cbc63ad344 | ||
|
|
844b655c76 | ||
|
|
db6c8f5a9a | ||
|
|
06331492e5 | ||
|
|
dabb5652fe | ||
|
|
4947faca83 | ||
|
|
9d1cb51acc | ||
|
|
1ff1f5777f | ||
|
|
7bc13bf237 | ||
|
|
f65a0f093b | ||
|
|
e7058f6aca | ||
|
|
785afe66cd | ||
|
|
af834012d0 | ||
|
|
de3f769f49 | ||
|
|
ed862050b2 | ||
|
|
3c6c1eb634 | ||
|
|
22851a9463 | ||
|
|
38df8156b9 | ||
|
|
542467fd6a | ||
|
|
5986542e3d | ||
|
|
5163313285 | ||
|
|
2201f3354a | ||
|
|
b639394d56 | ||
|
|
12c2d71bd2 | ||
|
|
c6f7e9f635 | ||
|
|
7151713d9c | ||
|
|
a5379de077 | ||
|
|
85d24f1354 | ||
|
|
b60a76f0db | ||
|
|
e0dc7659fb | ||
|
|
b7b4268138 | ||
|
|
e89a61399c | ||
|
|
6c90686880 | ||
|
|
331a66c027 | ||
|
|
6d509fa5c3 | ||
|
|
b62cf39359 | ||
|
|
49f44e65c0 | ||
|
|
32697fea42 | ||
|
|
c80a091f56 | ||
|
|
bbc6708976 | ||
|
|
6a01f6551a | ||
|
|
0c663a8aa2 | ||
|
|
be224ed94a | ||
|
|
c09da71078 | ||
|
|
cf5f0126ce | ||
|
|
a23e8907d7 | ||
|
|
4ca2805303 | ||
|
|
1b775faa43 | ||
|
|
0edefaeae5 | ||
|
|
a12132ccb6 | ||
|
|
9f044d89b5 | ||
|
|
191e3df84b | ||
|
|
6b6f8afdac | ||
|
|
25e333dbf4 | ||
|
|
856d13e9ff | ||
|
|
eb8f7f433c | ||
|
|
6a693a818f | ||
|
|
fa7d81095a | ||
|
|
16260082f6 | ||
|
|
7763785ed7 | ||
|
|
2baa889917 | ||
|
|
e60f43fff3 | ||
|
|
2ed1a9eaad | ||
|
|
1545f03a94 | ||
|
|
5e51751064 | ||
|
|
d87bc4d22c | ||
|
|
29dd96acf3 | ||
|
|
59a5c9fc79 | ||
|
|
c389a3c434 | ||
|
|
ec96a85f86 | ||
|
|
e30f5b9c96 | ||
|
|
f5b03af9d6 | ||
|
|
80c7823ac7 | ||
|
|
a443f003bb | ||
|
|
f6979648e8 | ||
|
|
2a4decc635 | ||
|
|
b9d19d3bb3 | ||
|
|
d8da041edf | ||
|
|
dd99371cda | ||
|
|
90aa6fc544 | ||
|
|
7b4fcc3e52 | ||
|
|
81b6b7eeb2 | ||
|
|
4b5974f600 | ||
|
|
a6926f4ce6 | ||
|
|
b6e972af79 | ||
|
|
e76ec19775 | ||
|
|
d797322fea | ||
|
|
5a5e34a744 | ||
|
|
93db7052ff | ||
|
|
f7170af7bd | ||
|
|
b78eef3eaa | ||
|
|
d5decea85c | ||
|
|
dea3ae3317 | ||
|
|
33c1e50235 | ||
|
|
5718ad1b53 | ||
|
|
4f3671e253 | ||
|
|
4e08bb1b66 | ||
|
|
69c5016b63 | ||
|
|
ce1bf58dc0 | ||
|
|
5eb00c9ee6 | ||
|
|
1f3b6b95aa | ||
|
|
118db41049 | ||
|
|
8390c3e50b | ||
|
|
168b5179e6 | ||
|
|
7697f6d400 | ||
|
|
235ebce5d5 | ||
|
|
e8a09d6499 | ||
|
|
edc67827d8 | ||
|
|
9e5cdef2ef | ||
|
|
c2f4a5e248 | ||
|
|
72d811b289 | ||
|
|
43731aa990 | ||
|
|
45f59fff3a | ||
|
|
58a4cfa132 | ||
|
|
333dd3f2fd | ||
|
|
c4f7dd77b1 | ||
|
|
f442b83573 | ||
|
|
768aaae25d | ||
|
|
eab997c557 | ||
|
|
9d73dc487d | ||
|
|
2575ac61ba | ||
|
|
6130144da1 | ||
|
|
68db31da44 | ||
|
|
1730b05078 | ||
|
|
1acbce733c | ||
|
|
b44316049b | ||
|
|
2e133e8ecb | ||
|
|
dfb2f4d7f2 | ||
|
|
10e9e4215f | ||
|
|
8125a211d3 | ||
|
|
818b8db433 | ||
|
|
8a522f5e7d | ||
|
|
ad4626edfc | ||
|
|
8d7e8933cf | ||
|
|
3ad21a409f | ||
|
|
fcbd105b82 | ||
|
|
b82dcf1387 | ||
|
|
b16b550150 | ||
|
|
d3471aef59 | ||
|
|
822555df0b | ||
|
|
102dc8bd02 | ||
|
|
e306ba0c85 | ||
|
|
4626d65ac1 | ||
|
|
38a80ea0e4 | ||
|
|
4b88ad2b0a | ||
|
|
590f954d6f | ||
|
|
bc5fc2b0f3 | ||
|
|
d0fb4e342e | ||
|
|
b53d0529db | ||
|
|
8c7988b525 | ||
|
|
dfffe4b5e8 | ||
|
|
538aa11904 | ||
|
|
6fa978af9a | ||
|
|
3f81af72f6 | ||
|
|
97f1cf08fb | ||
|
|
e047cec18a | ||
|
|
bdcf59d109 | ||
|
|
d3f1379dc8 | ||
|
|
b876d32452 | ||
|
|
50a6be3d58 | ||
|
|
5251db2278 | ||
|
|
92fca7cf01 | ||
|
|
e22f5bc048 | ||
|
|
f971d1e0bb | ||
|
|
752917acaa | ||
|
|
ea6fb52698 | ||
|
|
07a87e369c | ||
|
|
56c46e6da5 | ||
|
|
2db4ca1300 | ||
|
|
53bc415268 | ||
|
|
be537728df | ||
|
|
4e5b98b10f | ||
|
|
d4acd906bf | ||
|
|
d751ce66a3 | ||
|
|
3f0abd4dfd | ||
|
|
44a423d804 | ||
|
|
3e61e0490e | ||
|
|
def4919592 | ||
|
|
2d147d70e0 | ||
|
|
e29e64dffe | ||
|
|
a7ec259bd5 | ||
|
|
3e93e19767 | ||
|
|
532b065596 | ||
|
|
82c1e2315b | ||
|
|
8cc9eec535 | ||
|
|
dece65be31 | ||
|
|
3e6d29b3dd | ||
|
|
487135b497 | ||
|
|
91f648aa95 | ||
|
|
999931ded2 | ||
|
|
15dbcae725 | ||
|
|
01efb623da | ||
|
|
f854c5262d | ||
|
|
a91b754aaa | ||
|
|
b1623ff3d4 | ||
|
|
c2426ca45a | ||
|
|
276f419a3d | ||
|
|
c91b8bea01 | ||
|
|
7bdceca6ce | ||
|
|
6f9a263435 | ||
|
|
28a7865ed1 | ||
|
|
6e7335ac52 | ||
|
|
8115383dec | ||
|
|
3d1b017a60 | ||
|
|
bf14e5b018 | ||
|
|
fb3517453f | ||
|
|
bfca6beb28 | ||
|
|
f51e46d3d8 | ||
|
|
78a60cc1d9 | ||
|
|
935d3a9e42 | ||
|
|
35866f8485 | ||
|
|
6b4b644355 | ||
|
|
b9ec58e7a1 | ||
|
|
4644aed322 | ||
|
|
80da896859 | ||
|
|
b96dcb4401 | ||
|
|
5054f1784d | ||
|
|
788c0efda0 | ||
|
|
4d49d42702 | ||
|
|
f5192230e0 | ||
|
|
400e3eca7d | ||
|
|
b90c8d80fe | ||
|
|
a0491f6bfc | ||
|
|
a8df54cf5d | ||
|
|
9c4e43ee12 | ||
|
|
7a1887c525 | ||
|
|
907783f9ca | ||
|
|
f4f68fa021 | ||
|
|
b76e9e80a7 | ||
|
|
b8f677b2fe | ||
|
|
6e42fbae4d | ||
|
|
b8c0008061 | ||
|
|
cdce090c2a | ||
|
|
7994a3df8b | ||
|
|
e246c0852b | ||
|
|
5bb0c458cd | ||
|
|
c5b2f0945a | ||
|
|
1b0425bfe9 | ||
|
|
4db86286ee | ||
|
|
9308946715 | ||
|
|
d28eca6b7f | ||
|
|
8e26105232 | ||
|
|
ab52f334e2 | ||
|
|
c674f9f7ad | ||
|
|
537d30120a | ||
|
|
519267e1cb | ||
|
|
2c495fb70d | ||
|
|
401d1aec7b | ||
|
|
8299b1c036 | ||
|
|
9dd1e4dbdb | ||
|
|
47a3534eff | ||
|
|
2f39ff66f3 | ||
|
|
ddca183704 | ||
|
|
078ce6130c | ||
|
|
6d15c2a156 | ||
|
|
7d705c0677 | ||
|
|
c027328b91 | ||
|
|
65cb67e1c1 | ||
|
|
494f27c14c | ||
|
|
3c02b72084 | ||
|
|
75e2be35ba | ||
|
|
b0f9cbfd26 | ||
|
|
6afea18cde | ||
|
|
2e69ff4b97 | ||
|
|
6c70fe9334 | ||
|
|
4ccbd4581e | ||
|
|
6e262f6c3f | ||
|
|
52e10475a5 | ||
|
|
c0299a5a4b | ||
|
|
06eecb0dce | ||
|
|
96261a7742 | ||
|
|
d7c479fa1e | ||
|
|
f8c494e59c | ||
|
|
f6d304864b | ||
|
|
3593b4cd60 | ||
|
|
ef557b3fc1 | ||
|
|
9a94a4b7b8 | ||
|
|
e18518d731 | ||
|
|
e49bf21914 | ||
|
|
8b01d8f13b | ||
|
|
710da275c8 | ||
|
|
fd481eb725 | ||
|
|
b5bbdbbed5 | ||
|
|
0f78d8aa5c | ||
|
|
6a26200314 | ||
|
|
3f98aa1cfb | ||
|
|
a485121526 | ||
|
|
1f9e1cf175 | ||
|
|
ec402882da | ||
|
|
e7633e0e2c | ||
|
|
30aeb465b7 | ||
|
|
ff4993fc51 | ||
|
|
0a42ea8021 | ||
|
|
b8d024b59b | ||
|
|
9e1ccf4543 | ||
|
|
feecd75ff3 | ||
|
|
248bdcc149 | ||
|
|
e4e354834d | ||
|
|
d64a6d6255 | ||
|
|
510387a605 | ||
|
|
e99b2a8410 | ||
|
|
6608111315 | ||
|
|
075ebb255d | ||
|
|
3eb6a5b3b2 | ||
|
|
8ba1f17f72 | ||
|
|
e5f5a79e66 | ||
|
|
43f1b19767 | ||
|
|
7bebe4528f | ||
|
|
da63657cdd | ||
|
|
2b1d271888 | ||
|
|
47fb8a4fda | ||
|
|
ee7d9726df | ||
|
|
44b560a916 | ||
|
|
5657f6ebe8 | ||
|
|
19543b6b16 | ||
|
|
94a832a0c6 | ||
|
|
b56e994ecd | ||
|
|
8be11cdfdb | ||
|
|
d71a9602b5 | ||
|
|
1108bb7e85 | ||
|
|
ae8e5aa88d | ||
|
|
17f4acf6b1 | ||
|
|
2ce3f3037c | ||
|
|
29189a6d4a | ||
|
|
08f3c86b8a | ||
|
|
c6eb171b5b | ||
|
|
d26695cd2a | ||
|
|
01ab390b06 | ||
|
|
43c42295d3 | ||
|
|
52bc915120 | ||
|
|
cd9cabb955 | ||
|
|
b7253275fc | ||
|
|
e66a61c198 | ||
|
|
f8b3c78b19 | ||
|
|
4749746171 | ||
|
|
1ddd01c2a0 | ||
|
|
87ec3850b5 | ||
|
|
1b25a61c9e | ||
|
|
7bee8e8161 | ||
|
|
a545ff8264 | ||
|
|
5352234aef | ||
|
|
c3732f9d86 | ||
|
|
b95f3809fe | ||
|
|
e6a28b7753 | ||
|
|
62adea8b46 | ||
|
|
7d11db33c0 | ||
|
|
11fce4235b | ||
|
|
f500b4875f | ||
|
|
ba212c583e | ||
|
|
fd341e07da | ||
|
|
d59e2a229c | ||
|
|
5808fc6966 | ||
|
|
b40bf6a64d | ||
|
|
7a73e97922 | ||
|
|
3a97122e34 | ||
|
|
2f89a16314 | ||
|
|
0cd8c2e273 | ||
|
|
b4992673b2 | ||
|
|
46dce17970 | ||
|
|
f080627cba | ||
|
|
85fb20a1d1 | ||
|
|
428d203eac | ||
|
|
c10ca25f62 | ||
|
|
2e8f6f9c28 | ||
|
|
11e4c46f25 | ||
|
|
ff8d8752c7 | ||
|
|
e3cfc28718 |
@@ -25,7 +25,7 @@ runs:
|
||||
steps:
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
|
||||
- uses: actions/cache@v4
|
||||
- uses: actions/cache@v5
|
||||
if: ${{env.DEBUG == 'true'}}
|
||||
id: debug
|
||||
with:
|
||||
|
||||
@@ -36,7 +36,7 @@ runs:
|
||||
steps:
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
|
||||
- uses: actions/cache@v4
|
||||
- uses: actions/cache@v5
|
||||
if: ${{env.DEBUG == 'true' && inputs.cache-skip != 'true'}}
|
||||
id: debug
|
||||
with:
|
||||
|
||||
@@ -23,7 +23,7 @@ inputs:
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
- uses: actions/cache/restore@v4 # Cache for LLVM libcxx
|
||||
- uses: actions/cache/restore@v5 # Cache for LLVM libcxx
|
||||
with:
|
||||
path: ${{env.LLVM_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
@@ -32,14 +32,14 @@ runs:
|
||||
- uses: ./.github/actions/sanitize/mpi
|
||||
if: ${{inputs.par == 'true'}}
|
||||
|
||||
- uses: actions/cache/restore@v4 # Cache for Hypre
|
||||
- uses: actions/cache/restore@v5 # Cache for Hypre
|
||||
if: ${{inputs.par == 'true'}}
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
|
||||
- uses: actions/cache/restore@v4 # Cache for Metis
|
||||
- uses: actions/cache/restore@v5 # Cache for Metis
|
||||
if: ${{inputs.par == 'true'}}
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
@@ -51,13 +51,13 @@ runs:
|
||||
run: ln -s -f ${{env.HYPRE_DIR}} hypre && ln -s -f ${{env.METIS_DIR}} metis-4.0
|
||||
shell: bash
|
||||
|
||||
- uses: actions/cache/restore@v4 # Cache for LSAN suppression file
|
||||
- uses: actions/cache/restore@v5 # Cache for LSAN suppression file
|
||||
with:
|
||||
path: ${{env.LSAN_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: build-lsan-suppression-file
|
||||
|
||||
- uses: actions/checkout@v4 # Checkout the repository
|
||||
- uses: actions/checkout@v6 # Checkout the repository
|
||||
with:
|
||||
path: mfem
|
||||
# ref: ${{env.BRANCH}}
|
||||
|
||||
@@ -43,7 +43,7 @@ jobs:
|
||||
remove-docker-images: 'true'
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
|
||||
# It's easier to reference named variables than indexes of the matrix
|
||||
- name: Set Environment
|
||||
|
||||
@@ -153,7 +153,7 @@ jobs:
|
||||
# /home/runner/work/mfem/mfem/mfem
|
||||
# Note: Done now to access "install-hypre" and "install-metis" actions.
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
path: ${{ env.MFEM_TOP_DIR }}
|
||||
# Fetch the complete history for codecov to access commits ID
|
||||
@@ -225,7 +225,7 @@ jobs:
|
||||
- name: cache hypre
|
||||
id: hypre-cache
|
||||
if: matrix.mpi == 'par'
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
|
||||
@@ -255,7 +255,7 @@ jobs:
|
||||
- name: cache metis
|
||||
id: metis-cache
|
||||
if: matrix.mpi == 'par' && matrix.os != 'windows-latest'
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
@@ -270,7 +270,7 @@ jobs:
|
||||
- name: cache vcpkg (Windows)
|
||||
id: vcpkg-cache
|
||||
if: matrix.os == 'windows-latest'
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: vcpkg_cache
|
||||
key: ${{ runner.os }}-${{ matrix.mpi }}-vcpkg-v1
|
||||
@@ -295,7 +295,8 @@ jobs:
|
||||
export HOMEBREW_NO_INSTALL_CLEANUP=1
|
||||
brew update
|
||||
brew install enzyme
|
||||
ENZYME_LLVM=$(brew info enzyme | sed -n 's/^Required:.*\(llvm[^ ]*\).*/\1/p')
|
||||
ENZYME_LLVM=$(brew info enzyme | sed -n 's/^Required.*:.*\(llvm[^ ]*\).*/\1/p')
|
||||
echo "ENZYME_LLVM=$ENZYME_LLVM"
|
||||
LLVM_PREFIX=$(brew --prefix $ENZYME_LLVM)
|
||||
echo "LLVM_PREFIX=$LLVM_PREFIX" >> $GITHUB_ENV
|
||||
echo "OMPI_CC=$LLVM_PREFIX/bin/clang" >> $GITHUB_ENV
|
||||
|
||||
@@ -40,11 +40,11 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v2
|
||||
uses: github/codeql-action/init@v4
|
||||
with:
|
||||
languages: ${{ matrix.language }}
|
||||
# If you wish to specify custom queries, you can do so here or in a config file.
|
||||
@@ -57,7 +57,7 @@ jobs:
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below)
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v2
|
||||
uses: github/codeql-action/autobuild@v4
|
||||
|
||||
# ℹ️ Command-line programs to run using the OS shell.
|
||||
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
|
||||
@@ -70,4 +70,4 @@ jobs:
|
||||
# ./location_of_script_within_repo/buildscript.sh
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v2
|
||||
uses: github/codeql-action/analyze@v4
|
||||
|
||||
@@ -39,7 +39,7 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: checkout MFEM
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
path: mfem
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
|
||||
- name: Cache Hypre Install
|
||||
id: hypre-cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
|
||||
@@ -65,7 +65,7 @@ jobs:
|
||||
|
||||
- name: Cache Metis Install
|
||||
id: metis-cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
|
||||
@@ -38,7 +38,7 @@ jobs:
|
||||
github.event.pull_request.head.repo.full_name != github.repository)
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: copyright check
|
||||
id: copyright
|
||||
@@ -93,7 +93,7 @@ jobs:
|
||||
github.event.pull_request.head.repo.full_name != github.repository)
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: get astyle
|
||||
run: |
|
||||
@@ -110,7 +110,7 @@ jobs:
|
||||
github.event.pull_request.head.repo.full_name != github.repository)
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: get doxygen and graphviz
|
||||
run: |
|
||||
@@ -135,7 +135,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
|
||||
@@ -17,11 +17,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
name: 2.19.0
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
|
||||
@@ -27,13 +27,13 @@ jobs:
|
||||
llvm_use_sanitizer: "Undefined"
|
||||
name: ${{matrix.sanitizer}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.LLVM_DIR}}
|
||||
key: build-libcxx-${{env.LLVM_VER}}-${{matrix.sanitizer}}
|
||||
|
||||
@@ -17,11 +17,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
name: lsan.supp
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.LSAN_DIR}}
|
||||
key: build-lsan-suppression-file
|
||||
|
||||
@@ -17,11 +17,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
name: 4.0.3
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
|
||||
|
||||
@@ -28,7 +28,7 @@ jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/mfem
|
||||
with:
|
||||
par: ${{inputs.par}}
|
||||
@@ -40,7 +40,7 @@ jobs:
|
||||
env:
|
||||
ex: ${{inputs.par && 'ex1p' || 'ex1'}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
@@ -58,7 +58,7 @@ jobs:
|
||||
env:
|
||||
exclude: ${{inputs.par && '-E "_ser"' || ''}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
@@ -82,7 +82,7 @@ jobs:
|
||||
env:
|
||||
exclude: ${{inputs.par && '-E "_ser"' || ''}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
@@ -107,7 +107,7 @@ jobs:
|
||||
run: ${{inputs.par && '-R "_cpu_np"' || ''}}
|
||||
exclude: ${{inputs.par && '"unit_tests|debug"' || '"^unit_tests$|debug"'}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
@@ -131,7 +131,7 @@ jobs:
|
||||
env:
|
||||
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
@@ -165,7 +165,7 @@ jobs:
|
||||
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
|
||||
np: ${{inputs.par && '_np=2' || ''}}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/restore
|
||||
id: restore
|
||||
with:
|
||||
|
||||
+6
-2
@@ -443,12 +443,16 @@ miniapps/diag-smoothers/mg-abs-l1-jacobi
|
||||
miniapps/contact/contact
|
||||
miniapps/contact/ParaView
|
||||
|
||||
miniapps/plasma/pic/electrostatic-*
|
||||
!miniapps/plasma/pic/electrostatic-*.cpp
|
||||
miniapps/plasma/pic/*.csv
|
||||
|
||||
# Unit test binary and outputs
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
tests/unit/punit_tests
|
||||
tests/unit/gpu_unit_tests
|
||||
tests/unit/pgpu_unit_tests
|
||||
tests/unit/cunit_tests
|
||||
tests/unit/pcunit_tests
|
||||
tests/unit/sedov_tests_*
|
||||
tests/unit/psedov_tests_*
|
||||
tests/unit/tmop_pa_tests_*
|
||||
|
||||
@@ -8,6 +8,22 @@
|
||||
https://mfem.org
|
||||
|
||||
|
||||
Version 4.10 (development)
|
||||
==========================
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Replaced legacy simplex quadrature rules with symmetric positive-weight
|
||||
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
|
||||
rules guarantee all-positive weights and interior quadrature points,
|
||||
improving numerical stability. Higher orders fall back to Grundmann-Moller.
|
||||
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
|
||||
2015.
|
||||
Tet rules (d=1-13): Witherden & Vincent (ibid).
|
||||
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
|
||||
2022.
|
||||
|
||||
|
||||
Version 4.9.1 (development)
|
||||
===========================
|
||||
|
||||
|
||||
@@ -592,6 +592,13 @@ if (MFEM_USE_ENZYME)
|
||||
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_PROTEUS)
|
||||
enable_language(C)
|
||||
find_package(proteus REQUIRED PATHS "${PROTEUS_DIR}")
|
||||
message(STATUS "${PROTEUS_DIR}/include")
|
||||
include_directories("${PROTEUS_DIR}/include")
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
if (APPLE)
|
||||
@@ -728,6 +735,16 @@ mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
target_compile_features(mfem PUBLIC cxx_std_${CMAKE_CXX_STANDARD})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES} ${TPL_TARGETS})
|
||||
|
||||
if (MFEM_USE_PROTEUS)
|
||||
add_library(ClangProteusFlags INTERFACE IMPORTED)
|
||||
set_target_properties(ClangProteusFlags PROPERTIES
|
||||
INTERFACE_COMPILE_OPTIONS "-fpass-plugin=$<TARGET_FILE:ProteusPass>"
|
||||
)
|
||||
target_link_libraries(mfem PUBLIC ClangProteusFlags)
|
||||
target_link_libraries(mfem PUBLIC proteus)
|
||||
endif()
|
||||
|
||||
if (TPL_TARGETS)
|
||||
add_dependencies(mfem ${TPL_TARGETS})
|
||||
endif()
|
||||
|
||||
@@ -109,6 +109,10 @@ if (MFEM_USE_RAJA)
|
||||
find_dependency(RAJA)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_dependency(umpire)
|
||||
endif()
|
||||
|
||||
if (NOT TARGET mfem)
|
||||
include(${CMAKE_CURRENT_LIST_DIR}/MFEMTargets.cmake)
|
||||
endif (NOT TARGET mfem)
|
||||
|
||||
@@ -14,12 +14,12 @@
|
||||
# - UMPIRE_LIBRARIES
|
||||
# - UMPIRE_INCLUDE_DIRS
|
||||
|
||||
if (NOT umpire_DIR AND UMPIRE_DIR)
|
||||
set(umpire_DIR ${UMPIRE_DIR}/lib/cmake/umpire)
|
||||
if (NOT umpire_ROOT AND UMPIRE_DIR)
|
||||
set(umpire_ROOT ${UMPIRE_DIR})
|
||||
endif()
|
||||
message(STATUS "Looking for UMPIRE ...")
|
||||
message(STATUS " in UMPIRE_DIR = ${UMPIRE_DIR}")
|
||||
message(STATUS " umpire_DIR = ${umpire_DIR}")
|
||||
message(STATUS " umpire_ROOT = ${umpire_ROOT}")
|
||||
find_package(umpire CONFIG)
|
||||
set(UMPIRE_FOUND ${umpire_FOUND})
|
||||
set(UMPIRE_LIBRARIES "umpire")
|
||||
|
||||
@@ -157,4 +157,22 @@ constexpr real_t operator""_r(unsigned long long v)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
#ifdef NVTX_DBG_HPP
|
||||
#include NVTX_DBG_HPP
|
||||
#else
|
||||
#define db1(...)
|
||||
#define dbg(...)
|
||||
#define dbl(...)
|
||||
#define dba(...)
|
||||
#define dbc(...)
|
||||
#define NVTX_MARK_FUNCTION
|
||||
#define NVTX_MARK_BEGIN(...)
|
||||
#define NVTX_INI(...)
|
||||
#define NVTX_END(...)
|
||||
#define NVTX_MARK_INI(...)
|
||||
#define NVTX_MARK_END(...)
|
||||
#define NVTX_MARK(...)
|
||||
#define NVTX(...)
|
||||
#endif
|
||||
|
||||
#endif // MFEM_CONFIG_HPP
|
||||
|
||||
@@ -47,6 +47,7 @@ list(APPEND ALL_EXE_SRCS
|
||||
ex39.cpp
|
||||
ex40.cpp
|
||||
ex41.cpp
|
||||
jitplayground.cpp
|
||||
)
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
|
||||
+2
-2
@@ -5,9 +5,9 @@
|
||||
// Sample runs:
|
||||
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh
|
||||
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh
|
||||
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 462 -n 10 -o 2 -elast
|
||||
// mpirun -np 4 ex12p -m ../data/beam-tet.mesh -s 464 -n 10 -o 2 -elast
|
||||
// mpirun -np 4 ex12p -m ../data/beam-hex.mesh -s 3878
|
||||
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 81
|
||||
// mpirun -np 4 ex12p -m ../data/beam-wedge.mesh -s 82
|
||||
// mpirun -np 4 ex12p -m ../data/beam-tri.mesh -s 3877 -o 2 -sys
|
||||
// mpirun -np 4 ex12p -m ../data/beam-quad.mesh -s 4544 -n 6 -o 3 -elast
|
||||
// mpirun -np 4 ex12p -m ../data/beam-quad-nurbs.mesh
|
||||
|
||||
+27
-9
@@ -302,15 +302,21 @@ int main(int argc, char *argv[])
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock_r(vishost, visport);
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_r.precision(8);
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
|
||||
<< "window_title 'Exact: Real Part'" << flush;
|
||||
// Make sure all ranks have sent their real solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
|
||||
<< "window_title 'Exact: Imaginary Part'" << flush;
|
||||
// Make sure all ranks have sent their imaginary solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
}
|
||||
|
||||
// 11. Set up the parallel sesquilinear form a(.,.) on the finite element
|
||||
@@ -534,15 +540,21 @@ int main(int argc, char *argv[])
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock_r(vishost, visport);
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_r.precision(8);
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_r << "solution\n" << *pmesh << u.real()
|
||||
<< "window_title 'Solution: Real Part'" << flush;
|
||||
// Make sure all ranks have sent their real solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_i << "solution\n" << *pmesh << u.imag()
|
||||
<< "window_title 'Solution: Imaginary Part'" << flush;
|
||||
// Make sure all ranks have sent their imaginary solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
}
|
||||
if (visualization && exact_sol)
|
||||
{
|
||||
@@ -551,15 +563,21 @@ int main(int argc, char *argv[])
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock_r(vishost, visport);
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_r.precision(8);
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_r << "solution\n" << *pmesh << u_exact->real()
|
||||
<< "window_title 'Error: Real Part'" << flush;
|
||||
// Make sure all ranks have sent their real solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_i << "solution\n" << *pmesh << u_exact->imag()
|
||||
<< "window_title 'Error: Imaginary Part'" << flush;
|
||||
// Make sure all ranks have sent their imaginary solution before initiating
|
||||
// another set of GLVis connections (one from each rank):
|
||||
MPI_Barrier(pmesh->GetComm());
|
||||
}
|
||||
if (visualization)
|
||||
{
|
||||
|
||||
+11
-52
@@ -5,8 +5,8 @@
|
||||
// Sample runs:
|
||||
// ex37 -alpha 10
|
||||
// ex37 -alpha 10 -pv
|
||||
// ex37 -lambda 0.1 -mu 0.1
|
||||
// ex37 -o 2 -alpha 5.0 -mi 50 -vf 0.4 -ntol 1e-5
|
||||
// ex37 -lambda 0.1 -mu 0.1 -growth 1
|
||||
// ex37 -o 2 -alpha 10.0 -mi 50 -vf 0.4 -ntol 1e-5 -growth 1.5
|
||||
// ex37 -r 6 -o 1 -alpha 25.0 -epsilon 0.02 -mi 50 -ntol 1e-5
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to solve a
|
||||
@@ -55,53 +55,6 @@
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
/**
|
||||
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
|
||||
* ∫_Ω ρ dx = θ vol(Ω) as follows:
|
||||
*
|
||||
* 1. Compute the root of the R → R function
|
||||
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
|
||||
* 2. Set ψ ← ψ + c.
|
||||
*
|
||||
* @param psi a GridFunction to be updated
|
||||
* @param target_volume θ vol(Ω)
|
||||
* @param tol Newton iteration tolerance
|
||||
* @param max_its Newton maximum iteration number
|
||||
* @return real_t Final volume, ∫_Ω sigmoid(ψ)
|
||||
*/
|
||||
real_t proj(GridFunction &psi, real_t target_volume, real_t tol=1e-12,
|
||||
int max_its=10)
|
||||
{
|
||||
MappedGridFunctionCoefficient sigmoid_psi(&psi, sigmoid);
|
||||
MappedGridFunctionCoefficient der_sigmoid_psi(&psi, der_sigmoid);
|
||||
|
||||
LinearForm int_sigmoid_psi(psi.FESpace());
|
||||
int_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
|
||||
LinearForm int_der_sigmoid_psi(psi.FESpace());
|
||||
int_der_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(
|
||||
der_sigmoid_psi));
|
||||
bool done = false;
|
||||
for (int k=0; k<max_its; k++) // Newton iteration
|
||||
{
|
||||
int_sigmoid_psi.Assemble(); // Recompute f(c) with updated ψ
|
||||
const real_t f = int_sigmoid_psi.Sum() - target_volume;
|
||||
|
||||
int_der_sigmoid_psi.Assemble(); // Recompute df(c) with updated ψ
|
||||
const real_t df = int_der_sigmoid_psi.Sum();
|
||||
|
||||
const real_t dc = -f/df;
|
||||
psi += dc;
|
||||
if (abs(dc) < tol) { done = true; break; }
|
||||
}
|
||||
if (!done)
|
||||
{
|
||||
mfem_warning("Projection reached maximum iteration without converging. "
|
||||
"Result may not be accurate.");
|
||||
}
|
||||
int_sigmoid_psi.Assemble();
|
||||
return int_sigmoid_psi.Sum();
|
||||
}
|
||||
|
||||
/*
|
||||
* ---------------------------------------------------------------
|
||||
* ALGORITHM PREAMBLE
|
||||
@@ -180,10 +133,11 @@ int main(int argc, char *argv[])
|
||||
int ref_levels = 5;
|
||||
int order = 2;
|
||||
real_t alpha = 1.0;
|
||||
real_t growth = 2;
|
||||
real_t epsilon = 0.01;
|
||||
real_t vol_fraction = 0.5;
|
||||
int max_it = 1e3;
|
||||
real_t itol = 1e-1;
|
||||
real_t itol = 1e-2;
|
||||
real_t ntol = 1e-4;
|
||||
real_t rho_min = 1e-6;
|
||||
real_t lambda = 1.0;
|
||||
@@ -198,6 +152,8 @@ int main(int argc, char *argv[])
|
||||
"Order (degree) of the finite elements.");
|
||||
args.AddOption(&alpha, "-alpha", "--alpha-step-length",
|
||||
"Step length for gradient descent.");
|
||||
args.AddOption(&growth, "-growth", "--alpha-growth-rate",
|
||||
"Growth rate of step length for gradient descent.");
|
||||
args.AddOption(&epsilon, "-epsilon", "--epsilon-thickness",
|
||||
"Length scale for ρ.");
|
||||
args.AddOption(&max_it, "-mi", "--max-it",
|
||||
@@ -332,6 +288,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
FilterSolver->SetEssentialBoundary(ess_bdr_filter);
|
||||
FilterSolver->SetupFEM();
|
||||
FilterSolver->AssembleDiffusionBilinear();
|
||||
|
||||
BilinearForm mass(&control_fes);
|
||||
mass.AddDomainIntegrator(new InverseIntegrator(new MassIntegrator(one)));
|
||||
@@ -385,7 +342,7 @@ int main(int argc, char *argv[])
|
||||
// 11. Iterate:
|
||||
for (int k = 1; k <= max_it; k++)
|
||||
{
|
||||
if (k > 1) { alpha *= ((real_t) k) / ((real_t) k-1); }
|
||||
if (k > 1) { alpha = std::pow((real_t) k,growth); }
|
||||
|
||||
mfem::out << "\nStep = " << k << std::endl;
|
||||
|
||||
@@ -422,7 +379,9 @@ int main(int argc, char *argv[])
|
||||
|
||||
// Step 5 - Update design variable ψ ← proj(ψ - αG)
|
||||
psi.Add(-alpha, grad);
|
||||
const real_t material_volume = proj(psi, target_volume);
|
||||
GridFunction alpha_grad(grad);
|
||||
alpha_grad *= alpha;
|
||||
const real_t material_volume = proj(psi, alpha_grad, target_volume);
|
||||
|
||||
// Compute ||ρ - ρ_old|| in control fes.
|
||||
real_t norm_increment = zerogf.ComputeL1Error(succ_diff_rho);
|
||||
|
||||
+189
-29
@@ -137,7 +137,7 @@ public:
|
||||
exponent(exponent_), rho_min(rho_min_)
|
||||
{
|
||||
MFEM_ASSERT(rho_min_ >= 0.0, "rho_min must be >= 0");
|
||||
MFEM_ASSERT(rho_min_ < 1.0, "rho_min must be > 1");
|
||||
MFEM_ASSERT(rho_min_ < 1.0, "rho_min must be < 1");
|
||||
MFEM_ASSERT(u, "displacement field is not set");
|
||||
MFEM_ASSERT(rho_filter, "density field is not set");
|
||||
}
|
||||
@@ -231,9 +231,12 @@ private:
|
||||
FiniteElementCollection * fec = nullptr;
|
||||
FiniteElementSpace * fes = nullptr;
|
||||
Array<int> ess_bdr;
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> neumann_bdr;
|
||||
GridFunction * u = nullptr;
|
||||
LinearForm * b = nullptr;
|
||||
BilinearForm * a = nullptr;
|
||||
OperatorPtr A;
|
||||
bool parallel;
|
||||
#ifdef MFEM_USE_MPI
|
||||
ParMesh * pmesh = nullptr;
|
||||
@@ -267,6 +270,8 @@ public:
|
||||
void ResetFEM();
|
||||
void SetupFEM();
|
||||
|
||||
void UpdateEssentialTDofs();
|
||||
void AssembleDiffusionBilinear(bool update_ess_tdofs=true);
|
||||
void Solve();
|
||||
GridFunction * GetFEMSolution();
|
||||
LinearForm * GetLinearForm() {return b;}
|
||||
@@ -371,6 +376,130 @@ public:
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
|
||||
* ∫_Ω ρ dx = θ vol(Ω) as follows:
|
||||
*
|
||||
* 1. Compute the root of the R → R function
|
||||
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
|
||||
* using the Illinois method
|
||||
* 2. Set ψ ← ψ + c.
|
||||
*
|
||||
* @param psi a GridFunction to be updated
|
||||
* @param alpha_grad alpha multiplied by gradient
|
||||
* @param target_volume θ vol(Ω)
|
||||
* @param tol Illinois iteration tolerance
|
||||
* @param max_its Illinois maximum iteration number
|
||||
* @return real_t Final volume (∫_Ω sigmoid(ψ) dx)
|
||||
*/
|
||||
real_t proj(GridFunction &psi, GridFunction &alpha_grad, real_t target_volume,
|
||||
real_t tol = 1e-12, int max_its = 100)
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
FiniteElementSpace *fes = psi.FESpace();
|
||||
ParFiniteElementSpace *pfes = dynamic_cast<ParFiniteElementSpace*>(fes);
|
||||
#endif
|
||||
ConstantCoefficient zero_cf(0.0);
|
||||
real_t a = -alpha_grad.ComputeMaxError(zero_cf);
|
||||
real_t b = -a;
|
||||
real_t y = 0.0;
|
||||
|
||||
MappedGridFunctionCoefficient sigmoid_psi(
|
||||
&psi, [&y](const real_t x) { return sigmoid(x + y); });
|
||||
std::unique_ptr<LinearForm> int_sigmoid_psi;
|
||||
#ifdef MFEM_USE_MPI
|
||||
ParGridFunction *par_psi = dynamic_cast<ParGridFunction *>(&psi);
|
||||
if (par_psi)
|
||||
{
|
||||
int_sigmoid_psi.reset(new ParLinearForm(par_psi->ParFESpace()));
|
||||
}
|
||||
else
|
||||
{
|
||||
int_sigmoid_psi.reset(new LinearForm(psi.FESpace()));
|
||||
}
|
||||
#else
|
||||
int_sigmoid_psi.reset(new LinearForm(psi.FESpace()));
|
||||
#endif
|
||||
int_sigmoid_psi->AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
|
||||
|
||||
y = a;
|
||||
int_sigmoid_psi->Assemble();
|
||||
real_t f_a = int_sigmoid_psi->Sum(); // f_a := f(a) + θ vol(Ω)
|
||||
|
||||
y = b;
|
||||
int_sigmoid_psi->Assemble();
|
||||
real_t f_b = int_sigmoid_psi->Sum(); // f_b := f(b) + θ vol(Ω)
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pfes)
|
||||
{
|
||||
MPI_Allreduce(MPI_IN_PLACE, &f_a, 1, MPITypeMap<real_t>::mpi_type,
|
||||
MPI_SUM, MPI_COMM_WORLD);
|
||||
MPI_Allreduce(MPI_IN_PLACE, &f_b, 1, MPITypeMap<real_t>::mpi_type,
|
||||
MPI_SUM, MPI_COMM_WORLD);
|
||||
}
|
||||
#endif
|
||||
f_a -= target_volume; // f_a := f(a)
|
||||
f_b -= target_volume; // f_b := f(b)
|
||||
real_t c = 0.0;
|
||||
real_t f_c = 0.0;
|
||||
int side = 0;
|
||||
|
||||
bool done = false;
|
||||
for (int k=0; k < max_its; k++)
|
||||
{
|
||||
c = (f_a * b - f_b * a) / (f_a - f_b);
|
||||
|
||||
if (abs(b - a) < tol * abs(b + a)) { done = true; break; }
|
||||
|
||||
y = c;
|
||||
int_sigmoid_psi->Assemble();
|
||||
f_c = int_sigmoid_psi->Sum(); // f_c := f(c) + θ vol(Ω)
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pfes)
|
||||
{
|
||||
MPI_Allreduce(MPI_IN_PLACE, &f_c, 1, MPITypeMap<real_t>::mpi_type,
|
||||
MPI_SUM, MPI_COMM_WORLD);
|
||||
}
|
||||
#endif
|
||||
f_c -= target_volume; // f_c := f(c)
|
||||
|
||||
if (f_c * f_b > 0)
|
||||
{
|
||||
b = c;
|
||||
f_b = f_c;
|
||||
if (side == -1) { f_a /= 2.0; }
|
||||
side = -1;
|
||||
}
|
||||
else if (f_c * f_a > 0)
|
||||
{
|
||||
a = c;
|
||||
f_a = f_c;
|
||||
if (side == 1) { f_b /= 2.0; }
|
||||
side = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
done = true; break;
|
||||
}
|
||||
}
|
||||
if (!done)
|
||||
{
|
||||
mfem_warning("Projection reached maximum iteration without converging. "
|
||||
"Result may not be accurate.");
|
||||
}
|
||||
y = 0.0;
|
||||
psi += c;
|
||||
int_sigmoid_psi->Assemble();
|
||||
real_t material_volume = int_sigmoid_psi->Sum();
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pfes)
|
||||
{
|
||||
MPI_Allreduce(MPI_IN_PLACE, &material_volume, 1,
|
||||
MPITypeMap<real_t>::mpi_type, MPI_SUM, MPI_COMM_WORLD);
|
||||
}
|
||||
#endif
|
||||
return material_volume;
|
||||
}
|
||||
|
||||
// Poisson solver
|
||||
|
||||
@@ -422,12 +551,8 @@ void DiffusionSolver::SetupFEM()
|
||||
}
|
||||
}
|
||||
|
||||
void DiffusionSolver::Solve()
|
||||
void DiffusionSolver::UpdateEssentialTDofs()
|
||||
{
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
Array<int> ess_tdof_list;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (parallel)
|
||||
{
|
||||
@@ -440,7 +565,39 @@ void DiffusionSolver::Solve()
|
||||
#else
|
||||
fes->GetEssentialTrueDofs(ess_bdr,ess_tdof_list);
|
||||
#endif
|
||||
*u=0.0;
|
||||
}
|
||||
|
||||
void DiffusionSolver::AssembleDiffusionBilinear(bool update_ess_tdofs)
|
||||
{
|
||||
if (update_ess_tdofs)
|
||||
{
|
||||
UpdateEssentialTDofs();
|
||||
}
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (parallel)
|
||||
{
|
||||
a = new ParBilinearForm(pfes);
|
||||
}
|
||||
else
|
||||
{
|
||||
a = new BilinearForm(fes);
|
||||
}
|
||||
#else
|
||||
a = new BilinearForm(fes);
|
||||
#endif
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(*diffcf));
|
||||
if (masscf)
|
||||
{
|
||||
a->AddDomainIntegrator(new MassIntegrator(*masscf));
|
||||
}
|
||||
a->Assemble();
|
||||
a->FormSystemMatrix(ess_tdof_list, A);
|
||||
}
|
||||
|
||||
void DiffusionSolver::Solve()
|
||||
{
|
||||
Vector B, X;
|
||||
|
||||
if (b)
|
||||
{
|
||||
delete b;
|
||||
@@ -475,31 +632,33 @@ void DiffusionSolver::Solve()
|
||||
|
||||
b->Assemble();
|
||||
|
||||
BilinearForm * a = nullptr;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (parallel)
|
||||
{
|
||||
a = new ParBilinearForm(pfes);
|
||||
}
|
||||
else
|
||||
{
|
||||
a = new BilinearForm(fes);
|
||||
}
|
||||
#else
|
||||
a = new BilinearForm(fes);
|
||||
#endif
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(*diffcf));
|
||||
if (masscf)
|
||||
{
|
||||
a->AddDomainIntegrator(new MassIntegrator(*masscf));
|
||||
}
|
||||
a->Assemble();
|
||||
*u=0.0;
|
||||
if (essbdr_cf)
|
||||
{
|
||||
u->ProjectBdrCoefficient(*essbdr_cf,ess_bdr);
|
||||
}
|
||||
a->FormLinearSystem(ess_tdof_list, *u, *b, A, X, B);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (parallel)
|
||||
{
|
||||
X.SetSize(pfes->TrueVSize());
|
||||
B.SetSize(pfes->TrueVSize());
|
||||
dynamic_cast<ParGridFunction*>(u)->ParallelAssemble(X);
|
||||
dynamic_cast<ParLinearForm*>(b)->ParallelAssemble(B);
|
||||
dynamic_cast<ParBilinearForm*>(a)->ParallelEliminateTDofsInRHS(
|
||||
ess_tdof_list, X, B);
|
||||
}
|
||||
else
|
||||
{
|
||||
X.NewDataAndSize(u->GetData(), u->Size());
|
||||
B.NewDataAndSize(b->GetData(), b->Size());
|
||||
a->EliminateVDofsInRHS(ess_tdof_list, X, B);
|
||||
}
|
||||
#else
|
||||
X.NewDataAndSize(u->GetData(), u->Size());
|
||||
B.NewDataAndSize(b->GetData(), b->Size());
|
||||
a->EliminateVDofsInRHS(ess_tdof_list, X, B);
|
||||
#endif
|
||||
|
||||
CGSolver * cg = nullptr;
|
||||
Solver * M = nullptr;
|
||||
@@ -528,7 +687,6 @@ void DiffusionSolver::Solve()
|
||||
delete M;
|
||||
delete cg;
|
||||
a->RecoverFEMSolution(X, *b, *u);
|
||||
delete a;
|
||||
}
|
||||
|
||||
GridFunction * DiffusionSolver::GetFEMSolution()
|
||||
@@ -560,6 +718,8 @@ DiffusionSolver::~DiffusionSolver()
|
||||
#endif
|
||||
delete fec; fec = nullptr;
|
||||
delete b;
|
||||
A.Clear();
|
||||
delete a;
|
||||
}
|
||||
|
||||
|
||||
|
||||
+11
-60
@@ -4,8 +4,8 @@
|
||||
//
|
||||
// Sample runs:
|
||||
// mpirun -np 4 ex37p -alpha 10 -pv
|
||||
// mpirun -np 4 ex37p -lambda 0.1 -mu 0.1
|
||||
// mpirun -np 4 ex37p -o 2 -alpha 5.0 -mi 50 -vf 0.4 -ntol 1e-5
|
||||
// mpirun -np 4 ex37p -lambda 0.1 -mu 0.1 -growth 1
|
||||
// mpirun -np 4 ex37p -o 2 -alpha 10.0 -mi 50 -vf 0.4 -ntol 1e-5 -growth 1.5
|
||||
// mpirun -np 4 ex37p -r 6 -o 2 -alpha 10.0 -epsilon 0.02 -mi 50 -ntol 1e-5
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to solve a
|
||||
@@ -54,61 +54,6 @@
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
/**
|
||||
* @brief Bregman projection of ρ = sigmoid(ψ) onto the subspace
|
||||
* ∫_Ω ρ dx = θ vol(Ω) as follows:
|
||||
*
|
||||
* 1. Compute the root of the R → R function
|
||||
* f(c) = ∫_Ω sigmoid(ψ + c) dx - θ vol(Ω)
|
||||
* 2. Set ψ ← ψ + c.
|
||||
*
|
||||
* @param psi a GridFunction to be updated
|
||||
* @param target_volume θ vol(Ω)
|
||||
* @param tol Newton iteration tolerance
|
||||
* @param max_its Newton maximum iteration number
|
||||
* @return real_t Final volume, ∫_Ω sigmoid(ψ)
|
||||
*/
|
||||
real_t proj(ParGridFunction &psi, real_t target_volume, real_t tol=1e-12,
|
||||
int max_its=10)
|
||||
{
|
||||
MappedGridFunctionCoefficient sigmoid_psi(&psi, sigmoid);
|
||||
MappedGridFunctionCoefficient der_sigmoid_psi(&psi, der_sigmoid);
|
||||
|
||||
ParLinearForm int_sigmoid_psi(psi.ParFESpace());
|
||||
int_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(sigmoid_psi));
|
||||
ParLinearForm int_der_sigmoid_psi(psi.ParFESpace());
|
||||
int_der_sigmoid_psi.AddDomainIntegrator(new DomainLFIntegrator(
|
||||
der_sigmoid_psi));
|
||||
bool done = false;
|
||||
for (int k=0; k<max_its; k++) // Newton iteration
|
||||
{
|
||||
int_sigmoid_psi.Assemble(); // Recompute f(c) with updated ψ
|
||||
real_t f = int_sigmoid_psi.Sum();
|
||||
MPI_Allreduce(MPI_IN_PLACE, &f, 1, MPITypeMap<real_t>::mpi_type,
|
||||
MPI_SUM, MPI_COMM_WORLD);
|
||||
f -= target_volume;
|
||||
|
||||
int_der_sigmoid_psi.Assemble(); // Recompute df(c) with updated ψ
|
||||
real_t df = int_der_sigmoid_psi.Sum();
|
||||
MPI_Allreduce(MPI_IN_PLACE, &df, 1, MPITypeMap<real_t>::mpi_type,
|
||||
MPI_SUM, MPI_COMM_WORLD);
|
||||
|
||||
const real_t dc = -f/df;
|
||||
psi += dc;
|
||||
if (abs(dc) < tol) { done = true; break; }
|
||||
}
|
||||
if (!done)
|
||||
{
|
||||
mfem_warning("Projection reached maximum iteration without converging. "
|
||||
"Result may not be accurate.");
|
||||
}
|
||||
int_sigmoid_psi.Assemble();
|
||||
real_t material_volume = int_sigmoid_psi.Sum();
|
||||
MPI_Allreduce(MPI_IN_PLACE, &material_volume, 1,
|
||||
MPITypeMap<real_t>::mpi_type, MPI_SUM, MPI_COMM_WORLD);
|
||||
return material_volume;
|
||||
}
|
||||
|
||||
/*
|
||||
* ---------------------------------------------------------------
|
||||
* ALGORITHM PREAMBLE
|
||||
@@ -193,10 +138,11 @@ int main(int argc, char *argv[])
|
||||
int ref_levels = 5;
|
||||
int order = 2;
|
||||
real_t alpha = 1.0;
|
||||
real_t growth = 2;
|
||||
real_t epsilon = 0.01;
|
||||
real_t vol_fraction = 0.5;
|
||||
int max_it = 1e3;
|
||||
real_t itol = 1e-1;
|
||||
real_t itol = 1e-2;
|
||||
real_t ntol = 1e-4;
|
||||
real_t rho_min = 1e-6;
|
||||
real_t lambda = 1.0;
|
||||
@@ -211,6 +157,8 @@ int main(int argc, char *argv[])
|
||||
"Order (degree) of the finite elements.");
|
||||
args.AddOption(&alpha, "-alpha", "--alpha-step-length",
|
||||
"Step length for gradient descent.");
|
||||
args.AddOption(&growth, "-growth", "--alpha-growth-rate",
|
||||
"Growth rate of step length for gradient descent.");
|
||||
args.AddOption(&epsilon, "-epsilon", "--epsilon-thickness",
|
||||
"Length scale for ρ.");
|
||||
args.AddOption(&max_it, "-mi", "--max-it",
|
||||
@@ -359,6 +307,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
FilterSolver->SetEssentialBoundary(ess_bdr_filter);
|
||||
FilterSolver->SetupFEM();
|
||||
FilterSolver->AssembleDiffusionBilinear();
|
||||
|
||||
ParBilinearForm mass(&control_fes);
|
||||
mass.AddDomainIntegrator(new InverseIntegrator(new MassIntegrator(one)));
|
||||
@@ -412,7 +361,7 @@ int main(int argc, char *argv[])
|
||||
// 11. Iterate:
|
||||
for (int k = 1; k <= max_it; k++)
|
||||
{
|
||||
if (k > 1) { alpha *= ((real_t) k) / ((real_t) k-1); }
|
||||
if (k > 1) { alpha = std::pow((real_t) k,growth); }
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
@@ -452,7 +401,9 @@ int main(int argc, char *argv[])
|
||||
|
||||
// Step 5 - Update design variable ψ ← proj(ψ - αG)
|
||||
psi.Add(-alpha, grad);
|
||||
const real_t material_volume = proj(psi, target_volume);
|
||||
ParGridFunction alpha_grad(grad);
|
||||
alpha_grad *= alpha;
|
||||
const real_t material_volume = proj(psi, alpha_grad, target_volume);
|
||||
|
||||
// Compute ||ρ - ρ_old|| in control fes.
|
||||
real_t norm_increment = zerogf.ComputeL1Error(succ_diff_rho);
|
||||
|
||||
@@ -0,0 +1,536 @@
|
||||
#include <mfem.hpp>
|
||||
|
||||
#include "../fem/dfem/util.hpp"
|
||||
|
||||
#include <proteus/CppJitModule.h>
|
||||
|
||||
#include "jitplayground.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cctype>
|
||||
#include <cmath>
|
||||
#include <fstream>
|
||||
#include <initializer_list>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace util
|
||||
{
|
||||
constexpr std::string_view Dirname(std::string_view path)
|
||||
{
|
||||
const size_t last_sep = path.find_last_of("/\\");
|
||||
if (last_sep == std::string_view::npos) { return {}; }
|
||||
return path.substr(0, last_sep);
|
||||
}
|
||||
|
||||
constexpr std::string_view thisFileDir = Dirname(__FILE__);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static std::string TypeNameString()
|
||||
{
|
||||
return std::string(mfem::future::get_type_name<T>());
|
||||
}
|
||||
|
||||
template <typename Tuple, size_t... Is>
|
||||
static auto ParamTypeStringsImpl(std::index_sequence<Is...>)
|
||||
{
|
||||
return std::array<std::string, sizeof...(Is)>
|
||||
{
|
||||
TypeNameString<std::remove_reference_t<decltype(mfem::future::get<Is>(std::declval<Tuple&>()))>>()...
|
||||
};
|
||||
}
|
||||
|
||||
template <typename Tuple>
|
||||
static auto ParamTypeStrings()
|
||||
{
|
||||
return ParamTypeStringsImpl<Tuple>(
|
||||
std::make_index_sequence<mfem::future::tuple_size<Tuple>::value> {});
|
||||
}
|
||||
|
||||
static std::string_view Trim(std::string_view s)
|
||||
{
|
||||
size_t begin = 0;
|
||||
while (begin < s.size() && std::isspace(static_cast<unsigned char>(s[begin])))
|
||||
{
|
||||
++begin;
|
||||
}
|
||||
size_t end = s.size();
|
||||
while (end > begin &&
|
||||
std::isspace(static_cast<unsigned char>(s[end - 1])))
|
||||
{
|
||||
--end;
|
||||
}
|
||||
return s.substr(begin, end - begin);
|
||||
}
|
||||
|
||||
static bool IsValidIdentifier(std::string_view s)
|
||||
{
|
||||
if (s.empty()) { return false; }
|
||||
const unsigned char c0 = static_cast<unsigned char>(s[0]);
|
||||
if (!(std::isalpha(c0) || c0 == '_')) { return false; }
|
||||
for (size_t i = 1; i < s.size(); ++i)
|
||||
{
|
||||
const unsigned char c = static_cast<unsigned char>(s[i]);
|
||||
if (!(std::isalnum(c) || c == '_')) { return false; }
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool ParseJitDirective(std::string_view line,
|
||||
std::string &type,
|
||||
std::string &var,
|
||||
std::string &kind)
|
||||
{
|
||||
const size_t jit_pos = line.find("$JIT");
|
||||
if (jit_pos == std::string_view::npos) { return false; }
|
||||
|
||||
const size_t open = line.find('[', jit_pos);
|
||||
const size_t close = line.find(']', jit_pos);
|
||||
MFEM_VERIFY(open != std::string_view::npos &&
|
||||
close != std::string_view::npos &&
|
||||
close > open,
|
||||
"malformed $JIT directive (expected brackets): " << line);
|
||||
|
||||
const std::string_view payload = line.substr(open + 1, close - open - 1);
|
||||
const size_t comma1 = payload.find(',');
|
||||
const size_t comma2 = (comma1 == std::string_view::npos)
|
||||
? std::string_view::npos
|
||||
: payload.find(',', comma1 + 1);
|
||||
MFEM_VERIFY(comma1 != std::string_view::npos &&
|
||||
comma2 != std::string_view::npos,
|
||||
"malformed $JIT directive (expected 3 comma-separated fields): "
|
||||
<< line);
|
||||
|
||||
const std::string_view f0 = Trim(payload.substr(0, comma1));
|
||||
const std::string_view f1 = Trim(payload.substr(comma1 + 1,
|
||||
comma2 - comma1 - 1));
|
||||
const std::string_view f2 = Trim(payload.substr(comma2 + 1));
|
||||
MFEM_VERIFY(!f0.empty() && !f1.empty() && !f2.empty(),
|
||||
"malformed $JIT directive (empty field): " << line);
|
||||
|
||||
type.assign(f0);
|
||||
var.assign(f1);
|
||||
kind.assign(f2);
|
||||
return true;
|
||||
}
|
||||
|
||||
static std::string ReadFileOrEmpty(const std::string &fn)
|
||||
{
|
||||
std::ifstream file(fn);
|
||||
if (!file.is_open())
|
||||
{
|
||||
std::cerr << "could not open file " << fn << "\n";
|
||||
return {};
|
||||
}
|
||||
std::stringstream buffer;
|
||||
buffer << file.rdbuf();
|
||||
return buffer.str();
|
||||
}
|
||||
|
||||
static std::vector<std::string> ExtractJitVarNames(const std::string
|
||||
&kernel_code)
|
||||
{
|
||||
std::stringstream ss(kernel_code);
|
||||
std::string line;
|
||||
std::vector<std::string> var_names;
|
||||
std::unordered_set<std::string> seen_vars;
|
||||
|
||||
while (std::getline(ss, line))
|
||||
{
|
||||
std::string type, var, kind;
|
||||
if (ParseJitDirective(line, type, var, kind))
|
||||
{
|
||||
MFEM_VERIFY(IsValidIdentifier(var),
|
||||
"$JIT variable must be a valid identifier: " << var);
|
||||
MFEM_VERIFY(seen_vars.insert(var).second,
|
||||
"duplicate $JIT variable name: " << var);
|
||||
var_names.push_back(var);
|
||||
}
|
||||
}
|
||||
return var_names;
|
||||
}
|
||||
|
||||
static std::string RewriteKernelForJit(std::string kernel_code,
|
||||
const std::vector<std::string> &jit_values)
|
||||
{
|
||||
std::stringstream ss(kernel_code);
|
||||
std::string line;
|
||||
|
||||
std::string out;
|
||||
out.reserve(kernel_code.size() + 128);
|
||||
|
||||
bool have_pending = false;
|
||||
size_t pending_index = 0;
|
||||
std::string pending_type;
|
||||
std::string pending_var;
|
||||
std::unordered_set<std::string> seen_vars;
|
||||
|
||||
while (std::getline(ss, line))
|
||||
{
|
||||
line.push_back('\n');
|
||||
|
||||
if (have_pending)
|
||||
{
|
||||
MFEM_VERIFY(pending_index < jit_values.size(),
|
||||
"not enough JIT values provided");
|
||||
const size_t indent_end = line.find_first_not_of(" \t");
|
||||
const std::string indent =
|
||||
(indent_end == std::string::npos) ? std::string() :
|
||||
line.substr(0, indent_end);
|
||||
out += indent + "const " + pending_type + " " + pending_var + " = " +
|
||||
jit_values[pending_index] + ";\n";
|
||||
have_pending = false;
|
||||
++pending_index;
|
||||
continue;
|
||||
}
|
||||
|
||||
std::string type, var, kind;
|
||||
if (ParseJitDirective(line, type, var, kind))
|
||||
{
|
||||
MFEM_VERIFY(IsValidIdentifier(var),
|
||||
"$JIT variable must be a valid identifier: " << var);
|
||||
MFEM_VERIFY(kind == "generic",
|
||||
"unsupported $JIT kind: " << kind);
|
||||
MFEM_VERIFY(seen_vars.insert(var).second,
|
||||
"duplicate $JIT variable name: " << var);
|
||||
|
||||
pending_type = std::move(type);
|
||||
pending_var = std::move(var);
|
||||
have_pending = true;
|
||||
continue; // drop directive line
|
||||
}
|
||||
|
||||
out += line;
|
||||
}
|
||||
|
||||
MFEM_VERIFY(!have_pending,
|
||||
"$JIT directive must annotate a following line");
|
||||
MFEM_VERIFY(jit_values.size() == pending_index,
|
||||
"JIT value count must match number of $JIT directives");
|
||||
return out;
|
||||
}
|
||||
|
||||
static std::string GeneratedOutputPath(std::string_view original_path)
|
||||
{
|
||||
const size_t last_sep = original_path.find_last_of("/\\");
|
||||
const size_t dot = original_path.find_last_of('.');
|
||||
const bool dot_in_filename =
|
||||
(dot != std::string_view::npos) &&
|
||||
(last_sep == std::string_view::npos || dot > last_sep);
|
||||
|
||||
const std::string_view base =
|
||||
dot_in_filename ? original_path.substr(0, dot) : original_path;
|
||||
return std::string(base) + "_generated.hpp";
|
||||
}
|
||||
|
||||
static void WriteFileOrWarn(const std::string &path,
|
||||
const std::string &contents)
|
||||
{
|
||||
std::ofstream out(path);
|
||||
if (!out.is_open())
|
||||
{
|
||||
std::cerr << "could not write generated file " << path << "\n";
|
||||
return;
|
||||
}
|
||||
out << contents;
|
||||
}
|
||||
|
||||
class JitQFunction
|
||||
{
|
||||
public:
|
||||
template <typename ImplT, size_t N>
|
||||
JitQFunction(ImplT, const std::string &fn,
|
||||
const std::array<bool, N> &activity_map)
|
||||
{
|
||||
using qf_signature = typename
|
||||
mfem::future::get_function_signature<
|
||||
decltype(&ImplT::operator())>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
constexpr size_t nparams = mfem::future::tuple_size<qf_param_ts>::value;
|
||||
static_assert(N == nparams, "activity_map size must match qfunc arity");
|
||||
|
||||
this->fn = fn;
|
||||
this->nparams = nparams;
|
||||
this->activity_map.reserve(N);
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
this->activity_map.push_back(activity_map[i]);
|
||||
}
|
||||
{
|
||||
const auto param_types_arr = ParamTypeStrings<qf_param_ts>();
|
||||
this->param_types.assign(param_types_arr.begin(), param_types_arr.end());
|
||||
}
|
||||
this->return_type = TypeNameString<typename qf_signature::return_t>();
|
||||
this->return_is_void = std::is_same_v<typename qf_signature::return_t, void>;
|
||||
this->impl_type_name = TypeNameString<ImplT>();
|
||||
this->jit_var_names = ExtractJitVarNames(ReadFileOrEmpty(fn));
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run(std::string_view name,
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
auto ordered_values = MatchJitValues(jit_values);
|
||||
auto &mod = GetOrCreateModule(ordered_values);
|
||||
auto &instance = mod.instantiate(std::string(name), std::string());
|
||||
return instance.template run<ReturnT>(std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run_primal(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
return run<ReturnT>(qfunc_name, jit_values,
|
||||
std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run_derivative(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
return run<ReturnT>(qfunc_name + "_fwddiff", jit_values,
|
||||
std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
private:
|
||||
std::vector<std::string_view> MatchJitValues(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>>
|
||||
named_values) const
|
||||
{
|
||||
std::unordered_map<std::string_view, std::string_view> value_map;
|
||||
for (const auto &[name, value] : named_values)
|
||||
{
|
||||
value_map[name] = value;
|
||||
}
|
||||
|
||||
std::vector<std::string_view> ordered_values;
|
||||
ordered_values.reserve(jit_var_names.size());
|
||||
for (const auto &var_name : jit_var_names)
|
||||
{
|
||||
auto it = value_map.find(var_name);
|
||||
MFEM_VERIFY(it != value_map.end(),
|
||||
"missing JIT value for variable: " << var_name);
|
||||
ordered_values.push_back(it->second);
|
||||
}
|
||||
|
||||
MFEM_VERIFY(ordered_values.size() == named_values.size(),
|
||||
"provided " << named_values.size() << " JIT values but expected "
|
||||
<< jit_var_names.size());
|
||||
return ordered_values;
|
||||
}
|
||||
|
||||
|
||||
std::string BuildModuleCode(const std::vector<std::string> &jit_values) const
|
||||
{
|
||||
std::string module_code =
|
||||
RewriteKernelForJit(ReadFileOrEmpty(fn), jit_values);
|
||||
module_code += "\n\n";
|
||||
module_code += "// --- generated ---\n";
|
||||
module_code +=
|
||||
"template <typename return_type, typename... Args>\n"
|
||||
"return_type __enzyme_fwddiff(Args...);\n"
|
||||
"\n"
|
||||
"extern int enzyme_const;\n"
|
||||
"extern int enzyme_dup;\n"
|
||||
"\n";
|
||||
|
||||
// Generate a primal wrapper with the requested symbol name, so the kernel
|
||||
// header can just define the qfunc as a functor.
|
||||
//
|
||||
// Note: Proteus instantiates entrypoints via `qfunc_wrapper<>(...)` even
|
||||
// when there are no user template args, so keep the wrapper itself a
|
||||
// template (with a default parameter) while still doing literal `$JIT`
|
||||
// replacements in the kernel code.
|
||||
module_code += "template <typename = void>\n";
|
||||
module_code += return_type + " " +
|
||||
std::string(qfunc_name) + "(";
|
||||
bool first = true;
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (!first) { module_code += ", "; }
|
||||
first = false;
|
||||
module_code += param_types[i] + " Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += ")\n";
|
||||
module_code += "{\n";
|
||||
module_code += " " + impl_type_name + " qf;\n";
|
||||
if (return_is_void)
|
||||
{
|
||||
module_code += " ";
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " return ";
|
||||
}
|
||||
module_code += "qf(";
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (i) { module_code += ", "; }
|
||||
module_code += "Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += ");\n";
|
||||
module_code += "}\n\n";
|
||||
|
||||
module_code += "template <typename = void>\n";
|
||||
module_code += return_type + " " +
|
||||
std::string(qfunc_name) + "_fwddiff(";
|
||||
|
||||
first = true;
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (!first) { module_code += ", "; }
|
||||
first = false;
|
||||
module_code += param_types[i] + " Arg" + std::to_string(i);
|
||||
if (activity_map[i])
|
||||
{
|
||||
module_code += ", " + param_types[i] + " dArg" + std::to_string(i);
|
||||
}
|
||||
}
|
||||
module_code += ")\n";
|
||||
module_code += "{\n";
|
||||
if (return_is_void)
|
||||
{
|
||||
module_code += " __enzyme_fwddiff<void>(\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " return __enzyme_fwddiff<" +
|
||||
return_type + ">(\n";
|
||||
}
|
||||
module_code += " (void*)" + std::string(qfunc_name) + "<>";
|
||||
module_code += ",\n";
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (activity_map[i])
|
||||
{
|
||||
module_code += " enzyme_dup, Arg" + std::to_string(i) +
|
||||
", dArg" + std::to_string(i);
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " enzyme_const, Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += (i + 1 == nparams) ? ");\n" : ",\n";
|
||||
}
|
||||
module_code += "}\n";
|
||||
|
||||
WriteFileOrWarn(GeneratedOutputPath(fn), module_code);
|
||||
return module_code;
|
||||
}
|
||||
|
||||
proteus::CppJitModule &GetOrCreateModule(
|
||||
const std::vector<std::string_view> &jit_values)
|
||||
{
|
||||
std::string key;
|
||||
for (const auto &val : jit_values)
|
||||
{
|
||||
if (!key.empty()) { key += ","; }
|
||||
key += val;
|
||||
}
|
||||
|
||||
auto it = modules.find(key);
|
||||
if (it != modules.end())
|
||||
{
|
||||
return *it->second;
|
||||
}
|
||||
|
||||
std::vector<std::string> values(jit_values.begin(), jit_values.end());
|
||||
std::string code = BuildModuleCode(values);
|
||||
auto mod = std::make_unique<proteus::CppJitModule>("host", code,
|
||||
DefaultExtraArgs());
|
||||
auto [inserted, ok] = modules.emplace(key, std::move(mod));
|
||||
MFEM_VERIFY(ok, "failed to cache JIT module");
|
||||
return *inserted->second;
|
||||
}
|
||||
|
||||
static std::vector<std::string> DefaultExtraArgs()
|
||||
{
|
||||
return {"-fplugin=/Users/andrej1/local/enzyme/lib/ClangEnzyme-20.dylib"};
|
||||
}
|
||||
|
||||
std::string qfunc_name = "qfunc_wrapper";
|
||||
std::string fn;
|
||||
size_t nparams = 0;
|
||||
std::vector<bool> activity_map;
|
||||
std::vector<std::string> param_types;
|
||||
std::string return_type;
|
||||
bool return_is_void = false;
|
||||
std::string impl_type_name;
|
||||
std::vector<std::string> jit_var_names;
|
||||
std::unordered_map<std::string, std::unique_ptr<proteus::CppJitModule>> modules;
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
const size_t N = 4;
|
||||
const size_t M = 5;
|
||||
const double A = 123.4;
|
||||
|
||||
std::vector<double> X(N);
|
||||
std::vector<double> Y(N);
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
X[i] = static_cast<double>(i + 1);
|
||||
Y[i] = static_cast<double>(N - i);
|
||||
}
|
||||
|
||||
// // >>> user interface calls
|
||||
// const std::string kernel_path = std::string(util::thisFileDir) +
|
||||
// "/jitplayground.hpp";
|
||||
// JitQFunction qf(daxpy_op{}, kernel_path, std::array{false, true, false});
|
||||
// // <<< user interface calls
|
||||
|
||||
// // this will happen internally in dFEM
|
||||
|
||||
daxpy_op op;
|
||||
printf("\n\nfunction call\n");
|
||||
op(&A, X.data(), Y.data(), &N);
|
||||
|
||||
// reset X for the derivative test
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
X[i] = static_cast<double>(i + 1);
|
||||
Y[i] = static_cast<double>(N - i);
|
||||
}
|
||||
|
||||
std::vector<double> dX(N, 1.0);
|
||||
printf("\n\nforward diff call\n");
|
||||
daxpy_op_fwddiff(&A, X.data(), dX.data(), Y.data(), &N);
|
||||
|
||||
std::vector<double> dX_manual(N, A);
|
||||
|
||||
printf("\n\nderivative checks\n");
|
||||
std::cout << "dX: ";
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
std::cout << dX[i] << (i + 1 == N ? '\n' : ' ');
|
||||
}
|
||||
|
||||
std::cout << "dX_manual: ";
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
std::cout << dX_manual[i] << (i + 1 == N ? '\n' : ' ');
|
||||
}
|
||||
|
||||
double max_abs_err = 0.0;
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
max_abs_err = std::max(max_abs_err, std::abs(dX[i] - dX_manual[i]));
|
||||
}
|
||||
std::cout << "max |dX - dX_manual| = " << max_abs_err << "\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <vector>
|
||||
#include <type_traits>
|
||||
|
||||
#include "proteus/JitInterface.h"
|
||||
|
||||
struct daxpy_op
|
||||
{
|
||||
void operator()(
|
||||
const double *a,
|
||||
double *x,
|
||||
const double *y,
|
||||
const size_t *N) const
|
||||
{
|
||||
const size_t n = *N;
|
||||
auto lam = [=, n = proteus::jit_variable(n)]
|
||||
() __attribute__((annotate("jit")))
|
||||
{
|
||||
printf("N = %zu\n", n);
|
||||
for (size_t i = 0; i < n; ++i)
|
||||
{
|
||||
printf("x[%zu] = %f, y[%zu] = %f\n", i, x[i], i, y[i]);
|
||||
x[i] = *a * x[i] + y[i];
|
||||
printf("updated x[%zu] = %f\n", i, x[i]);
|
||||
}
|
||||
};
|
||||
|
||||
proteus::register_lambda(lam);
|
||||
|
||||
lam();
|
||||
}
|
||||
};
|
||||
|
||||
template <typename return_type, typename... Args>
|
||||
return_type __enzyme_fwddiff(Args...);
|
||||
|
||||
extern int enzyme_const;
|
||||
extern int enzyme_dup;
|
||||
|
||||
void daxpy_op_wrapper(const double * Arg0, double * Arg1,
|
||||
const double * Arg2, const size_t *Arg3)
|
||||
{
|
||||
daxpy_op qf;
|
||||
qf(Arg0, Arg1, Arg2, Arg3);
|
||||
}
|
||||
|
||||
void daxpy_op_fwddiff(const double * Arg0, double * Arg1,
|
||||
double * dArg1, const double * Arg2, const size_t *Arg3)
|
||||
{
|
||||
__enzyme_fwddiff<void>(
|
||||
(void*)daxpy_op_wrapper,
|
||||
enzyme_const, Arg0,
|
||||
enzyme_dup, Arg1, dArg1,
|
||||
enzyme_const, Arg2,
|
||||
enzyme_const, Arg3);
|
||||
}
|
||||
@@ -157,6 +157,8 @@ ex37-test-seq: ex37
|
||||
@$(call mfem-test,$<,, Serial example,-mi 3)
|
||||
ex37p-test-par: ex37p
|
||||
@$(call mfem-test,$<, $(RUN_MPI), Parallel example,-mi 3)
|
||||
ex39-test-seq: ex39
|
||||
@$(call mfem-test,$<,, Serial example,-m ../data/compass.mesh)
|
||||
ex41-test-seq: ex41
|
||||
@$(call mfem-test,$<,, Serial example,-tf 1.0)
|
||||
ex41p-test-par: ex41p
|
||||
|
||||
+20
-13
@@ -121,6 +121,11 @@ set(SRCS
|
||||
qinterp/eval_hdiv.cpp
|
||||
qinterp/grad_by_nodes.cpp
|
||||
qinterp/grad_by_vdim.cpp
|
||||
qinterp/grad_transpose.cpp
|
||||
qinterp/grad_transpose_by_nodes.cpp
|
||||
qinterp/grad_transpose_by_vdim.cpp
|
||||
qinterp/eval_transpose.cpp
|
||||
qinterp/eval_transpose_by_vdim.cpp
|
||||
qspace.cpp
|
||||
quadinterpolator.cpp
|
||||
quadinterpolator_face.cpp
|
||||
@@ -133,7 +138,7 @@ set(SRCS
|
||||
tmop/assemble/diag2.cpp
|
||||
tmop/assemble/grad2_limit.cpp
|
||||
tmop/assemble/grad2.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3.cpp
|
||||
tmop/assemble/grad3_limit.cpp
|
||||
tmop/assemble/grad3.cpp
|
||||
@@ -278,8 +283,10 @@ set(HDRS
|
||||
qfunction.hpp
|
||||
qinterp/det.hpp
|
||||
qinterp/eval.hpp
|
||||
qinterp/eval_transpose.hpp
|
||||
qinterp/eval_hdiv.hpp
|
||||
qinterp/grad.hpp
|
||||
qinterp/grad_transpose.hpp
|
||||
qspace.hpp
|
||||
quadinterpolator.hpp
|
||||
quadinterpolator_face.hpp
|
||||
@@ -313,36 +320,36 @@ set(HDRS
|
||||
)
|
||||
|
||||
if (MFEM_USE_SIDRE)
|
||||
list(APPEND SRCS sidredatacollection.cpp)
|
||||
list(APPEND HDRS sidredatacollection.hpp)
|
||||
list(APPEND SRCS sidredatacollection.cpp)
|
||||
list(APPEND HDRS sidredatacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CONDUIT)
|
||||
list(APPEND SRCS conduitdatacollection.cpp)
|
||||
list(APPEND HDRS conduitdatacollection.hpp)
|
||||
list(APPEND SRCS conduitdatacollection.cpp)
|
||||
list(APPEND HDRS conduitdatacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_ADIOS2)
|
||||
list(APPEND SRCS adios2datacollection.cpp)
|
||||
list(APPEND HDRS adios2datacollection.hpp)
|
||||
list(APPEND SRCS adios2datacollection.cpp)
|
||||
list(APPEND HDRS adios2datacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_FMS)
|
||||
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
|
||||
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
|
||||
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
|
||||
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
list(APPEND SRCS
|
||||
list(APPEND SRCS
|
||||
pbilinearform.cpp
|
||||
pfespace.cpp
|
||||
pgridfunc.cpp
|
||||
plinearform.cpp
|
||||
pnonlinearform.cpp
|
||||
prestriction.cpp)
|
||||
# If this list (HDRS -> HEADERS) is used for install, we probably want the
|
||||
# headers added all the time.
|
||||
list(APPEND HDRS
|
||||
# If this list (HDRS -> HEADERS) is used for install, we probably want the
|
||||
# headers added all the time.
|
||||
list(APPEND HDRS
|
||||
pbilinearform.hpp
|
||||
pfespace.hpp
|
||||
pgridfunc.hpp
|
||||
|
||||
@@ -729,7 +729,8 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
tr = mesh -> GetBdrFaceTransformations (i);
|
||||
if (tr != NULL)
|
||||
{
|
||||
fes -> GetElementVDofs (tr -> Elem1No, vdofs);
|
||||
mfem::DofTransformation doftrans;
|
||||
fes -> GetElementVDofs (tr -> Elem1No, vdofs, doftrans);
|
||||
fe1 = fes -> GetFE (tr -> Elem1No);
|
||||
// The fe2 object is really a dummy and not used on the boundaries,
|
||||
// but we can't dereference a NULL pointer, and we don't want to
|
||||
@@ -743,6 +744,7 @@ void BilinearForm::Assemble(int skip_zeros)
|
||||
|
||||
boundary_face_integs[k] -> AssembleFaceMatrix (*fe1, *fe2, *tr,
|
||||
elemmat);
|
||||
doftrans.TransformDual(elemmat);
|
||||
mat -> AddSubMatrix (vdofs, vdofs, elemmat, skip_zeros);
|
||||
}
|
||||
}
|
||||
@@ -1723,6 +1725,7 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
}
|
||||
}
|
||||
|
||||
DofTransformation dom_dof_trans, ran_dof_trans;
|
||||
for (int i = 0; i < trial_fes -> GetNBE(); i++)
|
||||
{
|
||||
const int bdr_attr = mesh->GetBdrAttribute(i);
|
||||
@@ -1731,8 +1734,8 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
ftr = mesh -> GetBdrFaceTransformations (i);
|
||||
if (ftr != NULL)
|
||||
{
|
||||
trial_fes->GetElementVDofs(ftr->Elem1No, trial_vdofs);
|
||||
test_fes->GetElementVDofs(ftr->Elem1No, test_vdofs);
|
||||
trial_fes->GetElementVDofs(ftr->Elem1No, trial_vdofs, dom_dof_trans);
|
||||
test_fes->GetElementVDofs(ftr->Elem1No, test_vdofs, ran_dof_trans);
|
||||
trial_fe1 = trial_fes->GetFE(ftr->Elem1No);
|
||||
test_fe1 = test_fes->GetFE(ftr->Elem1No);
|
||||
// The test_fe2 object is really a dummy and not used on the
|
||||
@@ -1748,6 +1751,7 @@ void MixedBilinearForm::Assemble(int skip_zeros)
|
||||
boundary_face_integs[k]->AssembleFaceMatrix(*trial_fe1, *test_fe1, *trial_fe2,
|
||||
*test_fe2,
|
||||
*ftr, elemmat);
|
||||
TransformDual(ran_dof_trans, dom_dof_trans, elemmat);
|
||||
mat->AddSubMatrix(test_vdofs, trial_vdofs, elemmat, skip_zeros);
|
||||
}
|
||||
}
|
||||
|
||||
+18
-13
@@ -2178,18 +2178,22 @@ class DiffusionIntegrator: public BilinearFormIntegrator
|
||||
{
|
||||
public:
|
||||
|
||||
using ApplyKernelType = void(*)(const int, const bool, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&,
|
||||
const Vector&, const Vector&,
|
||||
Vector&, const int, const int);
|
||||
using DiffusionApplyKernelType = void(*)(const int, const bool,
|
||||
const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&,
|
||||
const Vector&, const Vector&,
|
||||
Vector&, const int, const int);
|
||||
|
||||
using DiagonalKernelType = void(*)(const int, const bool, const Array<real_t>&,
|
||||
const Array<real_t>&, const Vector&, Vector&,
|
||||
const int, const int);
|
||||
using DiffusionDiagonalKernelType = void(*)(const int, const bool,
|
||||
const Array<real_t>&,
|
||||
const Array<real_t>&, const Vector&, Vector&,
|
||||
const int, const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
|
||||
MFEM_REGISTER_KERNELS(DiffusionApplyPAKernel, DiffusionApplyKernelType,
|
||||
(int, int, int));
|
||||
MFEM_REGISTER_KERNELS(DiffusionDiagonalPAKernel, DiffusionDiagonalKernelType,
|
||||
(int, int, int));
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
protected:
|
||||
@@ -2209,6 +2213,7 @@ private:
|
||||
const FiniteElementSpace *fespace;
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
public:
|
||||
int dim, ne, dofs1D, quad1D;
|
||||
Vector pa_data;
|
||||
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
|
||||
@@ -2350,8 +2355,8 @@ public:
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
static void AddSpecialization()
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiffusionApplyPAKernel::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiffusionDiagonalPAKernel::Specialization<DIM,D1D,Q1D>::Add();
|
||||
}
|
||||
protected:
|
||||
const IntegrationRule* GetDefaultIntegrationRule(
|
||||
@@ -2710,7 +2715,7 @@ public:
|
||||
|
||||
|
||||
/** Integrator for $(-Q u, \nabla v)$ for Nedelec ($u$) and $H^1$ ($v$) elements.
|
||||
This is equivalent to a weak divergence of the $H(curl$ basis functions. */
|
||||
This is equivalent to a weak divergence of the $H(curl)$ basis functions. */
|
||||
class VectorFEWeakDivergenceIntegrator: public BilinearFormIntegrator
|
||||
{
|
||||
protected:
|
||||
|
||||
@@ -52,6 +52,9 @@ public:
|
||||
/// Get the time for time dependent coefficients
|
||||
real_t GetTime() { return time; }
|
||||
|
||||
/// Returns dimension of the vector.
|
||||
int GetVDim() { return 1; }
|
||||
|
||||
/** @brief Evaluate the coefficient in the element described by @a T at the
|
||||
point @a ip. */
|
||||
/** @note When this method is called, the caller must make sure that the
|
||||
|
||||
+18
-5
@@ -492,6 +492,8 @@ void VisItDataCollection::SaveRootFile()
|
||||
to_padded_string(cycle, pad_digits_cycle) +
|
||||
".mfem_root";
|
||||
std::ofstream root_file(root_name);
|
||||
MFEM_VERIFY(root_file.is_open(),
|
||||
"Failed to open ofstream " << root_name);
|
||||
root_file << GetVisItRootString();
|
||||
if (!root_file)
|
||||
{
|
||||
@@ -977,7 +979,10 @@ void ParaViewDataCollection::Save()
|
||||
// Save the local part of the mesh and grid functions fields to the local
|
||||
// VTU file. Also save coefficient fields.
|
||||
{
|
||||
std::ofstream os(vtu_prefix + GenerateVTUFileName("proc", myid));
|
||||
std::string os_str = vtu_prefix + GenerateVTUFileName("proc", myid);
|
||||
std::ofstream os(os_str);
|
||||
MFEM_VERIFY(os.is_open(),
|
||||
"Failed to open ofstream " << os_str);
|
||||
os.precision(precision);
|
||||
SaveDataVTU(os, levels_of_detail);
|
||||
}
|
||||
@@ -989,7 +994,10 @@ void ParaViewDataCollection::Save()
|
||||
"QuadratureFunction output is not supported for "
|
||||
"ParaViewDataCollection on domain boundary!");
|
||||
const std::string &field_name = qfield.first;
|
||||
std::ofstream os(vtu_prefix + GenerateVTUFileName(field_name, myid));
|
||||
std::string os_str = vtu_prefix + GenerateVTUFileName(field_name, myid);
|
||||
std::ofstream os(os_str);
|
||||
MFEM_VERIFY(os.is_open(),
|
||||
"Failed to open ofstream " << os_str);
|
||||
qfield.second->SaveVTU(os, pv_data_format, GetCompressionLevel(), field_name);
|
||||
}
|
||||
|
||||
@@ -1000,7 +1008,10 @@ void ParaViewDataCollection::Save()
|
||||
{
|
||||
// Create the main PVTU file
|
||||
{
|
||||
std::ofstream pvtu_out(vtu_prefix + GeneratePVTUFileName("data"));
|
||||
std::string os_str = vtu_prefix + GeneratePVTUFileName("data");
|
||||
std::ofstream pvtu_out(os_str);
|
||||
MFEM_VERIFY(pvtu_out.is_open(),
|
||||
"Failed to open ofstream " << os_str);
|
||||
WritePVTUHeader(pvtu_out);
|
||||
|
||||
// Grid function fields and coefficient fields
|
||||
@@ -1055,8 +1066,10 @@ void ParaViewDataCollection::Save()
|
||||
const std::string &q_field_name = q_field.first;
|
||||
std::string q_fname = GeneratePVTUPath() + "/"
|
||||
+ GeneratePVTUFileName(q_field_name);
|
||||
|
||||
std::ofstream pvtu_out(col_path + "/" + q_fname);
|
||||
std::string os_str = col_path + "/" + q_fname;
|
||||
std::ofstream pvtu_out(os_str);
|
||||
MFEM_VERIFY(pvtu_out.is_open(),
|
||||
"Failed to open ofstream " << os_str);
|
||||
WritePVTUHeader(pvtu_out);
|
||||
int vec_dim = q_field.second->GetVDim();
|
||||
pvtu_out << "<PPointData>\n";
|
||||
|
||||
@@ -0,0 +1,587 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
|
||||
// #include "fem/kernels.hpp"
|
||||
#include "fem/kernels3d.hpp"
|
||||
namespace ker = mfem::kernels::internal;
|
||||
namespace low = mfem::kernels::internal::low;
|
||||
#include "fem/kernel_dispatch.hpp"
|
||||
|
||||
// #include "linalg/kernels.hpp"
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
#undef NVTX_COLOR
|
||||
#define NVTX_COLOR ::nvtx::kOrchid
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1>` */
|
||||
template<typename T, int n1>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2>` */
|
||||
template<typename T, int n1, int n2>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3>` */
|
||||
template<typename T, int n1, int n2, int n3>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2, n3>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2, int n3>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2, n3>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3, n4>` */
|
||||
template<typename T, int n1, int n2, int n3, int n4>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2, n3, n4>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3, n4>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2, int n3, int n4>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2, n3, n4>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3, n4>*>(ptr));
|
||||
}
|
||||
|
||||
|
||||
template <std::size_t N>
|
||||
MFEM_HOST_DEVICE inline
|
||||
std::array<real_t*, N>
|
||||
load_field_e_ptr(const std::array<DeviceTensor<2>, N> &fields_e,
|
||||
const int e)
|
||||
{
|
||||
std::array<real_t*, N> f;
|
||||
for_constexpr<N>([&](auto i) { f[i] = &fields_e[i](0, e); });
|
||||
return f;
|
||||
}
|
||||
|
||||
namespace qf
|
||||
{
|
||||
|
||||
template <int T_Q1D,
|
||||
size_t num_args,
|
||||
typename reg_t,
|
||||
typename qfunc_t,
|
||||
typename args_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel(reg_t &res /*output*/,
|
||||
reg_t ®,
|
||||
const real_t *rd,
|
||||
const int qx, const int qy, const int qz,
|
||||
const qfunc_t &qfunc, args_ts &args)
|
||||
{
|
||||
if constexpr (num_args == 2) // PAApply
|
||||
{
|
||||
// ∇u
|
||||
tensor<real_t, 3> &arg_0 = get<0>(args);
|
||||
arg_0[0] = reg[qz][qy][qx][0];
|
||||
arg_0[1] = reg[qz][qy][qx][1];
|
||||
arg_0[2] = reg[qz][qy][qx][2];
|
||||
|
||||
// D (PA data)
|
||||
tensor<real_t, 3, 3> &arg_1 = get<1>(args);
|
||||
|
||||
if constexpr (T_Q1D > 0)
|
||||
{
|
||||
const auto *D = (const real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
for (int k = 0; k < 3; k++)
|
||||
{
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
arg_1[k][j] = D[qx][qy][qz][k][j];
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(false);
|
||||
// const auto D = Reshape(r2, 3, 3, Q1D, Q1D, Q1D);
|
||||
// for (int j = 0; j < 3; j++)
|
||||
// {
|
||||
// for (int k = 0; k < 3; k++)
|
||||
// {
|
||||
// arg_1[k][j] = D(j, k, qz, qy, qx);
|
||||
// }
|
||||
// }
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// MFApply comes here
|
||||
assert(false);
|
||||
// MFEM_ABORT("Only two arguments (∇u and D) are supported in apply_kernel for now");
|
||||
}
|
||||
|
||||
const auto r = get<0>(apply(qfunc, args));
|
||||
|
||||
if constexpr (decltype(r)::ndim == 1)
|
||||
{
|
||||
// process_qf_result_from_reg(r0, qx, qy, qz, r);
|
||||
as_tensor<real_t, 3>(&res[qz][qy][qx][0]) = r;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(false);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace qf
|
||||
|
||||
#define MFEM_D2Q_MAX_SIZE 4
|
||||
static MFEM_CONSTANT real_t Bi[MFEM_D2Q_MAX_SIZE][8*8], Bo[8*8];
|
||||
static MFEM_CONSTANT real_t Gi[MFEM_D2Q_MAX_SIZE][8*8], Go[8*8];
|
||||
|
||||
template<size_t num_fields,
|
||||
size_t num_inputs,
|
||||
size_t num_outputs,
|
||||
typename restriction_cb_t,
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_fop_t>
|
||||
class NewActionCallback
|
||||
{
|
||||
restriction_cb_t &restriction_cb;
|
||||
qfunc_t &qfunc;
|
||||
input_t &inputs;
|
||||
const std::array<size_t, num_inputs> &input_to_field;
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps;
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps;
|
||||
const int num_entities;
|
||||
const int test_vdim;
|
||||
const int num_test_dof;
|
||||
const int dimension;
|
||||
const ThreadBlocks &thread_blocks;
|
||||
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info;
|
||||
const Array<int> &attributes;
|
||||
const output_fop_t &output_fop;
|
||||
const Array<int> *elem_attributes;
|
||||
// refs
|
||||
std::vector<Vector> &fields_e;
|
||||
Vector &residual_e;
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose;
|
||||
// args
|
||||
std::vector<Vector> &solutions_l;
|
||||
const std::vector<Vector> ¶meters_l;
|
||||
Vector &residual_l;
|
||||
|
||||
public:
|
||||
NewActionCallback() = delete;
|
||||
|
||||
NewActionCallback(const bool use_kernels_specialization,
|
||||
restriction_cb_t &restriction_cb,
|
||||
qfunc_t &qfunc,
|
||||
input_t &inputs,
|
||||
const std::array<size_t, num_inputs> &input_to_field,
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
const int num_entities,
|
||||
const int test_vdim,
|
||||
const int num_test_dof,
|
||||
const int dimension,
|
||||
const ThreadBlocks &thread_blocks,
|
||||
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info,
|
||||
const Array<int> &attributes,
|
||||
const output_fop_t &output_fop,
|
||||
const Array<int> *elem_attributes,
|
||||
// refs
|
||||
std::vector<Vector> &fields_e,
|
||||
Vector &residual_e,
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
|
||||
// args
|
||||
std::vector<Vector> &solutions_l,
|
||||
const std::vector<Vector> ¶meters_l,
|
||||
Vector &residual_l):
|
||||
restriction_cb(restriction_cb),
|
||||
qfunc(qfunc),
|
||||
inputs(inputs),
|
||||
input_to_field(input_to_field),
|
||||
input_dtq_maps(input_dtq_maps),
|
||||
output_dtq_maps(output_dtq_maps),
|
||||
num_entities(num_entities),
|
||||
test_vdim(test_vdim),
|
||||
num_test_dof(num_test_dof),
|
||||
dimension(dimension),
|
||||
thread_blocks(thread_blocks),
|
||||
shmem_info(shmem_info),
|
||||
attributes(attributes),
|
||||
output_fop(output_fop),
|
||||
elem_attributes(elem_attributes),
|
||||
fields_e(fields_e),
|
||||
residual_e(residual_e),
|
||||
output_restriction_transpose(output_restriction_transpose),
|
||||
solutions_l(solutions_l),
|
||||
parameters_l(parameters_l),
|
||||
residual_l(residual_l)
|
||||
{
|
||||
if (!use_kernels_specialization) { return; }
|
||||
NewActionCallbackKernels::template Specialization<3>::Add(); // 1
|
||||
NewActionCallbackKernels::template Specialization<4>::Add(); // 2
|
||||
NewActionCallbackKernels::template Specialization<5>::Add(); // 3
|
||||
NewActionCallbackKernels::template Specialization<6>::Add(); // 4
|
||||
NewActionCallbackKernels::template Specialization<7>::Add(); // 5
|
||||
NewActionCallbackKernels::template Specialization<8>::Add(); // 6
|
||||
}
|
||||
|
||||
template<int T_Q1D = 0>
|
||||
static void action_callback_new(const int d1d,
|
||||
restriction_cb_t &restriction_cb,
|
||||
qfunc_t &qfunc,
|
||||
[[maybe_unused]] input_t &inputs,
|
||||
[[maybe_unused]] const std::array<size_t, num_inputs> &input_to_field,
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
[[maybe_unused]] const int dimension,
|
||||
const int num_entities,
|
||||
[[maybe_unused]] const int test_vdim,
|
||||
[[maybe_unused]] const int num_test_dof,
|
||||
const ThreadBlocks &thread_blocks,
|
||||
[[maybe_unused]] SharedMemoryInfo<num_fields, num_inputs, num_outputs>
|
||||
&shmem_info,
|
||||
[[maybe_unused]] const Array<int> &attributes,
|
||||
[[maybe_unused]] const output_fop_t &output_fop,
|
||||
[[maybe_unused]] const Array<int> *elem_attributes,
|
||||
// refs
|
||||
std::vector<Vector> &fields_e,
|
||||
Vector &residual_e,
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
|
||||
// args
|
||||
std::vector<Vector> &solutions_l,
|
||||
const std::vector<Vector> ¶meters_l,
|
||||
Vector &residual_l,
|
||||
// fallback arguments
|
||||
const int q1d)
|
||||
{
|
||||
NVTX_MARK_FUNCTION;
|
||||
assert(dimension == 3);
|
||||
static_assert(MFEM_D2Q_MAX_SIZE >= num_inputs, "MFEM_D2Q_MAX_SIZE error");
|
||||
|
||||
constexpr int DIM = 3;
|
||||
|
||||
[[maybe_unused]] static bool ini = (for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
const auto dtq = input_dtq_maps[i];
|
||||
{
|
||||
const auto [q, _, p] = dtq.B.GetShape();
|
||||
const auto B = (const real_t*)input_dtq_maps[i].B;
|
||||
dbg("Loading Bi[{}]: q={} p={}", i.value, q, p);
|
||||
if (B) { Gpu(MemcpyToSymbol)(Bi[i], B, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
{
|
||||
const auto [q, _, p] = dtq.G.GetShape();
|
||||
const auto G = (const real_t*)input_dtq_maps[i].G;
|
||||
if (G) { Gpu(MemcpyToSymbol)(Gi[i], G, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
if constexpr (i == 0) // output B
|
||||
{
|
||||
const auto dtq_o = output_dtq_maps[0];
|
||||
const auto [q, _, p] = dtq_o.B.GetShape();
|
||||
const auto B = (const real_t*)dtq_o.B;
|
||||
if (B) { Gpu(MemcpyToSymbol)(Bo, B, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
if constexpr (i == 0) // output G
|
||||
{
|
||||
const auto dtq_o = output_dtq_maps[0];
|
||||
const auto [q, _, p] = dtq_o.G.GetShape();
|
||||
const auto G = (const real_t*)dtq_o.G;
|
||||
if (G) { Gpu(MemcpyToSymbol)(Go, G, (p*q)*sizeof(real_t)); }
|
||||
dbg("Loaded B and G to constant memory");
|
||||
}
|
||||
}), true);
|
||||
|
||||
// types
|
||||
using qf_signature =
|
||||
typename create_function_signature<decltype(&qfunc_t::operator())>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
restriction_cb(solutions_l, parameters_l, fields_e);
|
||||
|
||||
NVTX_INI("res=0");
|
||||
residual_e = 0.0;
|
||||
NVTX_END("res=0");
|
||||
|
||||
// auto wrapped_fields_e =
|
||||
// wrap_fields(fields_e, shmem_info.field_sizes, num_entities);
|
||||
|
||||
const bool has_attr = attributes.Size() > 0;
|
||||
const auto d_attr = attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes->Read();
|
||||
|
||||
// const int vdim = input.vdim;
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
const int NE = num_entities;
|
||||
constexpr int VDIM = 1;
|
||||
|
||||
const auto XE = Reshape(fields_e[0].Read(), d1d, d1d, d1d, VDIM, NE);
|
||||
const real_t *dx_ptr = fields_e[1].Read();
|
||||
|
||||
auto YE = Reshape(residual_e.ReadWrite(), d1d, d1d, d1d, VDIM, NE);
|
||||
|
||||
const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
|
||||
const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
|
||||
|
||||
NVTX_INI("forall");
|
||||
dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
|
||||
{
|
||||
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
|
||||
MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
|
||||
MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
|
||||
// real_t (&sm0_ptr)[MQ1][MQ1][MQ1][3] = sm0;
|
||||
// real_t (&sm1_ptr)[MQ1][MQ1][MQ1][3] = sm1;
|
||||
|
||||
low::regs3d_t<DIM, MQ1> reg;
|
||||
const real_t *rd = dx_ptr;
|
||||
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
// real_t (&sB_ptr)[MD1][MQ1] = sB;
|
||||
// real_t (&sG_ptr)[MD1][MQ1] = sG;
|
||||
|
||||
// Interpolate
|
||||
// for_constexpr<num_inputs>(
|
||||
// [ D1D, Q1D, MQ1, e,
|
||||
// &input_dtq_maps,
|
||||
// &sm0_ptr, &sm1_ptr,
|
||||
// &sB = sB_ptr, &sG = sG_ptr,
|
||||
// &inputs,
|
||||
// // &fields_e_ptr,
|
||||
// ®, &rd,
|
||||
// &input_to_field ] (auto i)
|
||||
{
|
||||
// const auto input = get<0/*i*/>(inputs);
|
||||
// using field_operator_t = std::decay_t<decltype(input)>;
|
||||
|
||||
// if constexpr (is_gradient_fop<field_operator_t>::value) // Grad
|
||||
{
|
||||
// const int vdim = input.vdim;
|
||||
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
|
||||
// const auto XE = Reshape(field_e_r, D1D, D1D, D1D, vdim);
|
||||
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bi[i]);
|
||||
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Gi[i]);
|
||||
low::LoadMatrix(d1d, q1d, B, sB);
|
||||
low::LoadMatrix(d1d, q1d, G, sG);
|
||||
// for (int c = 0; c < vdim; c++)
|
||||
// constexpr int c = 0;
|
||||
{
|
||||
low::LoadDofs3d(e, d1d, XE, sm0);
|
||||
low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
|
||||
}
|
||||
}
|
||||
// else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
|
||||
{
|
||||
// db1("Identity");
|
||||
// rd = fields_e_ptr[input_to_field[i]];
|
||||
// rd = dx_ptr;
|
||||
}
|
||||
// else if constexpr (is_weight_fop<field_operator_t>::value) // Weight
|
||||
// {
|
||||
// dbg("Weight");
|
||||
// rw = fields_e_ptr[input_to_field[i]]; // 🔥
|
||||
// }
|
||||
// else
|
||||
{
|
||||
// MFApply comes here
|
||||
// assert(false);
|
||||
// MFEM_ABORT("Only Grad and Identity field operators are supported");
|
||||
}
|
||||
}//); // for_constexpr<num_inputs>
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
#if 0
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
qf::apply_kernel<T_Q1D, num_inputs>
|
||||
(reg, reg, rd, qx, qy, qz, qfunc, qf_args);
|
||||
#elif 0
|
||||
real_t v[3], u[3] = { reg[qz][qy][qx][0],
|
||||
reg[qz][qy][qx][1],
|
||||
reg[qz][qy][qx][2]
|
||||
};
|
||||
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
kernels::Mult(3, 3, &D[qx][qy][qz][0][0], u, v);
|
||||
reg[qz][qy][qx][0] = v[0];
|
||||
reg[qz][qy][qx][1] = v[1];
|
||||
reg[qz][qy][qx][2] = v[2];
|
||||
#elif 0
|
||||
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
const auto args = decay_tuple<qf_param_ts>
|
||||
{
|
||||
{{ reg[qz][qy][qx][0], reg[qz][qy][qx][1], reg[qz][qy][qx][2] }},
|
||||
{{
|
||||
{{ D[qx][qy][qz][0][0], D[qx][qy][qz][0][1], D[qx][qy][qz][0][2] }},
|
||||
{{ D[qx][qy][qz][1][0], D[qx][qy][qz][1][1], D[qx][qy][qz][1][2] }},
|
||||
{{ D[qx][qy][qz][2][0], D[qx][qy][qz][2][1], D[qx][qy][qz][2][2] }}
|
||||
}
|
||||
}
|
||||
};
|
||||
const auto r = get<0>(apply(qfunc, args));
|
||||
reg[qz][qy][qx][0] = r[0];
|
||||
reg[qz][qy][qx][1] = r[1];
|
||||
reg[qz][qy][qx][2] = r[2];
|
||||
#elif 0
|
||||
auto u = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
const auto *d = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
auto D = as_tensor<real_t, 3, 3>(&d[qx][qy][qz][0][0]);
|
||||
auto r = D * u;
|
||||
reg[qz][qy][qx][0] = r[0];
|
||||
reg[qz][qy][qx][1] = r[1];
|
||||
reg[qz][qy][qx][2] = r[2];
|
||||
#else
|
||||
auto args = decay_tuple<qf_param_ts> {};
|
||||
get<0>(args) = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
if constexpr (T_Q1D > 0)
|
||||
{
|
||||
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
|
||||
}
|
||||
else
|
||||
{
|
||||
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
|
||||
}
|
||||
auto r = get<0>(apply(qfunc, args));
|
||||
if constexpr (decltype(r)::ndim == 1)
|
||||
{
|
||||
as_tensor<real_t, 3>(®[qz][qy][qx][0]) = r;
|
||||
}
|
||||
else { static_assert(false); }
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// Integrate
|
||||
// if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
|
||||
{
|
||||
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
|
||||
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
|
||||
low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
|
||||
low::WriteDofs3d(d1d, 0, e, reg, YE);
|
||||
}
|
||||
},
|
||||
num_entities, thread_blocks, 0, nullptr);
|
||||
NVTX_END("forall");
|
||||
|
||||
NVTX_INI("out^T");
|
||||
output_restriction_transpose(residual_e, residual_l);
|
||||
NVTX_END("out^T");
|
||||
}
|
||||
|
||||
using NewActionKernelType = decltype(&NewActionCallback::action_callback_new<>);
|
||||
MFEM_REGISTER_KERNELS(NewActionCallbackKernels, NewActionKernelType, (int));
|
||||
|
||||
void Apply(const int d1d, const int q1d)
|
||||
{
|
||||
db1();
|
||||
NewActionCallbackKernels::Run(q1d,
|
||||
// args
|
||||
d1d,
|
||||
restriction_cb,
|
||||
qfunc,
|
||||
inputs,
|
||||
input_to_field,
|
||||
input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
dimension,
|
||||
num_entities,
|
||||
test_vdim,
|
||||
num_test_dof,
|
||||
thread_blocks,
|
||||
shmem_info,
|
||||
attributes,
|
||||
output_fop,
|
||||
elem_attributes,
|
||||
fields_e,
|
||||
residual_e,
|
||||
output_restriction_transpose,
|
||||
solutions_l,
|
||||
parameters_l,
|
||||
residual_l,
|
||||
// fallback arguments
|
||||
q1d);
|
||||
}
|
||||
};
|
||||
|
||||
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
|
||||
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
|
||||
template<int T_Q1D>
|
||||
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
|
||||
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Kernel()
|
||||
{
|
||||
return action_callback_new<T_Q1D>;
|
||||
}
|
||||
|
||||
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
|
||||
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
|
||||
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
|
||||
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Fallback
|
||||
(int q1d)
|
||||
{
|
||||
dbg("\x1b[33mFallback q1d:{}", q1d);
|
||||
// MFEM_ABORT("No kernel for q1d=" << q1d);
|
||||
// return nullptr;
|
||||
return action_callback_new<>;
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -0,0 +1,111 @@
|
||||
#pragma once
|
||||
|
||||
#include "../util.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace GlobalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct Action
|
||||
{
|
||||
Action(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(std::move(qfunc)),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
|
||||
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
|
||||
|
||||
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
|
||||
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
gnqp = nqp * ctx.nentities;
|
||||
|
||||
xq_offsets.SetSize(ninputs + 1);
|
||||
xq_offsets[0] = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(inputs);
|
||||
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
|
||||
});
|
||||
xq_offsets.PartialSum();
|
||||
xq.Update(xq_offsets);
|
||||
|
||||
yq_offsets.SetSize(noutputs + 1);
|
||||
yq_offsets[0] = 0;
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
const auto output = get<i>(outputs);
|
||||
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
|
||||
});
|
||||
yq_offsets.PartialSum();
|
||||
yq.Update(yq_offsets);
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
|
||||
// E -> Q
|
||||
interpolate(input_to_infd, input_bases, xe, xq);
|
||||
|
||||
// Q -> Q
|
||||
static_assert(
|
||||
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
|
||||
"qfunc signature not supported by default backend Action");
|
||||
|
||||
detail::call_qfunc(
|
||||
qfunc, xq, yq, gnqp, input_qlayouts, output_qlayouts,
|
||||
std::make_index_sequence<ninputs> {},
|
||||
std::make_index_sequence<noutputs> {});
|
||||
|
||||
// Q -> E
|
||||
integrate(output_to_outfd, output_bases, yq, ye);
|
||||
}
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<FieldBasis, ninputs> input_bases;
|
||||
std::array<FieldBasis, noutputs> output_bases;
|
||||
|
||||
std::array<std::vector<int>, ninputs> input_qlayouts;
|
||||
std::array<std::vector<int>, noutputs> output_qlayouts;
|
||||
|
||||
int gnqp = 0;
|
||||
Array<int> xq_offsets, yq_offsets;
|
||||
mutable BlockVector xq, yq;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
#pragma once
|
||||
|
||||
#include "../fem/quadinterpolator.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
#include "../util.hpp"
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace GlobalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct DerivativeActionEnzyme
|
||||
{
|
||||
DerivativeActionEnzyme(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t &qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(qfunc),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
|
||||
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
|
||||
|
||||
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
|
||||
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
gnqp = nqp * ctx.nentities;
|
||||
|
||||
xq_offsets.SetSize(ninputs + 1);
|
||||
xq_offsets[0] = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(inputs);
|
||||
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
|
||||
});
|
||||
xq_offsets.PartialSum();
|
||||
xq.Update(xq_offsets);
|
||||
|
||||
yq_offsets.SetSize(noutputs + 1);
|
||||
yq_offsets[0] = 0;
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
const auto output = get<i>(outputs);
|
||||
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
|
||||
});
|
||||
yq_offsets.PartialSum();
|
||||
yq.Update(yq_offsets);
|
||||
|
||||
// For each dependent input in the dependency map we create a shadow
|
||||
// memory variable at the quadrature point level.
|
||||
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
|
||||
shadow_xq_offsets.SetSize(ninputs + 1);
|
||||
shadow_xq_offsets = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
if (activity_map[i])
|
||||
{
|
||||
shadow_xq_offsets[i + 1] =
|
||||
xq_offsets[i + 1] - xq_offsets[i];;
|
||||
}
|
||||
});
|
||||
shadow_xq_offsets.PartialSum();
|
||||
shadow_xq.Update(shadow_xq_offsets);
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
const Vector *de,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
// E -> Q
|
||||
interpolate(input_to_infd, input_bases, xe, xq);
|
||||
|
||||
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
|
||||
interpolate(input_to_infd, input_bases, xe, shadow_xq, activity_map);
|
||||
|
||||
// Q -> Q
|
||||
static_assert(
|
||||
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
|
||||
"qfunc signature not supported by default backend Action");
|
||||
|
||||
detail::enzyme_fwddiff<derivative_id, qfunc_t, inputs_t, outputs_t>(
|
||||
qfunc, xq, shadow_xq, yq, gnqp, input_qlayouts, output_qlayouts,
|
||||
std::make_index_sequence<ninputs> {},
|
||||
std::make_index_sequence<noutputs> {});
|
||||
|
||||
// Q -> E
|
||||
integrate(output_to_outfd, output_bases, yq, ye);
|
||||
}
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t &qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<FieldBasis, ninputs> input_bases;
|
||||
std::array<FieldBasis, noutputs> output_bases;
|
||||
|
||||
std::array<std::vector<int>, ninputs> input_qlayouts;
|
||||
std::array<std::vector<int>, noutputs> output_qlayouts;
|
||||
|
||||
int gnqp = 0;
|
||||
Array<int> xq_offsets, shadow_xq_offsets, yq_offsets;
|
||||
mutable BlockVector xq, shadow_xq, yq;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
#pragma once
|
||||
|
||||
#include "action.hpp"
|
||||
#include "derivative_action_enzyme.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct GlobalQFBackend
|
||||
{
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return GlobalQFImpl::Action(ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeDerivativeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return GlobalQFImpl::DerivativeActionEnzyme<
|
||||
derivative_id, qfunc_t, inputs_t, outputs_t>(
|
||||
ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
#pragma once
|
||||
|
||||
#include "../util.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace LocalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct Action
|
||||
{
|
||||
Action(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(std::move(qfunc)),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
|
||||
// Initialize DofToQuad maps for inputs
|
||||
for_constexpr<ninputs>([&](auto i)
|
||||
{
|
||||
const auto &fd = ctx.infds[input_to_infd[i]];
|
||||
std::visit([&](auto* space_ptr)
|
||||
{
|
||||
using T = std::decay_t<decltype(*space_ptr)>;
|
||||
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
|
||||
std::is_same_v<T, ParFiniteElementSpace>)
|
||||
{
|
||||
const auto *fe = space_ptr->GetTypicalFE();
|
||||
input_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
|
||||
}
|
||||
}, fd.data);
|
||||
});
|
||||
|
||||
// Initialize DofToQuad maps for outputs
|
||||
for_constexpr<noutputs>([&](auto i)
|
||||
{
|
||||
const auto &fd = ctx.outfds[output_to_outfd[i]];
|
||||
std::visit([&](auto* space_ptr)
|
||||
{
|
||||
using T = std::decay_t<decltype(*space_ptr)>;
|
||||
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
|
||||
std::is_same_v<T, ParFiniteElementSpace>)
|
||||
{
|
||||
const auto *fe = space_ptr->GetTypicalFE();
|
||||
output_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
|
||||
}
|
||||
}, fd.data);
|
||||
});
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
|
||||
// input_dtq_maps
|
||||
|
||||
// const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
|
||||
// const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
|
||||
|
||||
// dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
|
||||
// {
|
||||
// if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
|
||||
// MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
|
||||
// MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
|
||||
|
||||
// low::regs3d_t<DIM, MQ1> reg;
|
||||
// const real_t *rd = dx_ptr;
|
||||
|
||||
// MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
// {
|
||||
// low::LoadMatrix(d1d, q1d, B, sB);
|
||||
// low::LoadMatrix(d1d, q1d, G, sG);
|
||||
// {
|
||||
// low::LoadDofs3d(e, d1d, XE, sm0);
|
||||
// low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
|
||||
// }
|
||||
// }
|
||||
// // else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
|
||||
// {
|
||||
// // db1("Identity");
|
||||
// // rd = fields_e_ptr[input_to_field[i]];
|
||||
// // rd = dx_ptr;
|
||||
// }
|
||||
// }
|
||||
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
// {
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
// {
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
// {
|
||||
|
||||
// auto args = decay_tuple<qf_param_ts> {};
|
||||
// get<0>(args) = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
// if constexpr (T_Q1D > 0)
|
||||
// {
|
||||
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
|
||||
// }
|
||||
// else
|
||||
// {
|
||||
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
|
||||
// }
|
||||
// auto r = get<0>(apply(qfunc, args));
|
||||
// if constexpr (decltype(r)::ndim == 1)
|
||||
// {
|
||||
// as_tensor<real_t, 3>(®[qz][qy][qx][0]) = r;
|
||||
// }
|
||||
// else { static_assert(false); }
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
// MFEM_SYNC_THREAD;
|
||||
// // Integrate
|
||||
// // if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
|
||||
// {
|
||||
// // const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
|
||||
// // const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
|
||||
// low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
|
||||
// low::WriteDofs3d(d1d, 0, e, reg, YE);
|
||||
// }
|
||||
// },
|
||||
// num_entities, thread_blocks, 0, nullptr);
|
||||
}
|
||||
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<const DofToQuad*, ninputs> input_dtq_maps;
|
||||
std::array<const DofToQuad*, noutputs> output_dtq_maps;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
|
||||
#include "../../integrator_ctx.hpp"
|
||||
#include "action.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct LocalQFBackend
|
||||
{
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return LocalQFImpl::Action(ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeDerivativeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
MFEM_ABORT("LocalQFBackend does not support derivative actions.");
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,659 @@
|
||||
#pragma once
|
||||
|
||||
#include "../fem/quadinterpolator.hpp"
|
||||
#include "../util.hpp"
|
||||
#include "general/enzyme.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
template <size_t N, size_t... Is>
|
||||
constexpr std::array<bool, N> all_true_impl(std::index_sequence<Is...>)
|
||||
{
|
||||
return {{((void)Is, true)...}};
|
||||
}
|
||||
|
||||
template <size_t N>
|
||||
constexpr std::array<bool, N> all_true()
|
||||
{
|
||||
return all_true_impl<N>(std::make_index_sequence<N> {});
|
||||
}
|
||||
|
||||
struct FieldBasis
|
||||
{
|
||||
// E-vector -> Q-vector
|
||||
std::function<void(const Vector &, Vector &)> forward;
|
||||
|
||||
// Q-vector -> E-vector
|
||||
std::function<void(const Vector &, Vector &)> transpose;
|
||||
};
|
||||
|
||||
inline FieldBasis FromQI(const QuadratureInterpolator *qi,
|
||||
QuadratureInterpolator::EvalFlags mode)
|
||||
{
|
||||
return
|
||||
{
|
||||
[qi, mode](const Vector &xe, Vector &xq)
|
||||
{
|
||||
qi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
if (mode == QuadratureInterpolator::VALUES)
|
||||
{
|
||||
qi->Values(xe, xq);
|
||||
}
|
||||
else
|
||||
{
|
||||
qi->Derivatives(xe, xq);
|
||||
}
|
||||
},
|
||||
[qi, mode](const Vector &yq, Vector &ye)
|
||||
{
|
||||
Vector empty;
|
||||
qi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
if (mode == QuadratureInterpolator::VALUES)
|
||||
{
|
||||
qi->AddMultTranspose(QuadratureInterpolator::VALUES, yq, empty, ye);
|
||||
}
|
||||
else
|
||||
{
|
||||
qi->AddMultTranspose(QuadratureInterpolator::DERIVATIVES, empty, yq, ye);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// QuadratureFunction identity copy
|
||||
inline FieldBasis FromQF()
|
||||
{
|
||||
return
|
||||
{
|
||||
[](const Vector &xe, Vector &xq) { xq = xe; },
|
||||
[](const Vector &yq, Vector &ye) { ye = yq; }
|
||||
};
|
||||
}
|
||||
|
||||
// User-defined parameter space B
|
||||
inline FieldBasis FromPS(const Operator *B, const Operator *Bt)
|
||||
{
|
||||
return
|
||||
{
|
||||
[B](const Vector &xe, Vector &xq) { B->Mult(xe, xq); },
|
||||
[Bt](const Vector &yq, Vector &ye) { Bt->Mult(yq, ye); }
|
||||
};
|
||||
}
|
||||
|
||||
inline FieldBasis FieldBasisFromWeight(const IntegrationRule &ir)
|
||||
{
|
||||
return
|
||||
{
|
||||
[&ir](const Vector &, Vector &xq)
|
||||
{
|
||||
const int nqp = ir.GetNPoints();
|
||||
MFEM_ASSERT(xq.Size() % nqp == 0, "weight block has unexpected size");
|
||||
|
||||
const int ne = xq.Size() / nqp;
|
||||
const real_t *wref = ir.GetWeights().Read();
|
||||
|
||||
for (int e = 0; e < ne; e++)
|
||||
{
|
||||
std::memcpy(xq.GetData() + e*nqp, wref, nqp*sizeof(real_t));
|
||||
}
|
||||
},
|
||||
[](const Vector &, Vector &) {}
|
||||
};
|
||||
}
|
||||
|
||||
inline const FieldBasis GetFieldBasis(const FieldDescriptor &f,
|
||||
const IntegrationRule &ir,
|
||||
QuadratureInterpolator::EvalFlags mode)
|
||||
{
|
||||
return std::visit([&ir, &mode](auto && arg) -> FieldBasis
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
|
||||
if constexpr (std::is_same_v<T, const FiniteElementSpace *>)
|
||||
{
|
||||
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParFiniteElementSpace *>)
|
||||
{
|
||||
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return FromQF();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return FromPS(arg->GetB(), arg->GetBt());
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const IntegrationRule *>)
|
||||
{
|
||||
return FieldBasis{};
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>, "internal error");
|
||||
}
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void create_fieldbases(
|
||||
fops_t &fops,
|
||||
const std::array<size_t, nfops> &fop_to_fd,
|
||||
const std::vector<FieldDescriptor> &fds,
|
||||
const IntegrationRule &ir,
|
||||
std::array<FieldBasis, nfops> &bases)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
const auto fop = get<i>(fops);
|
||||
using fop_t = std::decay_t<decltype(fop)>;
|
||||
|
||||
const auto fd = fds[fop_to_fd[i]];
|
||||
|
||||
constexpr QuadratureInterpolator::EvalFlags dummy_mode =
|
||||
QuadratureInterpolator::VALUES;
|
||||
if constexpr (is_identity_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, dummy_mode);
|
||||
}
|
||||
else if constexpr (is_weight_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = FieldBasisFromWeight(ir);
|
||||
}
|
||||
else if constexpr (is_value_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::VALUES);
|
||||
}
|
||||
else if constexpr (is_gradient_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::DERIVATIVES);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void check_consistency(
|
||||
fops_t &fops,
|
||||
const std::array<size_t, nfops> &fop_to_fd,
|
||||
const std::vector<FieldDescriptor> &fields)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(fops);
|
||||
using input_t = std::decay_t<decltype(input)>;
|
||||
|
||||
const auto fd = fields[fop_to_fd[i]];
|
||||
|
||||
if constexpr (is_identity_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const QuadratureFunction *>(fd.data),
|
||||
"Identity FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
else if constexpr (is_weight_fop<input_t>::value)
|
||||
{
|
||||
}
|
||||
else if constexpr (is_value_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParameterSpace *>(fd.data),
|
||||
"Value FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
else if constexpr (is_gradient_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data),
|
||||
"Value FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t ninputs>
|
||||
void interpolate(
|
||||
const std::array<size_t, ninputs> &input_to_infd,
|
||||
const std::array<FieldBasis, ninputs> &input_bases,
|
||||
const std::vector<Vector *> &xe,
|
||||
BlockVector &xq,
|
||||
const std::array<bool, ninputs> &conditional = all_true<ninputs>())
|
||||
{
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
if (!conditional.empty() && !conditional[i]) { return; }
|
||||
|
||||
input_bases[i].forward(*xe[input_to_infd[i]], xq.GetBlock(i));
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t noutputs>
|
||||
void integrate(
|
||||
const std::array<size_t, noutputs> &output_to_outfd,
|
||||
const std::array<FieldBasis, noutputs> &output_bases,
|
||||
const BlockVector &yq,
|
||||
std::vector<Vector *> &ye)
|
||||
{
|
||||
for (auto v : ye) { *v = 0.0; }
|
||||
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
output_bases[i].transpose(yq.GetBlock(i), *ye[output_to_outfd[i]]);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
namespace detail
|
||||
{
|
||||
|
||||
template <typename T>
|
||||
struct is_tensor_array : std::false_type {};
|
||||
|
||||
template <typename scalar_t, int... Dims>
|
||||
struct is_tensor_array<tensor_array<scalar_t, Dims...>> : std::true_type {};
|
||||
|
||||
template <typename T>
|
||||
struct is_tensor_array_mut : std::false_type {};
|
||||
|
||||
template <typename scalar_t, int... Dims>
|
||||
struct is_tensor_array_mut<tensor_array<scalar_t, Dims...>> :
|
||||
std::bool_constant<!std::is_const_v<scalar_t>> {};
|
||||
|
||||
|
||||
template <typename ndarray_t>
|
||||
inline void set_layout_default(ndarray_t &a)
|
||||
{
|
||||
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
|
||||
|
||||
constexpr std::size_t nd = ndarray_t::rank();
|
||||
constexpr std::size_t td = ndarray_t::tensor_rank();
|
||||
std::array<std::size_t, nd + td> perm{};
|
||||
|
||||
for (std::size_t i = 0; i < td; i++) { perm[i] = nd + i; }
|
||||
for (std::size_t i = 0; i < nd; i++) { perm[td + i] = i; }
|
||||
|
||||
a.set_layout(perm);
|
||||
}
|
||||
|
||||
template <typename ndarray_t>
|
||||
inline void set_layout(ndarray_t& a, const std::vector<int>& layout)
|
||||
{
|
||||
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
|
||||
|
||||
constexpr std::size_t nd = ndarray_t::rank();
|
||||
constexpr std::size_t td = ndarray_t::tensor_rank();
|
||||
constexpr std::size_t N = nd + td;
|
||||
|
||||
// missing means default
|
||||
if (layout.empty()) { set_layout_default(a); return; }
|
||||
|
||||
MFEM_VERIFY(layout.size() == N,
|
||||
"layout size mismatch: expected " << N << " got " << layout.size());
|
||||
|
||||
// TODO: make a version of set_layout that takes `std::vector<int>`
|
||||
std::array<std::size_t, N> perm{};
|
||||
for (std::size_t i = 0; i < N; i++)
|
||||
{
|
||||
MFEM_VERIFY(layout[i] >= 0, "layout index must be >=0");
|
||||
perm[i] = static_cast<std::size_t>(layout[i]);
|
||||
}
|
||||
|
||||
a.set_layout(perm);
|
||||
}
|
||||
|
||||
/// Primary template: intentionally undefined — gives a clear error for unsupported types.
|
||||
template <typename T>
|
||||
struct tensor_array_traits;
|
||||
|
||||
/// Matches tensor<scalar_t, sizes...>
|
||||
template <typename scalar_t, int... sizes>
|
||||
struct tensor_array_traits<tensor<scalar_t, sizes...>>
|
||||
{
|
||||
using scalar_type = scalar_t;
|
||||
template <std::size_t ndims>
|
||||
using array_type = tensor_ndarray<scalar_t, ndims, sizes...>;
|
||||
};
|
||||
|
||||
/// Matches tensor_ndarray<scalar_t, ndims, tensor_sizes...>
|
||||
template <typename scalar_t, int ndims, int... tensor_sizes>
|
||||
struct tensor_array_traits<tensor_ndarray<scalar_t, ndims, tensor_sizes...>>
|
||||
{
|
||||
using scalar_type = scalar_t;
|
||||
template <std::size_t N>
|
||||
using array_type = tensor_ndarray<scalar_t, N, tensor_sizes...>;
|
||||
};
|
||||
|
||||
/// Entry point: explicit tensor type T as template argument.
|
||||
template <typename T, typename ptr_scalar_t, typename... dyn_sizes_t>
|
||||
decltype(auto) make_tensor_array(ptr_scalar_t *ptr,
|
||||
const std::vector<int>* layout,
|
||||
dyn_sizes_t... dynamic_sizes)
|
||||
{
|
||||
using traits = tensor_array_traits<T>;
|
||||
using array_t = typename traits::template array_type<sizeof...(dynamic_sizes)>;
|
||||
auto a = array_t(ptr, {std::size_t(dynamic_sizes)...});
|
||||
if (layout) { set_layout(a, *layout); }
|
||||
else { set_layout_default(a); }
|
||||
return a;
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename inputs_t, typename outputs_t>
|
||||
struct supports_tensor_array_qfunc
|
||||
{
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
static constexpr int ninputs = tuple_size<inputs_t>::value;
|
||||
static constexpr int noutputs = tuple_size<outputs_t>::value;
|
||||
static constexpr int nparams = tuple_size<qf_param_ts>::value;
|
||||
|
||||
template <std::size_t... Is>
|
||||
static constexpr bool InputsOk(std::index_sequence<Is...>)
|
||||
{
|
||||
return (is_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>::value && ...);
|
||||
}
|
||||
|
||||
template <std::size_t... Is>
|
||||
static constexpr bool OutputsOk(std::index_sequence<Is...>)
|
||||
{
|
||||
return (is_tensor_array_mut<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Is, qf_param_ts>::type>>>::value && ...);
|
||||
}
|
||||
|
||||
static constexpr bool value =
|
||||
(nparams == ninputs + noutputs) &&
|
||||
InputsOk(std::make_index_sequence<ninputs> {}) &&
|
||||
OutputsOk(std::make_index_sequence<noutputs> {});
|
||||
};
|
||||
|
||||
template <typename qfunc_t, std::size_t... Is, std::size_t... Os>
|
||||
inline void call_qfunc(
|
||||
const qfunc_t &qfunc,
|
||||
const BlockVector &xq,
|
||||
BlockVector &yq,
|
||||
int gnqp,
|
||||
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
|
||||
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
|
||||
std::index_sequence<Is...>,
|
||||
std::index_sequence<Os...>)
|
||||
{
|
||||
constexpr std::size_t ninputs = sizeof...(Is);
|
||||
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
auto inputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
auto outputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
std::apply([&](auto&&... args)
|
||||
{
|
||||
qfunc(args...);
|
||||
}, std::tuple_cat(inputs, outputs));
|
||||
}
|
||||
|
||||
template <typename func_t, typename... arg_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto qfunction_wrapper(const func_t &f, arg_ts...args)
|
||||
{
|
||||
return f(args...);
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, std::size_t I, typename Tuple, std::size_t... Is>
|
||||
constexpr std::array<bool, sizeof...(Is)>
|
||||
make_activity_array(std::index_sequence<Is...>)
|
||||
{
|
||||
return { (std::decay_t<typename tuple_element<Is, Tuple>::type>::GetFieldId() == derivative_id)... };
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, typename inputs_t, std::size_t... Is>
|
||||
constexpr auto make_activity_map_impl(std::index_sequence<Is...>)
|
||||
{
|
||||
constexpr std::size_t N = sizeof...(Is);
|
||||
|
||||
if constexpr (N == 0)
|
||||
return std::array<bool, 0> {};
|
||||
|
||||
return make_activity_array<derivative_id, 0, inputs_t>
|
||||
(std::make_index_sequence<N> {});
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, typename inputs_t>
|
||||
constexpr auto make_activity_map(inputs_t)
|
||||
{
|
||||
return make_activity_map_impl<derivative_id, inputs_t>(
|
||||
std::make_index_sequence<tuple_size<inputs_t>::value> {});
|
||||
}
|
||||
|
||||
namespace enzyme_detail
|
||||
{
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t, typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
do_enzyme_call(AccArgs... acc)
|
||||
{
|
||||
__enzyme_fwddiff<qf_return_t>(wrapper_fn, acc...);
|
||||
}
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t,
|
||||
size_t CurO, size_t NO,
|
||||
typename primals_t, typename derivs_t,
|
||||
typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
process_outputs(primals_t &primals, derivs_t &derivs, AccArgs... acc)
|
||||
{
|
||||
if constexpr (CurO == NO)
|
||||
{
|
||||
do_enzyme_call<wrapper_fn, qf_return_t>(acc...);
|
||||
}
|
||||
else
|
||||
{
|
||||
process_outputs<wrapper_fn, qf_return_t, CurO + 1, NO>(
|
||||
primals, derivs,
|
||||
acc...,
|
||||
enzyme_dupnoneed,
|
||||
&std::get<CurO>(primals),
|
||||
&std::get<CurO>(derivs));
|
||||
}
|
||||
}
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t,
|
||||
size_t CurI, size_t NI, bool... ActivityMap,
|
||||
typename inputs_t, typename shadows_t,
|
||||
typename primals_t, typename derivs_t,
|
||||
typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
process_inputs(inputs_t &inputs, shadows_t &shadows,
|
||||
primals_t &primals, derivs_t &derivs,
|
||||
AccArgs... acc)
|
||||
{
|
||||
if constexpr (CurI == NI)
|
||||
{
|
||||
constexpr size_t NO = std::tuple_size_v<primals_t>;
|
||||
process_outputs<wrapper_fn, qf_return_t, 0, NO>(
|
||||
primals, derivs, acc...);
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr bool active =
|
||||
std::array<bool, sizeof...(ActivityMap)> {ActivityMap...} [CurI];
|
||||
|
||||
if constexpr (active)
|
||||
{
|
||||
std::cout << "Input[" << CurI << "]: ACTIVE (enzyme_dup)\n"
|
||||
<< " primal ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n"
|
||||
<< " shadow ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(shadows))>() << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
std::cout << "Input[" << CurI << "]: INACTIVE (enzyme_const)\n"
|
||||
<< " primal ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n";
|
||||
}
|
||||
|
||||
if constexpr (active)
|
||||
{
|
||||
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
|
||||
inputs, shadows, primals, derivs,
|
||||
acc...,
|
||||
enzyme_dup,
|
||||
&std::get<CurI>(inputs),
|
||||
&std::get<CurI>(shadows));
|
||||
}
|
||||
else
|
||||
{
|
||||
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
|
||||
inputs, shadows, primals, derivs,
|
||||
acc...,
|
||||
enzyme_const,
|
||||
&std::get<CurI>(inputs));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace enzyme_detail
|
||||
|
||||
template <size_t derivative_id, typename qfunc_t, typename inputs_t, typename outputs_t,
|
||||
std::size_t... Is, std::size_t... Os>
|
||||
inline void enzyme_fwddiff(
|
||||
qfunc_t &qfunc,
|
||||
const BlockVector &xq,
|
||||
const BlockVector &shadow_xq,
|
||||
BlockVector &yq,
|
||||
const int &gnqp,
|
||||
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
|
||||
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
|
||||
std::index_sequence<Is...>,
|
||||
std::index_sequence<Os...>)
|
||||
{
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
constexpr std::size_t ninputs = sizeof...(Is);
|
||||
constexpr std::size_t noutputs = sizeof...(Os);
|
||||
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
using qf_return_t = typename qf_signature::return_t;
|
||||
|
||||
constexpr auto activity_map = make_activity_map<derivative_id>(inputs_t{});
|
||||
static_assert(activity_map.size() == ninputs, "activity map size mismatch");
|
||||
|
||||
std::cout << "activity_map: ";
|
||||
for (const auto &v : activity_map)
|
||||
{
|
||||
std::cout << v << " ";
|
||||
}
|
||||
std::cout << "\n";
|
||||
|
||||
auto inputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
auto shadows = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
shadow_xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
std::array<Vector, noutputs> primal_storage;
|
||||
((primal_storage[Os].SetSize(yq.GetBlock(Os).Size())), ...);
|
||||
|
||||
auto primals_out = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
primal_storage[Os].ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
auto derivs_out = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
using wrapper_fn_t = qf_return_t (*)(
|
||||
const qfunc_t &,
|
||||
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
|
||||
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...);
|
||||
|
||||
constexpr wrapper_fn_t wrapper_fn =
|
||||
qfunction_wrapper<qfunc_t,
|
||||
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
|
||||
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...>;
|
||||
|
||||
// wrapper_fn travels as a non-type template parameter throughout without
|
||||
// being stored.
|
||||
enzyme_detail::process_inputs<
|
||||
wrapper_fn,
|
||||
qf_return_t,
|
||||
0,
|
||||
ninputs,
|
||||
activity_map[Is]...
|
||||
>(inputs, shadows,
|
||||
primals_out, derivs_out,
|
||||
enzyme_const, &qfunc // seed: qfunc is always inactive
|
||||
);
|
||||
|
||||
#else
|
||||
MFEM_ABORT("enzyme_fwddiff requires MFEM_USE_ENZYME");
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
// Create quadrature function fop to fields map
|
||||
template <typename fops_t, size_t N = tuple_size<fops_t>::value, size_t M>
|
||||
void create_fop_to_fd(const fops_t &fops,
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
std::array<size_t, M> &fop_to_fd)
|
||||
{
|
||||
static_assert(N == M, "sizes must match");
|
||||
constexpr_for<0, N>([&](auto i)
|
||||
{
|
||||
const auto fop = get<i>(fops);
|
||||
fop_to_fd[i] = std::numeric_limits<size_t>::max();
|
||||
for (size_t j = 0; j < fields.size(); j++)
|
||||
{
|
||||
// TODO: output.GetFieldId() should probably store/return size_t
|
||||
if (static_cast<int>(fields[j].id) == fop.GetFieldId())
|
||||
{
|
||||
fop_to_fd[i] = j;
|
||||
}
|
||||
}
|
||||
// Handle Weight type. There is no FieldDescriptor for the weight.
|
||||
// TODO: Create weight descriptor for the weight for internal use?
|
||||
// TODO: this is a hack...
|
||||
if (is_weight_fop<std::remove_cv_t<decltype(fop)>>::value)
|
||||
{
|
||||
fop_to_fd[i] = 0;
|
||||
}
|
||||
else if (fop_to_fd[i] == std::numeric_limits<size_t>::max())
|
||||
{
|
||||
MFEM_ABORT("not found");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void create_qlayouts(const fops_t &fops,
|
||||
const std::unordered_map<std::type_index, std::vector<int>> &a,
|
||||
std::array<std::vector<int>, nfops> &b)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
using fop_t =
|
||||
std::remove_cv_t<std::remove_reference_t<decltype(get<i>(fops))>>;
|
||||
auto it = a.find(std::type_index(typeid(fop_t)));
|
||||
if (it != a.end()) { b[i] = it->second; }
|
||||
else { b[i].clear(); }
|
||||
});
|
||||
}
|
||||
|
||||
}
|
||||
+96
-21
@@ -11,44 +11,119 @@
|
||||
|
||||
#include "doperator.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
using namespace mfem;
|
||||
using namespace mfem::future;
|
||||
|
||||
void DifferentiableOperator::SetParameters(std::vector<Vector *> p) const
|
||||
DifferentiableOperator::DifferentiableOperator(
|
||||
const std::vector<FieldDescriptor> &infds,
|
||||
const std::vector<FieldDescriptor> &outfds,
|
||||
const ParMesh &mesh) :
|
||||
Operator(),
|
||||
mesh(mesh),
|
||||
infds(infds),
|
||||
outfds(outfds)
|
||||
{
|
||||
MFEM_ASSERT(parameters.size() == p.size(),
|
||||
"number of parameters doesn't match descriptors");
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
unionfds.clear();
|
||||
unionfds.insert(unionfds.end(), infds.begin(), infds.end());
|
||||
unionfds.insert(unionfds.end(), outfds.begin(), outfds.end());
|
||||
std::sort(unionfds.begin(), unionfds.end());
|
||||
auto last = std::unique(unionfds.begin(), unionfds.end());
|
||||
unionfds.erase(last, unionfds.end());
|
||||
|
||||
infields_l.resize(infds.size());
|
||||
for (size_t i = 0; i < infds.size(); i++)
|
||||
{
|
||||
p[i]->Read();
|
||||
parameters_l[i] = *p[i];
|
||||
infields_l[i] = new Vector(GetVSize(infds[i]));
|
||||
}
|
||||
|
||||
infields_e.resize(infds.size());
|
||||
}
|
||||
|
||||
DifferentiableOperator::DifferentiableOperator(
|
||||
const std::vector<FieldDescriptor> &solutions,
|
||||
const std::vector<FieldDescriptor> ¶meters,
|
||||
const ParMesh &mesh) :
|
||||
mesh(mesh),
|
||||
solutions(solutions),
|
||||
parameters(parameters)
|
||||
void DifferentiableOperator::SetMultLevel(MultLevel level)
|
||||
{
|
||||
fields.resize(solutions.size() + parameters.size());
|
||||
fields_e.resize(fields.size());
|
||||
solutions_l.resize(solutions.size());
|
||||
parameters_l.resize(parameters.size());
|
||||
mult_level = level;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < solutions.size(); i++)
|
||||
void DifferentiableOperator::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_ASSERT(!action_callbacks.empty(),
|
||||
"no integrators have been set");
|
||||
|
||||
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&x),
|
||||
"x needs to be a BlockVector");
|
||||
|
||||
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&y),
|
||||
"y needs to be a BlockVector");
|
||||
|
||||
const auto &bx = static_cast<const BlockVector &>(x);
|
||||
auto &by = static_cast<BlockVector &>(y);
|
||||
|
||||
Mult(bx, by);
|
||||
}
|
||||
|
||||
void DifferentiableOperator::DisableTensorProductStructure(bool disable)
|
||||
{
|
||||
use_tensor_product_structure = !disable;
|
||||
}
|
||||
|
||||
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
|
||||
size_t derivative_id, const Vector &x)
|
||||
{
|
||||
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
|
||||
derivative_action_callbacks.end(),
|
||||
"no derivative action has been found for ID " << derivative_id);
|
||||
|
||||
const size_t dfidx = FindIdx(derivative_id, infds);
|
||||
|
||||
// Get transpose callbacks if available, otherwise pass empty vector
|
||||
std::vector<derivative_action_t> transpose_callbacks;
|
||||
auto it = daction_transpose_callbacks.find(derivative_id);
|
||||
if (it != daction_transpose_callbacks.end())
|
||||
{
|
||||
fields[i] = solutions[i];
|
||||
transpose_callbacks = it->second;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
return std::make_shared<DerivativeOperator>(
|
||||
height,
|
||||
GetTrueVSize(infds[dfidx]),
|
||||
derivative_action_callbacks[derivative_id],
|
||||
transpose_callbacks,
|
||||
infds[dfidx],
|
||||
x,
|
||||
infds,
|
||||
outfds);
|
||||
}
|
||||
|
||||
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
|
||||
size_t derivative_id, const MultiVector &x)
|
||||
{
|
||||
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
|
||||
derivative_action_callbacks.end(),
|
||||
"no derivative action has been found for ID " << derivative_id);
|
||||
|
||||
const size_t dfidx = FindIdx(derivative_id, infds);
|
||||
|
||||
// Get transpose callbacks if available, otherwise pass empty vector
|
||||
std::vector<derivative_action_t> transpose_callbacks;
|
||||
auto it = daction_transpose_callbacks.find(derivative_id);
|
||||
if (it != daction_transpose_callbacks.end())
|
||||
{
|
||||
fields[i + solutions.size()] = parameters[i];
|
||||
transpose_callbacks = it->second;
|
||||
}
|
||||
|
||||
return std::make_shared<DerivativeOperator>(
|
||||
height,
|
||||
GetTrueVSize(infds[dfidx]),
|
||||
derivative_action_callbacks[derivative_id],
|
||||
transpose_callbacks,
|
||||
infds[dfidx],
|
||||
x,
|
||||
infds,
|
||||
outfds);
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
+236
-896
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,63 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../fespace.hpp"
|
||||
#include "parameterspace.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// @brief FieldDescriptor struct
|
||||
///
|
||||
/// This struct is used to store information about a field.
|
||||
struct FieldDescriptor
|
||||
{
|
||||
using data_variant_t =
|
||||
std::variant<const FiniteElementSpace *,
|
||||
const ParFiniteElementSpace *,
|
||||
const QuadratureFunction *,
|
||||
const ParameterSpace *>;
|
||||
|
||||
/// Field ID
|
||||
std::size_t id;
|
||||
|
||||
/// Field variant
|
||||
data_variant_t data;
|
||||
|
||||
/// Default constructor
|
||||
FieldDescriptor() :
|
||||
id(SIZE_MAX), data(data_variant_t{}) {}
|
||||
|
||||
/// Constructor
|
||||
template <typename T>
|
||||
FieldDescriptor(std::size_t field_id, const T* v) :
|
||||
id(field_id), data(v) {}
|
||||
|
||||
bool operator==(const FieldDescriptor& other) const
|
||||
{
|
||||
return id == other.id;
|
||||
}
|
||||
|
||||
bool operator<(const FieldDescriptor& other) const
|
||||
{
|
||||
return id < other.id;
|
||||
}
|
||||
|
||||
friend void swap(FieldDescriptor& a, FieldDescriptor& b)
|
||||
{
|
||||
using std::swap;
|
||||
swap(a.id, b.id);
|
||||
swap(a.data, b.data);
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct IntegratorContext
|
||||
{
|
||||
const ParMesh &mesh;
|
||||
const Array<int> *elem_attr;
|
||||
Array<int> attr;
|
||||
int nentities;
|
||||
const std::vector<FieldDescriptor> &infds;
|
||||
const std::vector<FieldDescriptor> &outfds;
|
||||
const std::vector<FieldDescriptor> &unionfds;
|
||||
const IntegrationRule &ir;
|
||||
std::unordered_map<std::type_index, std::vector<int>> &in_qlayouts;
|
||||
std::unordered_map<std::type_index, std::vector<int>> &out_qlayouts;
|
||||
};
|
||||
|
||||
}
|
||||
@@ -9,8 +9,23 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
// #define NVTX_COLOR nvtx::kPeru
|
||||
|
||||
#include "util.hpp"
|
||||
#include "fem/kernels.hpp"
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template <class T>
|
||||
inline std::enable_if_t<!std::numeric_limits<T>::is_integer, bool>
|
||||
AlmostEq(T x, T y, T tolerance = 15.0 * std::numeric_limits<T>::epsilon())
|
||||
{
|
||||
const T neg = std::abs(x - y);
|
||||
constexpr T min = std::numeric_limits<T>::min();
|
||||
constexpr T eps = std::numeric_limits<T>::epsilon();
|
||||
const T min_abs = std::min(std::abs(x), std::abs(y));
|
||||
if (std::abs(min_abs) == 0.0) { return neg < eps; }
|
||||
return (neg / (1.0 + std::max(min, min_abs))) < tolerance;
|
||||
}
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
@@ -30,6 +45,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
dbg("Value");
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
@@ -94,10 +110,11 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
// dbg("Gradient");
|
||||
const auto [q1d, B_dim, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
const auto field = Reshape(&std::as_const(field_e[0]), d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
@@ -106,7 +123,30 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
// static constexpr int DIM = 3;
|
||||
// MFEM_VERIFY(q1d <= MQ1, "q1d > MQ1");
|
||||
// MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
|
||||
// kernels::internal::d_regs3d_t<DIM, MQ1> r0, r1;
|
||||
// real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
|
||||
/*
|
||||
{
|
||||
assert(B_dim == 1 && "1D B required!");
|
||||
kernels::internal::LoadMatrix(d1d, q1d, B, sB);
|
||||
kernels::internal::LoadMatrix(d1d, q1d, G, sG);
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
assert(AlmostEq(B(qx, 0, dx), sB[dx][qx]));
|
||||
assert(AlmostEq(G(qx, 0, dx), sG[dx][qx]));
|
||||
}
|
||||
}
|
||||
}*/
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
@@ -117,7 +157,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, vd);
|
||||
const real_t f = field(dx, dy, dz, c);
|
||||
uv[0] += f * B(qx, 0, dx);
|
||||
uv[1] += f * G(qx, 0, dx);
|
||||
}
|
||||
@@ -163,19 +203,59 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
|
||||
}
|
||||
fqp(vd, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(vd, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(vd, 2, qx, qy, qz) = uvw[2];
|
||||
fqp(c, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(c, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(c, 2, qx, qy, qz) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/*
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
kernels::internal::LoadDofs3d(d1d, c, field, r0);
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, c);
|
||||
assert(AlmostEq(f, r0[d][dz][dy][dx]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
kernels::internal::Grad3d(d1d, q1d, smem, sB, sG, r0, r1, c);
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
if (!AlmostEq(fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]))
|
||||
{
|
||||
dbg("\x1b[31m[{}:d] {} {}", c, fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]);
|
||||
dbg("❌❌❌"), std::exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// dbg("✅✅✅✅✅✅✅✅✅✅✅✅✅✅✅");//, std::exit(EXIT_SUCCESS);
|
||||
}*/
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
|
||||
{
|
||||
// dbg("None");
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
@@ -518,6 +598,9 @@ void map_fields_to_quadrature_data(
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
// dbg();
|
||||
assert(use_sum_factorization && "❌ use_sum_factorization required");
|
||||
|
||||
// When the input_to_field map returns -1, this means the requested input
|
||||
// is the integration weight. Weights don't have a user defined field
|
||||
// attached to them and we create a dummy field which is not accessed
|
||||
@@ -578,6 +661,7 @@ void map_field_to_quadrature_data_conditional(
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
assert(false && "❌ condition not implemented");
|
||||
if (condition)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
@@ -619,6 +703,7 @@ void map_fields_to_quadrature_data_conditional(
|
||||
const std::array<bool, num_inputs> &conditions,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
assert(false && "❌ condition not implemented");
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
map_field_to_quadrature_data_conditional(
|
||||
@@ -627,7 +712,7 @@ void map_fields_to_quadrature_data_conditional(
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t num_inputs, typename field_operator_ts>
|
||||
template <int T_Q1D, size_t num_inputs, typename field_operator_ts>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_direction_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_inputs> &directions_qp,
|
||||
@@ -660,7 +745,7 @@ void map_direction_to_quadrature_data_conditional(
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
map_field_to_quadrature_data_tensor_product_3d<T_Q1D>(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
|
||||
@@ -43,7 +43,7 @@ public:
|
||||
/// Get spatial dimension
|
||||
///
|
||||
/// returns always 1.
|
||||
int Dimension() const
|
||||
constexpr int Dimension() const
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
@@ -74,11 +74,14 @@ public:
|
||||
return elem_restr.get();
|
||||
}
|
||||
|
||||
virtual const Operator* GetB() const = 0;
|
||||
|
||||
virtual const Operator* GetBt() const = 0;
|
||||
|
||||
protected:
|
||||
int vdim;
|
||||
DofToQuad dtq;
|
||||
mutable std::unique_ptr<Operator> prolongation;
|
||||
mutable std::unique_ptr<Operator> elem_restr;
|
||||
mutable std::unique_ptr<Operator> prolongation, elem_restr, B, Bt;
|
||||
};
|
||||
|
||||
/// @brief Uniform parameter space
|
||||
@@ -122,6 +125,18 @@ public:
|
||||
return lsize;
|
||||
}
|
||||
|
||||
const Operator* GetB() const override
|
||||
{
|
||||
MFEM_ABORT("UniformParameterSpace does not support GetB");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const Operator* GetBt() const override
|
||||
{
|
||||
MFEM_ABORT("UniformParameterSpace does not support GetBt");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
private:
|
||||
/// T-vector size
|
||||
int tsize;
|
||||
|
||||
@@ -243,6 +243,8 @@ void process_qf_arg(
|
||||
}
|
||||
}
|
||||
|
||||
// const tensor<real_t, DIM> ∇u
|
||||
// const tensor<real_t, DIM, DIM> D (PA_DATA)
|
||||
template <typename arg_type>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "tuple.hpp"
|
||||
#include "../linalg/tensor.hpp"
|
||||
|
||||
using namespace mfem::future;
|
||||
using mfem::future::tensor;
|
||||
|
||||
// Helper to add dimension to tensor type
|
||||
template<typename T, int qp>
|
||||
struct AddQPDimension;
|
||||
|
||||
// Specialization for tensor<real_t, dim>
|
||||
template<typename real_t, int dim, int qp>
|
||||
struct AddQPDimension<tensor<real_t, dim>, qp>
|
||||
{
|
||||
using type = tensor<real_t, dim, qp>;
|
||||
};
|
||||
|
||||
// Specialization for tensor<real_t, dim, dim>
|
||||
template<typename real_t, int dim, int qp>
|
||||
struct AddQPDimension<tensor<real_t, dim, dim>, qp>
|
||||
{
|
||||
using type = tensor<real_t, dim, dim, qp>;
|
||||
};
|
||||
|
||||
// Specialization for real_t (transforms to tensor<real_t, qp>)
|
||||
template<typename real_t, int qp>
|
||||
struct AddQPDimension
|
||||
{
|
||||
using type = tensor<real_t, qp>;
|
||||
};
|
||||
|
||||
// Helper to transform tuple
|
||||
template<typename Tuple, int qp>
|
||||
struct TransformTupleQP {};
|
||||
|
||||
// Specialization for mfem::future::tuple
|
||||
template<int qp, typename... Types>
|
||||
struct TransformTupleQP<mfem::future::tuple<Types...>, qp>
|
||||
{
|
||||
using type = mfem::future::tuple<typename AddQPDimension<Types, qp>::type...>;
|
||||
};
|
||||
|
||||
template<int qp, typename... Types>
|
||||
struct TransformTupleQP<std::tuple<Types...>, qp>
|
||||
{
|
||||
using type = std::tuple<typename AddQPDimension<Types, qp>::type...>;
|
||||
};
|
||||
|
||||
// Function to transform tuple type with qp dimension
|
||||
template<int qp, typename qf_param_ts>
|
||||
struct add_qp_dimension
|
||||
{
|
||||
using type = typename TransformTupleQP<qf_param_ts, qp>::type;
|
||||
};
|
||||
|
||||
// Helper alias template for cleaner usage
|
||||
template<int qp, typename qf_param_ts>
|
||||
using add_qp_dimension_t = typename add_qp_dimension<qp, qf_param_ts>::type;
|
||||
|
||||
// ...AddDomainIntegrator...
|
||||
// {
|
||||
// constexpr int Q1D = 4;
|
||||
// using qf_param_augmentd_ts = add_qp_dimension_t<Q1D, decay_tuple<qf_param_ts>>;
|
||||
// }
|
||||
+533
-50
@@ -21,6 +21,7 @@
|
||||
#include <type_traits>
|
||||
#include <numeric>
|
||||
#include <iomanip>
|
||||
#include <typeindex>
|
||||
|
||||
#include "../../general/communication.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
@@ -28,13 +29,19 @@
|
||||
#include "../fe/fe_base.hpp"
|
||||
#include "../fespace.hpp"
|
||||
#include "../pfespace.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../../mesh/mesh.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../quadinterpolator.hpp"
|
||||
|
||||
#include "fielddescriptor.hpp"
|
||||
#include "fieldoperator.hpp"
|
||||
#include "parameterspace.hpp"
|
||||
#include "tuple.hpp"
|
||||
|
||||
#undef NVTX_COLOR
|
||||
#define NVTX_COLOR ::nvtx::kLightBlue
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
@@ -75,7 +82,7 @@ constexpr void for_constexpr(lambda&& f,
|
||||
}
|
||||
|
||||
template <typename lambda>
|
||||
constexpr void for_constexpr(lambda&& f, std::integer_sequence<std::size_t>) {}
|
||||
constexpr void for_constexpr(lambda&&, std::integer_sequence<std::size_t>) {}
|
||||
|
||||
template <int... n, typename lambda>
|
||||
constexpr void for_constexpr(lambda&& f)
|
||||
@@ -84,7 +91,7 @@ constexpr void for_constexpr(lambda&& f)
|
||||
}
|
||||
|
||||
template <typename lambda, typename arg_t>
|
||||
constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg,
|
||||
constexpr void for_constexpr_with_arg(lambda&&, arg_t&&,
|
||||
std::integer_sequence<std::size_t>)
|
||||
{
|
||||
// Base case - do nothing for empty sequence
|
||||
@@ -108,6 +115,16 @@ constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg)
|
||||
indices{});
|
||||
}
|
||||
|
||||
template <auto start, auto end, auto inc = 1, typename F>
|
||||
constexpr void constexpr_for(F&& f)
|
||||
{
|
||||
if constexpr (start < end)
|
||||
{
|
||||
f(std::integral_constant<decltype(start), start>());
|
||||
constexpr_for<start + inc, end, inc>(f);
|
||||
}
|
||||
}
|
||||
|
||||
template <std::size_t I, typename Tuple, std::size_t... Is>
|
||||
std::array<bool, sizeof...(Is)>
|
||||
make_dependency_array(const Tuple& inputs, std::index_sequence<Is...>)
|
||||
@@ -444,6 +461,21 @@ struct create_function_signature<output_t (*)(input_ts...)>
|
||||
using type = FunctionSignature<output_t(input_ts...)>;
|
||||
};
|
||||
|
||||
template <typename...>
|
||||
using void_t = void;
|
||||
|
||||
template <typename T, typename = void>
|
||||
struct get_function_signature
|
||||
{
|
||||
using type = typename create_function_signature<T>::type;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct get_function_signature<T, void_t<decltype(&T::operator())>>
|
||||
{
|
||||
using type = typename create_function_signature<decltype(&T::operator())>::type;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
constexpr int GetFieldId()
|
||||
{
|
||||
@@ -538,38 +570,12 @@ auto get_marked_entries(
|
||||
/// @param t the tuple to filter fields from.
|
||||
/// @returns a tuple containing only the fields with field IDs not equal to -1.
|
||||
template <typename... Ts>
|
||||
constexpr auto filter_fields(const std::tuple<Ts...>& t)
|
||||
constexpr auto filter_fields(const std::tuple<Ts...>&)
|
||||
{
|
||||
return std::tuple_cat(
|
||||
std::conditional_t<Ts::GetFieldId() != -1, std::tuple<Ts>, std::tuple<>> {}...);
|
||||
}
|
||||
|
||||
/// @brief FieldDescriptor struct
|
||||
///
|
||||
/// This struct is used to store information about a field.
|
||||
struct FieldDescriptor
|
||||
{
|
||||
using data_variant_t =
|
||||
std::variant<const FiniteElementSpace *,
|
||||
const ParFiniteElementSpace *,
|
||||
const ParameterSpace *>;
|
||||
|
||||
/// Field ID
|
||||
std::size_t id;
|
||||
|
||||
/// Field variant
|
||||
data_variant_t data;
|
||||
|
||||
/// Default constructor
|
||||
FieldDescriptor() :
|
||||
id(SIZE_MAX), data(data_variant_t{}) {}
|
||||
|
||||
/// Constructor
|
||||
template <typename T>
|
||||
FieldDescriptor(std::size_t field_id, const T* v) :
|
||||
id(field_id), data(v) {}
|
||||
};
|
||||
|
||||
namespace dfem
|
||||
{
|
||||
template <class... T> constexpr bool always_false = false;
|
||||
@@ -599,7 +605,7 @@ struct ThreadBlocks
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_shmem(func_t f, int n)
|
||||
__global__ void forall_kernel_extern_shmem(func_t f, int n)
|
||||
{
|
||||
int i = blockIdx.x;
|
||||
extern __shared__ real_t shmem[];
|
||||
@@ -608,23 +614,48 @@ __global__ void forall_kernel_shmem(func_t f, int n)
|
||||
f(i, shmem);
|
||||
}
|
||||
}
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_static_smem(func_t f, int n)
|
||||
{
|
||||
int i = blockIdx.x;
|
||||
if (i >= n) { return; }
|
||||
f(i, nullptr);
|
||||
}
|
||||
template <int MAX_THREADS_PER_BLOCK, typename func_t>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void forall_kernel_static_smem_launch_bounds(func_t f, int n)
|
||||
{
|
||||
for (int k = blockIdx.x; k < n; k += gridDim.x) { f(k, nullptr); }
|
||||
}
|
||||
#endif
|
||||
|
||||
template <typename func_t>
|
||||
template </*typename kernel_tag,*/ typename func_t>
|
||||
void forall(func_t f,
|
||||
const int &N,
|
||||
const ThreadBlocks &blocks,
|
||||
int num_shmem = 0,
|
||||
[[maybe_unused]] const ThreadBlocks &blocks,
|
||||
[[maybe_unused]] int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
db1();
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
// int gridsize = (N + Z - 1) / Z;
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
db1("num_bytes:{}", num_bytes);
|
||||
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
forall_kernel_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
// ForallKernel<kernel_tag>::run<<<N, block_size, num_bytes>>>(f, N);
|
||||
if (num_bytes > 0)
|
||||
{
|
||||
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
forall_kernel_static_smem<<<N, block_size>>>(f, N);
|
||||
}
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
@@ -635,6 +666,7 @@ void forall(func_t f,
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
db1("CPU_MASK");
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
for (int i = 0; i < N; i++)
|
||||
@@ -648,6 +680,69 @@ void forall(func_t f,
|
||||
}
|
||||
}
|
||||
|
||||
namespace dfem
|
||||
{
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK = 0, typename func_t>
|
||||
void forall(func_t f,
|
||||
const int &N,
|
||||
[[maybe_unused]] const ThreadBlocks &blocks,
|
||||
[[maybe_unused]] int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
db1();
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
db1("num_bytes:{}", num_bytes);
|
||||
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
|
||||
db1("MAX_THREADS_PER_BLOCK:{}", MAX_THREADS_PER_BLOCK);
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
if constexpr (MAX_THREADS_PER_BLOCK > 0)
|
||||
{
|
||||
assert(num_bytes == 0);
|
||||
forall_kernel_static_smem_launch_bounds
|
||||
<MAX_THREADS_PER_BLOCK><<<N, block_size>>> (f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(MAX_THREADS_PER_BLOCK == 0);
|
||||
if (num_bytes == 0)
|
||||
{
|
||||
forall_kernel_static_smem<<<N, block_size>>>(f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
}
|
||||
}
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
#endif
|
||||
// MFEM_DEVICE_SYNC; // ⚠️
|
||||
#endif
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
db1("CPU_MASK");
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
f(i, shmem);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("no compute backend available");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// @todo To be removed.
|
||||
class FDJacobian : public Operator
|
||||
{
|
||||
@@ -772,6 +867,10 @@ int GetVSize(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetVSize();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->Size();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetVSize();
|
||||
@@ -810,6 +909,10 @@ void GetElementVDofs(const FieldDescriptor &f, int el, Array<int> &vdofs)
|
||||
{
|
||||
arg->GetElementVDofs(el, vdofs);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
MFEM_ABORT("internal error");
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
MFEM_ABORT("internal error");
|
||||
@@ -844,6 +947,10 @@ int GetTrueVSize(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetTrueVSize();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->Size();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetTrueVSize();
|
||||
@@ -874,6 +981,10 @@ int GetVDim(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
@@ -909,6 +1020,10 @@ int GetDimension(const FieldDescriptor &f)
|
||||
return arg->GetMesh()->Dimension() - 1;
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->GetSpace()->GetMesh()->Dimension();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->Dimension();
|
||||
@@ -921,6 +1036,36 @@ int GetDimension(const FieldDescriptor &f)
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
inline
|
||||
std::variant<const QuadratureInterpolator *, const Operator *>get_qinterp(
|
||||
const FieldDescriptor &f,
|
||||
const IntegrationRule &ir)
|
||||
{
|
||||
return std::visit([&ir](auto && arg) -> const QuadratureInterpolator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
|
||||
std::is_same_v<T, const ParFiniteElementSpace *>)
|
||||
{
|
||||
return arg->GetQuadratureInterpolator(ir);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
// QuadratureFunction doesn't need a QuadratureInterpolator
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>, "internal error");
|
||||
}
|
||||
|
||||
return nullptr; // Unreachable, but avoids compiler warning
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
/// @brief Get the prolongation operator for a field descriptor.
|
||||
///
|
||||
@@ -929,6 +1074,7 @@ int GetDimension(const FieldDescriptor &f)
|
||||
inline
|
||||
const Operator *get_prolongation(const FieldDescriptor &f)
|
||||
{
|
||||
NVTX("get P");
|
||||
return std::visit([](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -937,6 +1083,10 @@ const Operator *get_prolongation(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetProlongationMatrix();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetProlongationMatrix();
|
||||
@@ -959,6 +1109,7 @@ inline
|
||||
const Operator *get_element_restriction(const FieldDescriptor &f,
|
||||
ElementDofOrdering o)
|
||||
{
|
||||
NVTX("get ER");
|
||||
return std::visit([&o](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -967,6 +1118,10 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
|
||||
{
|
||||
return arg->GetElementRestriction(o);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetElementRestriction(o);
|
||||
@@ -994,6 +1149,7 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
|
||||
FaceType ft,
|
||||
L2FaceValues m)
|
||||
{
|
||||
NVTX("get FR");
|
||||
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -1002,6 +1158,11 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
|
||||
{
|
||||
return arg->GetFaceRestriction(o, ft, m);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
// QuadratureFunction does not support face restrictions
|
||||
MFEM_ABORT("internal error");
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
// ParameterSpace does not support face restrictions
|
||||
@@ -1027,6 +1188,7 @@ inline
|
||||
const Operator *get_restriction(const FieldDescriptor &f,
|
||||
const ElementDofOrdering &o)
|
||||
{
|
||||
NVTX("get R");
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
return get_element_restriction(f, o);
|
||||
@@ -1052,12 +1214,14 @@ inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
|
||||
get_restriction_transpose(
|
||||
const FieldDescriptor &f,
|
||||
const ElementDofOrdering &o,
|
||||
const fop_t &fop)
|
||||
[[maybe_unused]] const fop_t &fop)
|
||||
{
|
||||
NVTX("get R^T");
|
||||
if constexpr (is_sum_fop<fop_t>::value)
|
||||
{
|
||||
auto RT = [=](const Vector &v_e, Vector &v_l)
|
||||
{
|
||||
NVTX("R^T sum");
|
||||
v_l += v_e;
|
||||
};
|
||||
return std::make_tuple(RT, 1);
|
||||
@@ -1067,6 +1231,7 @@ get_restriction_transpose(
|
||||
const Operator *R = get_restriction<entity_t>(f, o);
|
||||
std::function<void(const Vector&, Vector&)> RT = [=](const Vector &x, Vector &y)
|
||||
{
|
||||
NVTX("R^T+");
|
||||
R->AddMultTranspose(x, y);
|
||||
};
|
||||
return std::make_tuple(RT, R->Height());
|
||||
@@ -1086,11 +1251,26 @@ get_restriction_transpose(
|
||||
inline
|
||||
void prolongation(const FieldDescriptor field, const Vector &x, Vector &field_l)
|
||||
{
|
||||
NVTX("P");
|
||||
const auto P = get_prolongation(field);
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
field_l.SetSize(P->Height());
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("P->Mult");
|
||||
P->Mult(x, field_l);
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const FieldDescriptor &field, const Vector &field_l, Vector &x)
|
||||
{
|
||||
const auto P = get_prolongation(field);
|
||||
x.SetSize(P->Width());
|
||||
P->MultTranspose(field_l, x);
|
||||
}
|
||||
|
||||
/// @brief Apply the prolongation operator to a vector of fields.
|
||||
///
|
||||
/// x is a long vector containing the data for all fields on tdofs and
|
||||
@@ -1107,6 +1287,7 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
const Vector &x,
|
||||
std::array<Vector, M> &fields_l)
|
||||
{
|
||||
NVTX("P");
|
||||
int data_offset = 0;
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
@@ -1114,9 +1295,14 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
const int width = P->Width();
|
||||
// const Vector x_i(x.GetData() + data_offset, width);
|
||||
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
fields_l[i].SetSize(P->Height());
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
fields_l[i].SetSize(P->Height());
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("P->Mult");
|
||||
P->Mult(x_i, fields_l[i]);
|
||||
NVTX_END("P->Mult");
|
||||
data_offset += width;
|
||||
}
|
||||
}
|
||||
@@ -1130,20 +1316,259 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
/// @param fields the array of field descriptors.
|
||||
/// @param x the input vector in tdofs.
|
||||
/// @param fields_l the array of output vectors in vdofs.
|
||||
// inline
|
||||
// void prolongation(const std::vector<FieldDescriptor> fields,
|
||||
// const Vector &x,
|
||||
// std::vector<Vector> &fields_l)
|
||||
// {
|
||||
// int data_offset = 0;
|
||||
// for (std::size_t i = 0; i < fields.size(); i++)
|
||||
// {
|
||||
// const auto P = get_prolongation(fields[i]);
|
||||
// const int width = P->Width();
|
||||
// const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
// fields_l[i].SetSize(P->Height());
|
||||
// P->Mult(x_i, fields_l[i]);
|
||||
// data_offset += width;
|
||||
// }
|
||||
// }
|
||||
|
||||
inline
|
||||
void prolongation(const std::vector<FieldDescriptor> fields,
|
||||
const Vector &x,
|
||||
std::vector<Vector> &fields_l)
|
||||
void prolongation(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const BlockVector &x,
|
||||
std::vector<Vector *> &x_l)
|
||||
{
|
||||
int data_offset = 0;
|
||||
for (std::size_t i = 0; i < fields.size(); i++)
|
||||
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
|
||||
"error " << x.NumBlocks() << " vs " << x_l.size());
|
||||
for (int i = 0; i < x.NumBlocks(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
const int width = P->Width();
|
||||
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
fields_l[i].SetSize(P->Height());
|
||||
P->Mult(x_i, fields_l[i]);
|
||||
data_offset += width;
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
*x_l[i] = x.GetBlock(i);
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Width() << " vs " << x.GetBlock(i).Size());
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
P->Mult(x.GetBlock(i), *x_l[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const MultiVector &x,
|
||||
std::vector<Vector *> &x_l)
|
||||
{
|
||||
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
|
||||
"error " << x.NumBlocks() << " vs " << x_l.size());
|
||||
for (int i = 0; i < x.NumBlocks(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
*x_l[i] = x[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
MFEM_ASSERT(P->Width() == x[i].Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Width() << " vs " << x[i].Size());
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
P->Mult(x[i], *x_l[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
BlockVector &x)
|
||||
{
|
||||
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
|
||||
"error " << x_l.size() << " vs " << x.NumBlocks());
|
||||
for (size_t i = 0; i < x_l.size(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
x.GetBlock(i) = *x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Width() << " vs " << x.GetBlock(i).Size());
|
||||
P->MultTranspose(*x_l[i], x.GetBlock(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
MultiVector &x)
|
||||
{
|
||||
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
|
||||
"error " << x_l.size() << " vs " << x.NumBlocks());
|
||||
for (size_t i = 0; i < x_l.size(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
x[i] = *x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
MFEM_ASSERT(P->Width() == x[i].Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Width() << " vs " << x[i].Size());
|
||||
P->MultTranspose(*x_l[i], x[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void restriction(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
std::vector<Vector *> &x_e)
|
||||
{
|
||||
MFEM_ASSERT(x_l.size() == x_e.size(),
|
||||
"internal error " << x_l.size() << " vs " << x_e.size());
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (R == nullptr)
|
||||
{
|
||||
s = x_l[i]->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
s = R->Height();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (x_e[i] == nullptr)
|
||||
{
|
||||
x_e[i] = new Vector(s);
|
||||
}
|
||||
x_e[i]->SetSize(s);
|
||||
|
||||
if (R == nullptr)
|
||||
{
|
||||
x_e[i] = x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(R->Width() == x_l[i]->Size(),
|
||||
"restriction not applicable to given input data size " <<
|
||||
R->Width() << " vs " << x_l[i]->Size());
|
||||
R->Mult(*x_l[i], *x_e[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void prepare_residual(
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
std::vector<Vector *> &r_e)
|
||||
{
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
if (std::holds_alternative<const QuadratureFunction *>(fields[i].data))
|
||||
{
|
||||
const auto fd = std::get<const QuadratureFunction *>(fields[i].data);
|
||||
s = fd->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
s = R->Height();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (r_e[i] == nullptr)
|
||||
{
|
||||
r_e[i] = new Vector(s);
|
||||
}
|
||||
else
|
||||
{
|
||||
r_e[i]->SetSize(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void restriction_transpose(
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
const std::vector<Vector *> &x_e,
|
||||
std::vector<Vector *> &x_l)
|
||||
{
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
// TODO: if nullptr, assume Identity
|
||||
if (R == nullptr)
|
||||
{
|
||||
s = x_e[i]->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
s = R->Width();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (x_l[i] == nullptr)
|
||||
{
|
||||
x_l[i] = new Vector(s);
|
||||
}
|
||||
x_l[i]->SetSize(s);
|
||||
|
||||
// TODO: if nullptr, assume Identity
|
||||
if (R == nullptr)
|
||||
{
|
||||
x_l[i] = x_e[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
R->MultTranspose(*x_e[i], *x_l[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1152,6 +1577,7 @@ void get_lvectors(const std::vector<FieldDescriptor> fields,
|
||||
const Vector &x,
|
||||
std::vector<Vector> &fields_l)
|
||||
{
|
||||
NVTX("get_lvectors");
|
||||
int data_offset = 0;
|
||||
for (std::size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
@@ -1178,13 +1604,15 @@ template <typename fop_t>
|
||||
inline
|
||||
std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
const FieldDescriptor &f,
|
||||
const fop_t &fop,
|
||||
[[maybe_unused]] const fop_t &fop,
|
||||
MPI_Comm mpi_comm)
|
||||
{
|
||||
NVTX("get P^T");
|
||||
if constexpr (is_sum_fop<fop_t>::value)
|
||||
{
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T sum");
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
real_t local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM, mpi_comm);
|
||||
@@ -1195,6 +1623,7 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
{
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T Identity");
|
||||
y = r_local;
|
||||
};
|
||||
return PT;
|
||||
@@ -1202,6 +1631,7 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
const Operator *P = get_prolongation(f);
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T");
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
return PT;
|
||||
@@ -1220,12 +1650,19 @@ void restriction(const FieldDescriptor u,
|
||||
Vector &field_e,
|
||||
ElementDofOrdering ordering)
|
||||
{
|
||||
NVTX("R");
|
||||
const auto R = get_restriction<entity_t>(u, ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l.Size(),
|
||||
"restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
field_e.SetSize(height);
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("R->Mult");
|
||||
R->Mult(u_l, field_e);
|
||||
NVTX_END("R->Mult");
|
||||
}
|
||||
|
||||
/// @brief Apply the restriction operator to a vector of fields.
|
||||
@@ -1243,14 +1680,29 @@ void restriction(const std::vector<FieldDescriptor> u,
|
||||
ElementDofOrdering ordering,
|
||||
const int offset = 0)
|
||||
{
|
||||
NVTX("R");
|
||||
for (std::size_t i = 0; i < u.size(); i++)
|
||||
{
|
||||
const auto R = get_restriction<entity_t>(u[i], ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l[i].Size(),
|
||||
"restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
// NVTX_INI("SetSize");
|
||||
fields_e[i + offset].SetSize(height);
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
// NVTX_END("SetSize");
|
||||
|
||||
// NVTX_INI("R->Mult");
|
||||
if (dynamic_cast<const IdentityOperator*>(R))
|
||||
{
|
||||
NVTX("Identity");
|
||||
fields_e[i + offset].NewMemoryAndSize(u_l[i].GetMemory(), u_l[i].Size(), false);
|
||||
}
|
||||
else
|
||||
{
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
}
|
||||
// NVTX_END("R->Mult");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1262,14 +1714,21 @@ void element_restriction(const std::array<FieldDescriptor, N> u,
|
||||
ElementDofOrdering ordering,
|
||||
const int offset = 0)
|
||||
{
|
||||
NVTX("ER");
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
const auto R = get_element_restriction(u[i], ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l[i].Size(),
|
||||
"element restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
fields_e[i + offset].SetSize(height);
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("R->Mult");
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
NVTX_END("R->Mult");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1326,6 +1785,10 @@ const DofToQuad *GetDofToQuad(const FieldDescriptor &f,
|
||||
return &arg->GetTypicalTraceElement()->GetDofToQuad(ir, mode);
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return &arg->GetDofToQuad();
|
||||
@@ -1457,7 +1920,7 @@ create_descriptors_to_fields_map(
|
||||
|
||||
auto f = [&](auto &fop, auto &map)
|
||||
{
|
||||
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
|
||||
if constexpr (is_weight_fop<std::decay_t<decltype(fop)>>::value)
|
||||
{
|
||||
// TODO-bug: stealing dimension from the first field
|
||||
fop.dim = GetDimension<entity_t>(fields[0]);
|
||||
@@ -1587,7 +2050,7 @@ get_shmem_info(
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
const int &num_entities,
|
||||
const input_t &inputs,
|
||||
[[maybe_unused]] const input_t &inputs,
|
||||
const int &num_qp,
|
||||
const std::vector<int> &input_size_on_qp,
|
||||
const int &residual_size_on_qp,
|
||||
@@ -2342,5 +2805,25 @@ std::array<DofToQuadMap, num_fields> create_dtq_maps(
|
||||
std::make_index_sequence<num_fields> {});
|
||||
}
|
||||
|
||||
struct QLayoutEntry
|
||||
{
|
||||
std::type_index type;
|
||||
std::vector<int> layout;
|
||||
|
||||
template <class Fop>
|
||||
QLayoutEntry(Fop, std::initializer_list<int> idx) :
|
||||
type(typeid(Fop)), layout(idx) {}
|
||||
};
|
||||
|
||||
static void ExtractQLayouts(
|
||||
const std::initializer_list<QLayoutEntry> entries,
|
||||
std::unordered_map<std::type_index, std::vector<int>>& out)
|
||||
{
|
||||
for (const auto& e : entries)
|
||||
{
|
||||
out[e.type] = e.layout;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
#endif
|
||||
|
||||
+6
-6
@@ -320,8 +320,8 @@ public:
|
||||
error estimation procedure where the flux averaging is replaced by a global
|
||||
L2 projection (requiring a mass matrix solve).
|
||||
|
||||
The required BilinearFormIntegrator must implement the methods
|
||||
ComputeElementFlux() and ComputeFluxEnergy().
|
||||
The required BilinearFormIntegrator must implement the method
|
||||
ComputeElementFlux().
|
||||
|
||||
Implemented for the parallel case only.
|
||||
*/
|
||||
@@ -357,8 +357,8 @@ protected:
|
||||
|
||||
public:
|
||||
/** @brief Construct a new L2ZienkiewiczZhuEstimator object.
|
||||
@param integ This BilinearFormIntegrator must implement the methods
|
||||
ComputeElementFlux() and ComputeFluxEnergy().
|
||||
@param integ This BilinearFormIntegrator must implement the method
|
||||
ComputeElementFlux().
|
||||
@param sol The solution field whose error is to be estimated.
|
||||
@param flux_fes The L2ZienkiewiczZhuEstimator assumes ownership of this
|
||||
FiniteElementSpace and will call its Update() method when
|
||||
@@ -382,8 +382,8 @@ public:
|
||||
{ }
|
||||
|
||||
/** @brief Construct a new L2ZienkiewiczZhuEstimator object.
|
||||
@param integ This BilinearFormIntegrator must implement the methods
|
||||
ComputeElementFlux() and ComputeFluxEnergy().
|
||||
@param integ This BilinearFormIntegrator must implement the method
|
||||
ComputeElementFlux().
|
||||
@param sol The solution field whose error is to be estimated.
|
||||
@param flux_fes The L2ZienkiewiczZhuEstimator does NOT assume ownership
|
||||
of this FiniteElementSpace; will call its Update() method
|
||||
|
||||
+3
-3
@@ -349,7 +349,7 @@ public:
|
||||
vector-valued finite elements, which is also the width of the
|
||||
DenseMatrix argument in
|
||||
CalcPhysVShape(ElementTransformation &Trans, DenseMatrix &shape). */
|
||||
int GetPhysRangeDim(int /* space_dim */) const { return vdim; }
|
||||
virtual int GetPhysRangeDim(int /* space_dim */) const { return vdim; }
|
||||
|
||||
/** Returns the dimension of the curl for vector-valued finite elements,
|
||||
which is also the width of the DenseMatrix argument in
|
||||
@@ -360,7 +360,7 @@ public:
|
||||
finite elements, which is also the width of the DenseMatrix argument in
|
||||
CalcPhysCurlShape(ElementTransformation &Trans, DenseMatrix &curl_shape).
|
||||
*/
|
||||
int GetPhysCurlDim(int /* space_dim */) const { return cdim; }
|
||||
virtual int GetPhysCurlDim(int /* space_dim */) const { return cdim; }
|
||||
|
||||
/// Returns the Geometry::Type of the reference element.
|
||||
Geometry::Type GetGeomType() const { return geom_type; }
|
||||
@@ -1017,7 +1017,7 @@ public:
|
||||
VectorFiniteElement(int D, Geometry::Type G, int Do, int O, int M,
|
||||
int F = FunctionSpace::Pk);
|
||||
|
||||
int GetPhysRangeDim(int space_dim) const { return space_dim; }
|
||||
int GetPhysRangeDim(int space_dim) const override { return space_dim; }
|
||||
};
|
||||
|
||||
/// @brief Class for computing 1D special polynomials and their associated basis
|
||||
|
||||
+4
-4
@@ -663,8 +663,8 @@ public:
|
||||
const int cb_type = BasisType::GaussLobatto,
|
||||
const int ob_type = BasisType::GaussLegendre);
|
||||
|
||||
int GetPhysRangeDim(int space_dim) const { return 2; }
|
||||
int GetPhysCurlDim(int space_dim) const { return 1; }
|
||||
int GetPhysRangeDim(int space_dim) const override { return 2; }
|
||||
int GetPhysCurlDim(int space_dim) const override { return 1; }
|
||||
|
||||
void CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const override;
|
||||
@@ -708,8 +708,8 @@ private:
|
||||
DenseMatrix &I) const;
|
||||
|
||||
public:
|
||||
int GetPhysRangeDim(int space_dim) const { return 3; }
|
||||
int GetPhysCurlDim(int space_dim) const { return 3; }
|
||||
int GetPhysRangeDim(int space_dim) const override { return 3; }
|
||||
int GetPhysCurlDim(int space_dim) const override { return 3; }
|
||||
|
||||
using FiniteElement::CalcVShape;
|
||||
using FiniteElement::CalcPhysCurlShape;
|
||||
|
||||
+4
-4
@@ -510,8 +510,8 @@ public:
|
||||
RT_R2D_SegmentElement(const int p,
|
||||
const int ob_type = BasisType::GaussLegendre);
|
||||
|
||||
int GetPhysRangeDim(int space_dim) const { return 2; }
|
||||
int GetPhysCurlDim(int space_dim) const { return 0; }
|
||||
int GetPhysRangeDim(int space_dim) const override { return 2; }
|
||||
int GetPhysCurlDim(int space_dim) const override { return 0; }
|
||||
|
||||
void CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const override;
|
||||
@@ -550,8 +550,8 @@ private:
|
||||
DenseMatrix &I) const;
|
||||
|
||||
public:
|
||||
int GetPhysRangeDim(int space_dim) const { return 3; }
|
||||
int GetPhysCurlDim(int space_dim) const { return 0; }
|
||||
int GetPhysRangeDim(int space_dim) const override { return 3; }
|
||||
int GetPhysCurlDim(int space_dim) const override { return 0; }
|
||||
|
||||
using FiniteElement::CalcVShape;
|
||||
|
||||
|
||||
+1
-1
@@ -52,7 +52,7 @@
|
||||
#include "bounds.hpp"
|
||||
#include "particleset.hpp"
|
||||
|
||||
#include "dfem/doperator.hpp"
|
||||
// #include "dfem/doperator.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include "pfespace.hpp"
|
||||
|
||||
@@ -3934,6 +3934,16 @@ const FiniteElement *FiniteElementSpace::GetBE(int i) const
|
||||
return BE;
|
||||
}
|
||||
|
||||
const FiniteElement *FiniteElementSpace::GetTypicalBE() const
|
||||
{
|
||||
if (mesh->GetNBE() > 0) { return GetBE(0); }
|
||||
|
||||
Geometry::Type geom = mesh->GetTypicalFaceGeometry();
|
||||
const FiniteElement *be = fec->FiniteElementForGeometry(geom);
|
||||
MFEM_VERIFY(be != nullptr, "Could not determine a typical BE!");
|
||||
return be;
|
||||
}
|
||||
|
||||
const FiniteElement *FiniteElementSpace::GetFaceElement(int i) const
|
||||
{
|
||||
MFEM_VERIFY(!IsVariableOrder(), "not implemented");
|
||||
@@ -3964,6 +3974,11 @@ const FiniteElement *FiniteElementSpace::GetFaceElement(int i) const
|
||||
return fe;
|
||||
}
|
||||
|
||||
const FiniteElement *FiniteElementSpace::GetTypicalFaceElement() const
|
||||
{
|
||||
return fec->FiniteElementForGeometry(mesh->GetTypicalFaceGeometry());
|
||||
}
|
||||
|
||||
const FiniteElement *FiniteElementSpace::GetEdgeElement(int i,
|
||||
int variant) const
|
||||
{
|
||||
|
||||
+13
-1
@@ -839,7 +839,7 @@ public:
|
||||
Note: For vector-valued elements, the results pads up the range dimension
|
||||
to the spatial dimension. E.g., consider a stack of 5 vector-valued
|
||||
elements each representing 2D vectors, living in a 3 dimensional space.
|
||||
Then this fucntion would give 15, not 10.
|
||||
Then this function would give 15, not 10.
|
||||
*/
|
||||
int GetVectorDim() const;
|
||||
|
||||
@@ -1323,12 +1323,24 @@ public:
|
||||
associated with i'th boundary face in the mesh object. */
|
||||
const FiniteElement *GetBE(int i) const;
|
||||
|
||||
/// @brief Return a "typical" boundary element.
|
||||
///
|
||||
/// This can be used in situations where the local mesh partition may be
|
||||
/// empty.
|
||||
const FiniteElement *GetTypicalBE() const;
|
||||
|
||||
/** @brief Returns pointer to the FiniteElement in the FiniteElementCollection
|
||||
associated with i'th face in the mesh object. Faces in this case refer
|
||||
to the MESHDIM-1 primitive so in 2D they are segments and in 1D they are
|
||||
points.*/
|
||||
const FiniteElement *GetFaceElement(int i) const;
|
||||
|
||||
/// @brief Return a "typical" face element.
|
||||
///
|
||||
/// This can be used in situations where the local mesh partition may be
|
||||
/// empty.
|
||||
const FiniteElement *GetTypicalFaceElement() const;
|
||||
|
||||
/** @brief Returns pointer to the FiniteElement in the FiniteElementCollection
|
||||
associated with i'th edge in the mesh object. */
|
||||
const FiniteElement *GetEdgeElement(int i, int variant = 0) const;
|
||||
|
||||
+84
-74
@@ -345,27 +345,6 @@ void GridFunction::ComputeFlux(BilinearFormIntegrator &blfi,
|
||||
}
|
||||
}
|
||||
|
||||
int GridFunction::VectorDim() const
|
||||
{
|
||||
const FiniteElement *fe = fes->GetTypicalFE();
|
||||
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
|
||||
{
|
||||
return fes->GetVDim();
|
||||
}
|
||||
return fes->GetVDim()*std::max(fes->GetMesh()->SpaceDimension(),
|
||||
fe->GetRangeDim());
|
||||
}
|
||||
|
||||
int GridFunction::CurlDim() const
|
||||
{
|
||||
const FiniteElement *fe = fes->GetTypicalFE();
|
||||
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
|
||||
{
|
||||
return 2 * fes->GetMesh()->SpaceDimension() - 3;
|
||||
}
|
||||
return fes->GetVDim()*fe->GetCurlDim();
|
||||
}
|
||||
|
||||
void GridFunction::GetTrueDofs(Vector &tv) const
|
||||
{
|
||||
const SparseMatrix *R = fes->GetRestrictionMatrix();
|
||||
@@ -2050,6 +2029,18 @@ void GridFunction::AccumulateAndCountBdrValues(
|
||||
Coefficient *coeff[], VectorCoefficient *vcoeff, const Array<int> &attr,
|
||||
Array<int> &values_counter)
|
||||
{
|
||||
if (vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
|
||||
"vcoeff vdim != fes VDim");
|
||||
MFEM_VERIFY(fes->GetTypicalBE()->GetMapType() == FiniteElement::VALUE &&
|
||||
fes->GetTypicalBE()->GetRangeType() ==
|
||||
FiniteElement::SCALAR,
|
||||
"Can only call ProjectBdrCoefficient on scalar value-type "
|
||||
"boundary elements. "
|
||||
"Did you intended to call ProjectBdrCoefficientNormal or "
|
||||
"ProjectBdrCoefficientTangent for vector finite elements?");
|
||||
}
|
||||
Array<int> vdofs;
|
||||
Vector vc;
|
||||
|
||||
@@ -2202,6 +2193,9 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
|
||||
VectorCoefficient &vcoeff, const Array<int> &bdr_attr,
|
||||
Array<int> &values_counter)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetTypicalBE()->GetPhysRangeDim(
|
||||
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
|
||||
"vcoeff vdim != PhysRangeDim");
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
@@ -2355,6 +2349,9 @@ void GridFunction::ProjectDeltaCoefficient(DeltaCoefficient &delta_coeff,
|
||||
|
||||
void GridFunction::ProjectCoefficient(Coefficient &coeff, ProjectType type)
|
||||
{
|
||||
MFEM_VERIFY(
|
||||
VectorDim() == 1,
|
||||
"Cannot project scalar Coefficient onto vector GridFunction");
|
||||
DeltaCoefficient *delta_c = dynamic_cast<DeltaCoefficient *>(&coeff);
|
||||
DofTransformation doftrans;
|
||||
Array<int> vdofs;
|
||||
@@ -2630,6 +2627,7 @@ void GridFunction::ProjectCoefficient(
|
||||
void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
|
||||
ProjectType type)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
|
||||
Array<int> vdofs;
|
||||
Vector vals;
|
||||
DofTransformation doftrans;
|
||||
@@ -2945,6 +2943,7 @@ void GridFunction::ProjectCoefficientElementL2(VectorCoefficient &vcoeff)
|
||||
void GridFunction::ProjectCoefficient(
|
||||
VectorCoefficient &vcoeff, Array<int> &dofs)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
|
||||
int el = -1;
|
||||
ElementTransformation *T = NULL;
|
||||
const FiniteElement *fe = NULL;
|
||||
@@ -2974,6 +2973,7 @@ void GridFunction::ProjectCoefficient(
|
||||
|
||||
void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff, int attribute)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
|
||||
int i;
|
||||
Array<int> vdofs;
|
||||
Vector vals;
|
||||
@@ -3030,9 +3030,14 @@ void GridFunction::ProjectCoefficient(Coefficient *coeff[])
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
|
||||
Array<int> &dof_attr)
|
||||
void GridFunction::ProjectDiscCoefficient(
|
||||
std::variant<Coefficient*, VectorCoefficient*> coeff, Array<int> &dof_attr)
|
||||
{
|
||||
std::visit([&](auto* c)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == c->GetVDim(), "coeff vdim != VectorDim()");
|
||||
}, coeff);
|
||||
|
||||
Array<int> vdofs;
|
||||
Vector vals;
|
||||
|
||||
@@ -3046,7 +3051,10 @@ void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
|
||||
{
|
||||
fes->GetElementVDofs(i, vdofs);
|
||||
vals.SetSize(vdofs.Size());
|
||||
fes->GetFE(i)->Project(coeff, *fes->GetElementTransformation(i), vals);
|
||||
std::visit([&](auto* c)
|
||||
{
|
||||
fes->GetFE(i)->Project(*c, *fes->GetElementTransformation(i), vals);
|
||||
}, coeff);
|
||||
|
||||
// the values in shared dofs are determined from the element with maximal
|
||||
// attribute
|
||||
@@ -3062,17 +3070,15 @@ void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
|
||||
{
|
||||
Array<int> dof_attr;
|
||||
ProjectDiscCoefficient(coeff, dof_attr);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
|
||||
{
|
||||
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
|
||||
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
|
||||
|
||||
MFEM_VERIFY(
|
||||
VectorDim() == 1,
|
||||
"Cannot project a scalar coefficient onto a vector GridFunction");
|
||||
|
||||
Array<int> zones_per_vdof;
|
||||
AccumulateAndCountZones(coeff, type, zones_per_vdof);
|
||||
|
||||
@@ -3082,6 +3088,7 @@ void GridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
|
||||
void GridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff,
|
||||
AvgType type)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == coeff.GetVDim(), "coeff vdim != VectorDim()");
|
||||
Array<int> zones_per_vdof;
|
||||
AccumulateAndCountZones(coeff, type, zones_per_vdof);
|
||||
|
||||
@@ -3137,52 +3144,33 @@ void GridFunction::ProjectBdrCoefficient(Coefficient *coeff[],
|
||||
}
|
||||
|
||||
void GridFunction::ProjectBdrCoefficientNormal(
|
||||
VectorCoefficient &vcoeff, const Array<int> &bdr_attr)
|
||||
Coefficient *coeff, VectorCoefficient *vcoeff, const Array<int> &bdr_attr)
|
||||
{
|
||||
#if 0
|
||||
// implementation for the case when the face dofs are integrals of the
|
||||
// normal component.
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
int dim = vcoeff.GetVDim();
|
||||
Vector vc(dim), nor(dim), lvec, shape;
|
||||
|
||||
for (int i = 0; i < fes->GetNBE(); i++)
|
||||
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
|
||||
MFEM_VERIFY(fes->GetTypicalBE()->GetRangeType() == FiniteElement::SCALAR &&
|
||||
fes->GetTypicalBE()->GetMapType() == FiniteElement::INTEGRAL,
|
||||
"Not an RT FE space!");
|
||||
if (vcoeff)
|
||||
{
|
||||
if (bdr_attr[fes->GetBdrAttribute(i)-1] == 0)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
fe = fes->GetBE(i);
|
||||
T = fes->GetBdrElementTransformation(i);
|
||||
int intorder = 2*fe->GetOrder(); // !!!
|
||||
const IntegrationRule &ir = IntRules.Get(fe->GetGeomType(), intorder);
|
||||
int nd = fe->GetDof();
|
||||
lvec.SetSize(nd);
|
||||
shape.SetSize(nd);
|
||||
lvec = 0.0;
|
||||
for (int j = 0; j < ir.GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
T->SetIntPoint(&ip);
|
||||
vcoeff.Eval(vc, *T, ip);
|
||||
CalcOrtho(T->Jacobian(), nor);
|
||||
fe->CalcShape(ip, shape);
|
||||
lvec.Add(ip.weight * (vc * nor), shape);
|
||||
}
|
||||
fes->GetBdrElementDofs(i, dofs);
|
||||
SetSubVector(dofs, lvec);
|
||||
MFEM_VERIFY(vcoeff->GetVDim() == fes->GetMesh()->SpaceDimension(),
|
||||
"vcoeff vdim (" << vcoeff->GetVDim()
|
||||
<< ") != SpaceDimension ("
|
||||
<< fes->GetMesh()->SpaceDimension() << ")");
|
||||
}
|
||||
#else
|
||||
|
||||
// implementation for the case when the face dofs are scaled point
|
||||
// values of the normal component.
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
int dim = vcoeff.GetVDim();
|
||||
Vector vc(dim), nor(dim), lvec;
|
||||
Vector vc, nor, lvec;
|
||||
DofTransformation doftrans;
|
||||
if (vcoeff)
|
||||
{
|
||||
const int dim = vcoeff->GetVDim();
|
||||
vc.SetSize(dim);
|
||||
nor.SetSize(dim);
|
||||
}
|
||||
|
||||
for (int i = 0; i < fes->GetNBE(); i++)
|
||||
{
|
||||
@@ -3198,15 +3186,22 @@ void GridFunction::ProjectBdrCoefficientNormal(
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
T->SetIntPoint(&ip);
|
||||
vcoeff.Eval(vc, *T, ip);
|
||||
CalcOrtho(T->Jacobian(), nor);
|
||||
lvec(j) = (vc * nor);
|
||||
if (coeff)
|
||||
{
|
||||
const real_t c = coeff->Eval(*T, ip);
|
||||
lvec(j) = c * T->Weight();
|
||||
}
|
||||
else if (vcoeff)
|
||||
{
|
||||
vcoeff->Eval(vc, *T, ip);
|
||||
CalcOrtho(T->Jacobian(), nor);
|
||||
lvec(j) = (vc * nor);
|
||||
}
|
||||
}
|
||||
fes->GetBdrElementDofs(i, dofs, doftrans);
|
||||
doftrans.TransformPrimal(lvec);
|
||||
SetSubVector(dofs, lvec);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void GridFunction::ProjectBdrCoefficientTangent(
|
||||
@@ -5007,6 +5002,14 @@ real_t ExtrudeCoefficient::Eval(ElementTransformation &T,
|
||||
return sol_in.Eval(*T_in, ip);
|
||||
}
|
||||
|
||||
void VectorExtrudeCoefficient::Eval(Vector &v, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
ElementTransformation *T_in =
|
||||
mesh_in->GetElementTransformation(T.ElementNo / n);
|
||||
T_in->SetIntPoint(&ip);
|
||||
sol_in.Eval(v, *T_in, ip);
|
||||
}
|
||||
|
||||
GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
|
||||
GridFunction *sol, const int ny)
|
||||
@@ -5057,10 +5060,17 @@ GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
|
||||
return NULL;
|
||||
}
|
||||
FiniteElementSpace *solfes2d;
|
||||
// assuming sol is scalar
|
||||
solfes2d = new FiniteElementSpace(mesh2d, solfec2d);
|
||||
const int vdim = sol->FESpace()->GetVDim();
|
||||
solfes2d = new FiniteElementSpace(mesh2d, solfec2d, vdim);
|
||||
sol2d = new GridFunction(solfes2d);
|
||||
sol2d->MakeOwner(solfec2d);
|
||||
if (vdim > 1)
|
||||
{
|
||||
VectorGridFunctionCoefficient vcsol(sol);
|
||||
VectorExtrudeCoefficient vc2d(mesh, vcsol, ny);
|
||||
sol2d->ProjectCoefficient(vc2d);
|
||||
}
|
||||
else
|
||||
{
|
||||
GridFunctionCoefficient csol(sol);
|
||||
ExtrudeCoefficient c2d(mesh, csol, ny);
|
||||
@@ -5758,4 +5768,4 @@ std::pair<real_t, real_t> GridFunction::EstimateFunctionMaximum(
|
||||
return std::make_pair(global_max_lower, global_max_upper);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
+90
-18
@@ -23,6 +23,7 @@
|
||||
#include <limits>
|
||||
#include <ostream>
|
||||
#include <string>
|
||||
#include <variant>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -79,10 +80,18 @@ protected:
|
||||
bool wcoef,
|
||||
int subdomain);
|
||||
|
||||
/** Project a discontinuous vector coefficient in a continuous space and
|
||||
return in dof_attr the maximal attribute of the elements containing each
|
||||
degree of freedom. */
|
||||
void ProjectDiscCoefficient(VectorCoefficient &coeff, Array<int> &dof_attr);
|
||||
/** @brief Project a discontinuous (vector) coefficient as a grid function on
|
||||
a continuous finite element space. Return in dof_attr the maximal
|
||||
attribute of the elements containing each degree of freedom. */
|
||||
virtual void ProjectDiscCoefficient(
|
||||
std::variant<Coefficient*, VectorCoefficient*> coeff, Array<int> &dof_attr);
|
||||
|
||||
/** @brief Project a discontinuous (vector) coefficient as a grid function on
|
||||
a continuous finite element space. The values in shared dofs are
|
||||
determined from the element with maximal attribute. */
|
||||
virtual void ProjectDiscCoefficient(
|
||||
std::variant<Coefficient*, VectorCoefficient*> coeff)
|
||||
{ Array<int> dof_attr; ProjectDiscCoefficient(coeff, dof_attr); };
|
||||
|
||||
/** Helper function for ProjectCoefficientElementL2 */
|
||||
void ProjectCoefficientElementL2_(Coefficient &coeff, Vector &sol, Vector &Va);
|
||||
@@ -150,11 +159,13 @@ public:
|
||||
|
||||
FiniteElementCollection *OwnFEC() { return fec_owned; }
|
||||
|
||||
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the underlying #fes
|
||||
int VectorDim() const;
|
||||
/** @brief Shortcut for calling FiniteElementSpace::GetVectorDim() on the
|
||||
underlying #fes */
|
||||
int VectorDim() const { return fes->GetVectorDim(); }
|
||||
|
||||
/// Shortcut for calling FiniteElementSpace::GetCurlDim() on the underlying #fes
|
||||
int CurlDim() const;
|
||||
/** @brief Shortcut for calling FiniteElementSpace::GetCurlDim() on the
|
||||
underlying #fes */
|
||||
int CurlDim() const { return fes->GetCurlDim(); }
|
||||
|
||||
/// Read only access to the (optional) internal true-dof Vector.
|
||||
const Vector &GetTrueVector() const
|
||||
@@ -513,10 +524,17 @@ public:
|
||||
but using an array of scalar coefficients for each component. */
|
||||
void ProjectCoefficient(Coefficient *coeff[]);
|
||||
|
||||
/** @brief Project a discontinuous coefficient as a grid function on
|
||||
a continuous finite element space. The values in shared dofs are
|
||||
determined from the element with maximal attribute. */
|
||||
virtual void ProjectDiscCoefficient(Coefficient &coeff)
|
||||
{ ProjectDiscCoefficient(&coeff); }
|
||||
|
||||
/** @brief Project a discontinuous vector coefficient as a grid function on
|
||||
a continuous finite element space. The values in shared dofs are
|
||||
determined from the element with maximal attribute. */
|
||||
virtual void ProjectDiscCoefficient(VectorCoefficient &coeff);
|
||||
virtual void ProjectDiscCoefficient(VectorCoefficient &coeff)
|
||||
{ ProjectDiscCoefficient(&coeff); }
|
||||
|
||||
enum AvgType {ARITHMETIC, HARMONIC};
|
||||
/** @brief Projects a discontinuous coefficient so that the values in shared
|
||||
@@ -532,6 +550,9 @@ public:
|
||||
std::unique_ptr<GridFunction> ProlongateToMaxOrder() const;
|
||||
|
||||
protected:
|
||||
void ProjectBdrCoefficientNormal(Coefficient *coeff, VectorCoefficient *vcoeff,
|
||||
const Array<int> &attr);
|
||||
|
||||
/** @brief Accumulates (depending on @a type) the values of @a coeff at all
|
||||
shared vdofs and counts in how many zones each vdof appears. */
|
||||
void AccumulateAndCountZones(Coefficient &coeff, AvgType type,
|
||||
@@ -656,15 +677,26 @@ public:
|
||||
virtual void ProjectBdrCoefficient(Coefficient *coeff[],
|
||||
const Array<int> &attr);
|
||||
|
||||
/** Project the normal component of the given VectorCoefficient on
|
||||
the boundary. Only boundary attributes that are marked in
|
||||
'bdr_attr' are projected. Assumes RT-type VectorFE GridFunction. */
|
||||
/** @brief Project the normal component of the given VectorCoefficient on
|
||||
the boundary. */
|
||||
/** Only boundary attributes that are marked in @a bdr_attr are
|
||||
projected. Assumes RT-type vector finite element GridFunction. */
|
||||
void ProjectBdrCoefficientNormal(VectorCoefficient &vcoeff,
|
||||
const Array<int> &bdr_attr);
|
||||
const Array<int> &bdr_attr)
|
||||
{ ProjectBdrCoefficientNormal(NULL, &vcoeff, bdr_attr); }
|
||||
|
||||
/** @brief Project the given Coefficient in the normal direction on the
|
||||
boundary. */
|
||||
/** Only boundary attributes that are marked in @a bdr_attr are projected.
|
||||
Assumes RT-type vector finite element GridFunction. */
|
||||
void ProjectBdrCoefficientNormal(Coefficient &coeff,
|
||||
const Array<int> &bdr_attr)
|
||||
{ ProjectBdrCoefficientNormal(&coeff, NULL, bdr_attr); }
|
||||
|
||||
/** @brief Project the tangential components of the given VectorCoefficient
|
||||
on the boundary. Only boundary attributes that are marked in @a bdr_attr
|
||||
are projected. Assumes ND-type VectorFE GridFunction. */
|
||||
on the boundary. */
|
||||
/** Only boundary attributes that are marked in @a bdr_attr
|
||||
are projected. Assumes ND-type vector finite element GridFunction. */
|
||||
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
|
||||
const Array<int> &bdr_attr);
|
||||
|
||||
@@ -1914,7 +1946,7 @@ real_t ComputeElementLpDistance(real_t p, int i,
|
||||
GridFunction& gf1, GridFunction& gf2);
|
||||
|
||||
|
||||
/// Class used for extruding scalar GridFunctions
|
||||
/// Class used for extruding a scalar coefficient
|
||||
class ExtrudeCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
@@ -1922,13 +1954,53 @@ private:
|
||||
Mesh *mesh_in;
|
||||
Coefficient &sol_in;
|
||||
public:
|
||||
/// Constructs an instance of VectorExtrudeCoefficient
|
||||
/**
|
||||
* @param m 1D mesh
|
||||
* @param s 1D vector coefficient
|
||||
* @param n_ number of transverse elements of the extruded mesh
|
||||
*/
|
||||
ExtrudeCoefficient(Mesh *m, Coefficient &s, int n_)
|
||||
: n(n_), mesh_in(m), sol_in(s) { }
|
||||
: n(n_), mesh_in(m), sol_in(s)
|
||||
{ MFEM_VERIFY(n > 0, "Number of transverse elements must be positive!"); }
|
||||
|
||||
real_t Eval(ElementTransformation &T, const IntegrationPoint &ip) override;
|
||||
|
||||
virtual ~ExtrudeCoefficient() { }
|
||||
};
|
||||
|
||||
/// Extrude a scalar 1D GridFunction, after extruding the mesh with Extrude1D.
|
||||
/// Class used for extruding a vector coefficient
|
||||
class VectorExtrudeCoefficient : public VectorCoefficient
|
||||
{
|
||||
private:
|
||||
int n;
|
||||
Mesh *mesh_in;
|
||||
VectorCoefficient &sol_in;
|
||||
public:
|
||||
/// Constructs an instance of VectorExtrudeCoefficient
|
||||
/**
|
||||
* @param m 1D mesh
|
||||
* @param s 1D vector coefficient
|
||||
* @param n_ number of transverse elements of the extruded mesh
|
||||
*/
|
||||
VectorExtrudeCoefficient(Mesh *m, VectorCoefficient &s, int n_)
|
||||
: VectorCoefficient(s.GetVDim()), n(n_), mesh_in(m), sol_in(s)
|
||||
{ MFEM_VERIFY(n > 0, "Number of transverse elements must be positive!"); }
|
||||
|
||||
void Eval(Vector &v, ElementTransformation &T,
|
||||
const IntegrationPoint &ip) override;
|
||||
using VectorCoefficient::Eval;
|
||||
|
||||
virtual ~VectorExtrudeCoefficient() { }
|
||||
};
|
||||
|
||||
/// Extrude a 1D GridFunction, after extruding the mesh with Extrude1D()
|
||||
/**
|
||||
* @param mesh 1D mesh
|
||||
* @param mesh2d extruded mesh
|
||||
* @param sol grid function
|
||||
* @param ny number of transverse elements of the extruded mesh
|
||||
*/
|
||||
GridFunction *Extrude1DGridFunction(Mesh *mesh, Mesh *mesh2d,
|
||||
GridFunction *sol, const int ny);
|
||||
|
||||
|
||||
+6
-5
@@ -490,7 +490,7 @@ void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
|
||||
}
|
||||
DEV.find_device = true;
|
||||
|
||||
const int id = gsl_comm->id, np = gsl_comm->np;
|
||||
const unsigned int id = gsl_comm->id, np = gsl_comm->np;
|
||||
|
||||
gsl_mfem_ref.SetSize(points_cnt * dim);
|
||||
gsl_mfem_elem.SetSize(points_cnt);
|
||||
@@ -652,7 +652,7 @@ void FindPointsGSLIB::FindPointsOnDevice(const Vector &point_pos,
|
||||
{
|
||||
const int pp = hash_offset[i];
|
||||
/* don't send back to where it just came from */
|
||||
if (pp == p->proc)
|
||||
if (static_cast<unsigned>(pp) == p->proc)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
@@ -1068,7 +1068,7 @@ void FindPointsGSLIB::InterpolateOnDevice(const Vector &field_in_evec,
|
||||
sarray_transfer(struct evalOutPt_t, &outpt, proc, 1, cr);
|
||||
|
||||
opt = (evalOutPt_t *)outpt.ptr;
|
||||
for (int index = 0; index < outpt.n; index++)
|
||||
for (size_t index = 0; index < outpt.n; index++)
|
||||
{
|
||||
int idx = ordering == Ordering::byNODES ?
|
||||
opt->index + i*points_cnt :
|
||||
@@ -1413,7 +1413,7 @@ void FindPointsGSLIB::SetupSplitMeshesAndIntegrationRules(const int order)
|
||||
{
|
||||
MFEM_VERIFY(mesh, "Setup FindPointsGSLIB with mesh first.");
|
||||
const int dof1D = order+1;
|
||||
const int dim = mesh->Dimension();
|
||||
dim = mesh->Dimension();
|
||||
|
||||
SetupSplitMeshes();
|
||||
if (dim == 2)
|
||||
@@ -2254,7 +2254,8 @@ void FindPointsGSLIB::DistributeInterpolatedValues(const Vector &int_vals,
|
||||
sarray_transfer(struct out_pt, outpt, proc, 1, cr);
|
||||
|
||||
// Store received data
|
||||
MFEM_VERIFY(outpt->n == points_cnt, "Incompatible size. Number of points "
|
||||
MFEM_VERIFY(outpt->n == static_cast<size_t>(points_cnt),
|
||||
"Incompatible size. Number of points "
|
||||
"received does not match the number of points originally "
|
||||
"found using FindPoints.");
|
||||
|
||||
|
||||
@@ -202,13 +202,19 @@ protected:
|
||||
const int dof1dsol, const int ordering);
|
||||
|
||||
public:
|
||||
/// Serial constructor
|
||||
FindPointsGSLIB();
|
||||
|
||||
/// Serial constructor + setup with given Mesh (see \ref Setup)
|
||||
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Constructor for ParMesh
|
||||
FindPointsGSLIB(MPI_Comm comm_);
|
||||
|
||||
/// Constructor + setup with given ParMesh (see \ref Setup)
|
||||
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
@@ -254,7 +254,7 @@ get_edge(const double *elx[2], const double *wtend, int ei,
|
||||
edge.dxdn[d] = workspace + (2 + d) * pN; //dxdn and dydn at DOFs along edge
|
||||
}
|
||||
|
||||
if (side_init != (1u << ei))
|
||||
if (static_cast<unsigned>(side_init) != (1u << ei))
|
||||
{
|
||||
#define ELX(d, j, k) elx[d][j + k * pN] // assumes lexicographic ordering
|
||||
for (int d = 0; d < 2; ++d)
|
||||
|
||||
@@ -294,7 +294,7 @@ get_face(const double *elx[3], const double *wtend, int fi, double *workspace,
|
||||
face.dxdn[d] = workspace+(3+d)*p_Nfr;
|
||||
}
|
||||
|
||||
if (side_init != (1u << fi))
|
||||
if (static_cast<unsigned>(side_init) != (1u << fi))
|
||||
{
|
||||
const int e_stride[3] = {1, pN, pN*pN};
|
||||
#define ELX(d, j, k, l) elx[d][j*e_stride[d1]+k*e_stride[d2]+l*e_stride[dn]]
|
||||
@@ -342,7 +342,7 @@ get_edge(const double *elx[3], const double *wtend, int ei, double *workspace,
|
||||
|
||||
if (jidx >= 3*pN) { return edge; }
|
||||
|
||||
if (side_init != (64u << ei))
|
||||
if (static_cast<unsigned>(side_init) != (64u << ei))
|
||||
{
|
||||
const int e_stride[3] = {1, pN, pN*pN};
|
||||
#define ELX(d, j, k, l) elx[d][j*e_stride[de]+k*e_stride[dn1]+l*e_stride[dn2]]
|
||||
|
||||
@@ -1064,6 +1064,8 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad X
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
|
||||
@@ -1084,6 +1086,8 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad Y
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
|
||||
@@ -1105,6 +1109,8 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad Z + Q-function
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
|
||||
@@ -1217,20 +1223,23 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
|
||||
namespace
|
||||
{
|
||||
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
|
||||
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
|
||||
using DiffusionApplyKernelType =
|
||||
DiffusionIntegrator::DiffusionApplyKernelType;
|
||||
|
||||
using DiffusionDiagonalKernelType =
|
||||
DiffusionIntegrator::DiffusionDiagonalKernelType;
|
||||
}
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
DiffusionApplyKernelType DiffusionIntegrator::DiffusionApplyPAKernel::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
inline DiffusionApplyKernelType
|
||||
DiffusionIntegrator::DiffusionApplyPAKernel::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PADiffusionApply2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionApply3D; }
|
||||
@@ -1238,15 +1247,16 @@ ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
}
|
||||
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
|
||||
DiffusionDiagonalKernelType
|
||||
DiffusionIntegrator::DiffusionDiagonalPAKernel::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline DiagonalKernelType
|
||||
DiffusionIntegrator::DiagonalPAKernels::Fallback(int DIM, int, int)
|
||||
inline DiffusionDiagonalKernelType
|
||||
DiffusionIntegrator::DiffusionDiagonalPAKernel::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
|
||||
|
||||
@@ -31,8 +31,8 @@ void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
const Array<real_t> &B = maps->B;
|
||||
const Array<real_t> &G = maps->G;
|
||||
const Vector &Dv = pa_data;
|
||||
DiagonalPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
|
||||
diag, dofs1D, quad1D);
|
||||
DiffusionDiagonalPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
|
||||
diag, dofs1D, quad1D);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,8 +68,8 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
|
||||
Gt, Dv, x, y, dofs1D, quad1D);
|
||||
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
|
||||
Gt, Dv, x, y, dofs1D, quad1D);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -174,9 +174,9 @@ void DiffusionIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
abs_pa_data.Abs();
|
||||
auto abs_maps = maps->Abs();
|
||||
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
|
||||
abs_pa_data, x, y, dofs1D, quad1D);
|
||||
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
|
||||
abs_pa_data, x, y, dofs1D, quad1D);
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::AddAbsMultTransposePA(const Vector &x,
|
||||
|
||||
@@ -197,15 +197,21 @@ static void EAHdivAssemble3D(const int NE,
|
||||
// Assemble (one row per thread)
|
||||
MFEM_FOREACH_THREAD(idx_i, x, NDOF)
|
||||
{
|
||||
// NOTE: due to an llvm backend bug, usage of the modulus operator
|
||||
// has been removed from this foreach section.
|
||||
const int ic = idx_i / NDOF_C;
|
||||
const int idx_ii = idx_i % NDOF_C;
|
||||
const int idx_ii = idx_i - ic * NDOF_C; // idx_i % NDOF_C
|
||||
|
||||
const int nx_i = (ic == 0) ? D1D : D1D-1;
|
||||
const int ny_i = (ic == 1) ? D1D : D1D-1;
|
||||
|
||||
const int ix = idx_ii % nx_i;
|
||||
const int iy = (idx_ii / nx_i) % ny_i;
|
||||
const int iz = (idx_ii / nx_i) / ny_i;
|
||||
const int qx_i = idx_ii / nx_i;
|
||||
const int ix = idx_ii - qx_i * nx_i; // idx_ii % nx_i
|
||||
|
||||
const int qy_i = qx_i / ny_i;
|
||||
const int iy = qx_i - qy_i * ny_i; // (idx_ii / nx_i) % ny_i
|
||||
|
||||
const int iz = qy_i; // (idx_ii / nx_i) / ny_i
|
||||
|
||||
const real_t (&Bi1)[MQ1][MD1] = (ic == 0) ? r_Bc : r_Bo;
|
||||
const real_t (&Bi2)[MQ1][MD1] = (ic == 1) ? r_Bc : r_Bo;
|
||||
@@ -214,14 +220,18 @@ static void EAHdivAssemble3D(const int NE,
|
||||
for (int idx_j = 0; idx_j < NDOF; ++idx_j)
|
||||
{
|
||||
const int jc = idx_j / NDOF_C;
|
||||
const int idx_jj = idx_j % NDOF_C;
|
||||
const int idx_jj = idx_j - jc * NDOF_C; // idx_j % NDOF_C
|
||||
|
||||
const int nx_j = (jc == 0) ? D1D : D1D-1;
|
||||
const int ny_j = (jc == 1) ? D1D : D1D-1;
|
||||
|
||||
const int jx = idx_jj % nx_j;
|
||||
const int jy = (idx_jj / nx_j) % ny_j;
|
||||
const int jz = (idx_jj / nx_j) / ny_j;
|
||||
const int qx_j = idx_jj / nx_j;
|
||||
const int jx = idx_jj - qx_j * nx_j; // idx_jj % nx_j
|
||||
|
||||
const int qy_j = qx_j / ny_j;
|
||||
const int jy = qx_j - qy_j * ny_j; // (idx_jj / nx_j) % ny_j
|
||||
|
||||
const int jz = qy_j; // (idx_jj / nx_j) / ny_j
|
||||
|
||||
const real_t (&Bj1)[MQ1][MD1] = (jc == 0) ? r_Bc : r_Bo;
|
||||
const real_t (&Bj2)[MQ1][MD1] = (jc == 1) ? r_Bc : r_Bo;
|
||||
|
||||
+811
-327
File diff suppressed because it is too large
Load Diff
+63
-64
@@ -43,56 +43,52 @@ public:
|
||||
index = i;
|
||||
}
|
||||
|
||||
void Set3w(const real_t x1, const real_t x2, const real_t x3, const real_t w)
|
||||
{ x = x1; y = x2; z = x3; weight = w; }
|
||||
void Set2w(const real_t x1, const real_t x2, const real_t w)
|
||||
{ x = x1; y = x2; weight = w; }
|
||||
void Set1w(const real_t x1, const real_t w)
|
||||
{ x = x1; weight = w; }
|
||||
|
||||
void Set3w(const real_t *p) { Set3w(p[0], p[1], p[2], p[3]); }
|
||||
void Set2w(const real_t *p) { Set2w(p[0], p[1], p[2]); }
|
||||
void Set1w(const real_t *p) { Set1w(p[0], p[1]); }
|
||||
|
||||
void Set3(const real_t x1, const real_t x2, const real_t x3)
|
||||
{ x = x1; y = x2; z = x3; }
|
||||
void Set2(const real_t x1, const real_t x2)
|
||||
{ x = x1; y = x2; }
|
||||
void Set1(const real_t x1)
|
||||
{ x = x1; }
|
||||
|
||||
void Set3(const real_t *p) { Set3(p[0], p[1], p[2]); }
|
||||
void Set2(const real_t *p) { Set2(p[0], p[1]); }
|
||||
void Set1(const real_t *p) { Set1(p[0]); }
|
||||
|
||||
void Set(const real_t x1, const real_t x2, const real_t x3, const real_t w)
|
||||
{ Set3w(x1, x2, x3, w); }
|
||||
|
||||
void Set(const real_t *p, const int dim)
|
||||
{
|
||||
MFEM_ASSERT(1 <= dim && dim <= 3, "invalid dim: " << dim);
|
||||
x = p[0];
|
||||
if (dim > 1)
|
||||
switch (dim)
|
||||
{
|
||||
y = p[1];
|
||||
if (dim > 2)
|
||||
{
|
||||
z = p[2];
|
||||
}
|
||||
case 3: Set3(p); break;
|
||||
case 2: Set2(p); break;
|
||||
case 1: Set1(p); break;
|
||||
}
|
||||
}
|
||||
|
||||
void Get(real_t *p, const int dim) const
|
||||
{
|
||||
MFEM_ASSERT(1 <= dim && dim <= 3, "invalid dim: " << dim);
|
||||
p[0] = x;
|
||||
if (dim > 1)
|
||||
switch (dim)
|
||||
{
|
||||
p[1] = y;
|
||||
if (dim > 2)
|
||||
{
|
||||
p[2] = z;
|
||||
}
|
||||
case 3: p[2] = z;
|
||||
case 2: p[1] = y;
|
||||
case 1: p[0] = x;
|
||||
}
|
||||
}
|
||||
|
||||
void Set(const real_t x1, const real_t x2, const real_t x3, const real_t w)
|
||||
{ x = x1; y = x2; z = x3; weight = w; }
|
||||
|
||||
void Set3w(const real_t *p) { x = p[0]; y = p[1]; z = p[2]; weight = p[3]; }
|
||||
|
||||
void Set3(const real_t x1, const real_t x2, const real_t x3)
|
||||
{ x = x1; y = x2; z = x3; }
|
||||
|
||||
void Set3(const real_t *p) { x = p[0]; y = p[1]; z = p[2]; }
|
||||
|
||||
void Set2w(const real_t x1, const real_t x2, const real_t w)
|
||||
{ x = x1; y = x2; weight = w; }
|
||||
|
||||
void Set2w(const real_t *p) { x = p[0]; y = p[1]; weight = p[2]; }
|
||||
|
||||
void Set2(const real_t x1, const real_t x2) { x = x1; y = x2; }
|
||||
|
||||
void Set2(const real_t *p) { x = p[0]; y = p[1]; }
|
||||
|
||||
void Set1w(const real_t x1, const real_t w) { x = x1; weight = w; }
|
||||
|
||||
void Set1w(const real_t *p) { x = p[0]; weight = p[1]; }
|
||||
};
|
||||
|
||||
/// Class for an integration rule - an Array of IntegrationPoint.
|
||||
@@ -125,18 +121,6 @@ private:
|
||||
void AddTriPoints3b(const int off, const real_t b, const real_t weight)
|
||||
{ AddTriPoints3(off, (1. - b)/2., b, weight); }
|
||||
|
||||
void AddTriPoints3R(const int off, const real_t a, const real_t b,
|
||||
const real_t c, const real_t weight)
|
||||
{
|
||||
IntPoint(off + 0).Set2w(a, b, weight);
|
||||
IntPoint(off + 1).Set2w(c, a, weight);
|
||||
IntPoint(off + 2).Set2w(b, c, weight);
|
||||
}
|
||||
|
||||
void AddTriPoints3R(const int off, const real_t a, const real_t b,
|
||||
const real_t weight)
|
||||
{ AddTriPoints3R(off, a, b, 1. - a - b, weight); }
|
||||
|
||||
void AddTriPoints6(const int off, const real_t a, const real_t b,
|
||||
const real_t c, const real_t weight)
|
||||
{
|
||||
@@ -183,14 +167,6 @@ private:
|
||||
AddTetPoints3(off + 1, a, 1. - 3.*a, weight);
|
||||
}
|
||||
|
||||
// given b, add the permutations of (a,a,a,b), where 3*a + b = 1
|
||||
void AddTetPoints4b(const int off, const real_t b, const real_t weight)
|
||||
{
|
||||
const real_t a = (1. - b)/3.;
|
||||
IntPoint(off).Set(a, a, a, weight);
|
||||
AddTetPoints3(off + 1, a, b, weight);
|
||||
}
|
||||
|
||||
// add the permutations of (a,a,b,b), 2*(a + b) = 1
|
||||
void AddTetPoints6(const int off, const real_t a, const real_t weight)
|
||||
{
|
||||
@@ -209,14 +185,37 @@ private:
|
||||
AddTetPoints6(off + 6, a, bc, cb, weight);
|
||||
}
|
||||
|
||||
// given (b,c), add the permutations of (a,a,b,c), 2*a + b + c = 1
|
||||
void AddTetPoints12bc(const int off, const real_t b, const real_t c,
|
||||
const real_t weight)
|
||||
// add all 24 permutations of (a,b,c,d) where a+b+c+d = 1, all distinct
|
||||
void AddTetPoints24(const int off, const real_t a, const real_t b,
|
||||
const real_t c, const real_t weight)
|
||||
{
|
||||
const real_t a = (1. - b - c)/2.;
|
||||
AddTetPoints3(off, a, b, weight);
|
||||
AddTetPoints3(off + 3, a, c, weight);
|
||||
AddTetPoints6(off + 6, a, b, c, weight);
|
||||
const real_t d = 1. - a - b - c;
|
||||
// all 24 permutations of 4 distinct barycentric coordinates
|
||||
// permuting which coordinate goes to x, y, z (4th is 1-x-y-z)
|
||||
IntPoint(off + 0).Set(a, b, c, weight);
|
||||
IntPoint(off + 1).Set(a, b, d, weight);
|
||||
IntPoint(off + 2).Set(a, c, b, weight);
|
||||
IntPoint(off + 3).Set(a, c, d, weight);
|
||||
IntPoint(off + 4).Set(a, d, b, weight);
|
||||
IntPoint(off + 5).Set(a, d, c, weight);
|
||||
IntPoint(off + 6).Set(b, a, c, weight);
|
||||
IntPoint(off + 7).Set(b, a, d, weight);
|
||||
IntPoint(off + 8).Set(b, c, a, weight);
|
||||
IntPoint(off + 9).Set(b, c, d, weight);
|
||||
IntPoint(off + 10).Set(b, d, a, weight);
|
||||
IntPoint(off + 11).Set(b, d, c, weight);
|
||||
IntPoint(off + 12).Set(c, a, b, weight);
|
||||
IntPoint(off + 13).Set(c, a, d, weight);
|
||||
IntPoint(off + 14).Set(c, b, a, weight);
|
||||
IntPoint(off + 15).Set(c, b, d, weight);
|
||||
IntPoint(off + 16).Set(c, d, a, weight);
|
||||
IntPoint(off + 17).Set(c, d, b, weight);
|
||||
IntPoint(off + 18).Set(d, a, b, weight);
|
||||
IntPoint(off + 19).Set(d, a, c, weight);
|
||||
IntPoint(off + 20).Set(d, b, a, weight);
|
||||
IntPoint(off + 21).Set(d, b, c, weight);
|
||||
IntPoint(off + 22).Set(d, c, a, weight);
|
||||
IntPoint(off + 23).Set(d, c, b, weight);
|
||||
}
|
||||
|
||||
public:
|
||||
|
||||
@@ -207,6 +207,28 @@ inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input DIM vector at element offset into given register tensor
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int d1d, const int c,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][d][dz][dy][dx] = X(dx, dy, dz, c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input VDIM*DIM vector into given register tensor, specific component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d, const int c,
|
||||
@@ -332,6 +354,28 @@ inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 3D DIM vector into given device tensor for specific component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int d1d, const int c,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
DeviceTensor<4, real_t> &Y)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
for (int d = 0; d < DIM; ++d)
|
||||
{
|
||||
Y(dx, dy, dz, c) += X(c, d, dz, dy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar contraction, X direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractX2d(const int d1d, const int q1d,
|
||||
|
||||
@@ -0,0 +1,332 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include "kernels.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem::kernels::internal::low
|
||||
{
|
||||
|
||||
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
|
||||
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
|
||||
template <int DIM, int N>
|
||||
// struct regs3d_device_wrapper: mfem::future::tensor<real_t, DIM, 0, 0, 0> {};
|
||||
struct regs3d_device_wrapper: mfem::future::tensor<real_t, 0, 0, 0, DIM> {};
|
||||
template <int DIM, int N>
|
||||
using regs3d_t = regs3d_device_wrapper<DIM, N>;
|
||||
#else
|
||||
template <int DIM, int N>
|
||||
using regs3d_t = mfem::future::tensor<real_t, N, N, N, DIM>;
|
||||
// using regs3d_t = mfem::future::tensor<real_t, DIM, N, N, N>;
|
||||
#endif
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// Load 2D matrix into shared memory
|
||||
template <int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
|
||||
const real_t *M, real_t (*N)[MQ1])
|
||||
{
|
||||
if (MFEM_THREAD_ID(z) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
|
||||
{
|
||||
N[dy][qx] = M[dy * q1d + qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
|
||||
const DeviceTensor<5, const real_t> &XE,
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
sm0[dz][dy][dx][0] = XE(dx, dy, dz, 0, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 1/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradX(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dx = 0; dx < d1d; ++dx)
|
||||
{
|
||||
const auto x = sm0[dz][dy][dx][0];
|
||||
u = std::fma(B[dx][qx], x, u);
|
||||
v = std::fma(G[dx][qx], x, v);
|
||||
}
|
||||
sm1[dz][dy][qx][0] = u;
|
||||
sm1[dz][dy][qx][1] = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 2/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradY(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dy = 0; dy < d1d; ++dy)
|
||||
{
|
||||
u = std::fma(sm1[dz][dy][qx][1], B[dy][qy], u);
|
||||
v = std::fma(sm1[dz][dy][qx][0], G[dy][qy], v);
|
||||
w = std::fma(sm1[dz][dy][qx][0], B[dy][qy], w);
|
||||
}
|
||||
sm0[dz][qy][qx][0] = u;
|
||||
sm0[dz][qy][qx][1] = v;
|
||||
sm0[dz][qy][qx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradZ(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u[3] = {0.0, 0.0, 0.0};
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
u[0] = std::fma(B[dz][qz], sm0[dz][qy][qx][0], u[0]);
|
||||
u[1] = std::fma(B[dz][qz], sm0[dz][qy][qx][1], u[1]);
|
||||
u[2] = std::fma(G[dz][qz], sm0[dz][qy][qx][2], u[2]);
|
||||
}
|
||||
reg[qz][qy][qx][0] = u[0];
|
||||
reg[qz][qy][qx][1] = u[1];
|
||||
reg[qz][qy][qx][2] = u[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D scalar gradient
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
GradX(d1d, q1d, B, G, sm0, sm1); // Grad X
|
||||
GradY(d1d, q1d, B, G, sm1, sm0); // Grad Y
|
||||
GradZ(d1d, q1d, B, G, sm0, reg); // Grad Z
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 1/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dX(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
sm1[qz][qy][qx][0] = reg[qz][qy][qx][0];
|
||||
sm1[qz][qy][qx][1] = reg[qz][qy][qx][1];
|
||||
sm1[qz][qy][qx][2] = reg[qz][qy][qx][2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qx = 0; qx < q1d; ++qx)
|
||||
{
|
||||
u = std::fma(sm1[qz][qy][qx][0], G[dx][qx], u);
|
||||
v = std::fma(sm1[qz][qy][qx][1], B[dx][qx], v);
|
||||
w = std::fma(sm1[qz][qy][qx][2], B[dx][qx], w);
|
||||
}
|
||||
sm0[qz][qy][dx][0] = u;
|
||||
sm0[qz][qy][dx][1] = v;
|
||||
sm0[qz][qy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 2/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dY(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qy = 0; qy < q1d; ++qy)
|
||||
{
|
||||
u = std::fma(sm0[qz][qy][dx][0], B[dy][qy], u);
|
||||
v = std::fma(sm0[qz][qy][dx][1], G[dy][qy], v);
|
||||
w = std::fma(sm0[qz][qy][dx][2], B[dy][qy], w);
|
||||
}
|
||||
sm1[qz][dy][dx][0] = u;
|
||||
sm1[qz][dy][dx][1] = v;
|
||||
sm1[qz][dy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dZ(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < q1d; ++qz)
|
||||
{
|
||||
u = std::fma(sm1[qz][dy][dx][0], B[dz][qz], u);
|
||||
v = std::fma(sm1[qz][dy][dx][1], B[dz][qz], v);
|
||||
w = std::fma(sm1[qz][dy][dx][2], G[dz][qz], w);
|
||||
}
|
||||
reg[dz][dy][dx][0] = u;
|
||||
reg[dz][dy][dx][1] = v;
|
||||
reg[dz][dy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D scalar gradient transposed
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
GradTranspose3dX(d1d, q1d, B, G, reg, sm1, sm0); // Grad^T X
|
||||
GradTranspose3dY(d1d, q1d, B, G, sm0, sm1); // Grad^T Y
|
||||
GradTranspose3dZ(d1d, q1d, B, G, sm1, reg); // Grad^T Z
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int d1d,
|
||||
const int c, const int e,
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
const DeviceTensor<5, real_t> &YE)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
const real_t u = reg[dz][dy][dx][0];
|
||||
const real_t v = reg[dz][dy][dx][1];
|
||||
const real_t w = reg[dz][dy][dx][2];
|
||||
YE(dx, dy, dz, c, e) += (u + v + w);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::kernels::internal
|
||||
+3
-1
@@ -297,7 +297,8 @@ void LinearForm::Assemble()
|
||||
tr = mesh->GetBdrFaceTransformations(i);
|
||||
if (tr != NULL)
|
||||
{
|
||||
fes -> GetElementVDofs (tr -> Elem1No, vdofs);
|
||||
mfem::DofTransformation doftrans;
|
||||
fes -> GetElementVDofs (tr -> Elem1No, vdofs, doftrans);
|
||||
for (int k = 0; k < boundary_face_integs.Size(); k++)
|
||||
{
|
||||
if (boundary_face_integs_marker[k] &&
|
||||
@@ -307,6 +308,7 @@ void LinearForm::Assemble()
|
||||
boundary_face_integs[k]->
|
||||
AssembleRHSElementVect(*fes->GetFE(tr->Elem1No),
|
||||
*tr, elemvect);
|
||||
doftrans.TransformDual(elemvect);
|
||||
AddElementVector (vdofs, elemvect);
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -164,8 +164,8 @@ private:
|
||||
|
||||
public:
|
||||
/// Constructs the domain integrator $ (Q, \nabla v) $
|
||||
DomainLFGradIntegrator(VectorCoefficient &QF)
|
||||
: DeltaLFIntegrator(QF), Q(QF) { }
|
||||
DomainLFGradIntegrator(VectorCoefficient &QF, const IntegrationRule *ir = NULL)
|
||||
: DeltaLFIntegrator(QF, ir), Q(QF) { }
|
||||
|
||||
bool SupportsDevice() const override { return true; }
|
||||
|
||||
|
||||
@@ -224,6 +224,9 @@ public:
|
||||
/** @see GetGradient(const Vector &) */
|
||||
Operator &GetGradient(const Vector &x, bool finalize) const;
|
||||
|
||||
/// Suppress a warning about hiding overloaded virtual function.
|
||||
using Operator::GetGradient;
|
||||
|
||||
/// Update the NonlinearForm to propagate updates of the associated FE space.
|
||||
/** After calling this method, the essential boundary conditions need to be
|
||||
set again. */
|
||||
|
||||
+15
-1
@@ -545,6 +545,8 @@ void ParGridFunction::GetElementDofValues(int el, Vector &dof_vals) const
|
||||
|
||||
void ParGridFunction::ProjectCoefficient(Coefficient &coeff, ProjectType type)
|
||||
{
|
||||
MFEM_VERIFY(VectorDim() == 1,
|
||||
"Cannot project scalar coefficient onto vector ParGridFunction");
|
||||
DeltaCoefficient *delta_c = dynamic_cast<DeltaCoefficient *>(&coeff);
|
||||
|
||||
if (delta_c == NULL)
|
||||
@@ -715,7 +717,8 @@ void ParGridFunction::ProjectCoefficientElementL2(VectorCoefficient &vcoeff)
|
||||
}
|
||||
|
||||
|
||||
void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
|
||||
void ParGridFunction::ProjectDiscCoefficient(
|
||||
std::variant<Coefficient*, VectorCoefficient*> coeff)
|
||||
{
|
||||
// local maximal element attribute for each dof
|
||||
Array<int> ldof_attr;
|
||||
@@ -761,6 +764,9 @@ void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &coeff)
|
||||
|
||||
void ParGridFunction::ProjectDiscCoefficient(Coefficient &coeff, AvgType type)
|
||||
{
|
||||
MFEM_VERIFY(
|
||||
VectorDim() == 1,
|
||||
"Cannot project scalar coefficient onto a vector ParGridFunction");
|
||||
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
|
||||
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
|
||||
|
||||
@@ -786,6 +792,8 @@ void ParGridFunction::ProjectDiscCoefficient(VectorCoefficient &vcoeff,
|
||||
// Harmonic (x1 ... xn) = [ (1/x1 + ... + 1/xn) / n ]^-1.
|
||||
// Arithmetic(x1 ... xn) = (x1 + ... + xn) / n.
|
||||
|
||||
MFEM_VERIFY(VectorDim() == vcoeff.GetVDim(), "vcoeff vdim != VectorDim()");
|
||||
|
||||
// Number of zones that contain a given dof.
|
||||
Array<int> zones_per_vdof;
|
||||
AccumulateAndCountZones(vcoeff, type, zones_per_vdof);
|
||||
@@ -858,6 +866,12 @@ void ParGridFunction::ProjectBdrCoefficient(
|
||||
#endif
|
||||
}
|
||||
|
||||
void ParGridFunction::ProjectBdrCoefficient(VectorCoefficient &vcoeff,
|
||||
const Array<int> &attr)
|
||||
{
|
||||
ProjectBdrCoefficient(NULL, &vcoeff, attr);
|
||||
}
|
||||
|
||||
void ParGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
|
||||
const Array<int> &bdr_attr)
|
||||
{
|
||||
|
||||
+7
-7
@@ -63,6 +63,12 @@ protected:
|
||||
void ProjectBdrCoefficient(Coefficient *coeff[], VectorCoefficient *vcoeff,
|
||||
const Array<int> &attr);
|
||||
|
||||
/** @brief Project a discontinuous (vector) coefficient as a grid function on
|
||||
a continuous finite element space. The values in shared dofs are
|
||||
determined from the element with maximal attribute. */
|
||||
virtual void ProjectDiscCoefficient(
|
||||
std::variant<Coefficient*, VectorCoefficient*> coeff) override;
|
||||
|
||||
public:
|
||||
ParGridFunction() { pfes = NULL; }
|
||||
|
||||
@@ -268,11 +274,6 @@ public:
|
||||
ProjectType type = ProjectType::DEFAULT) override;
|
||||
|
||||
using GridFunction::ProjectDiscCoefficient;
|
||||
/** @brief Project a discontinuous vector coefficient as a grid function on
|
||||
a continuous finite element space. The values in shared dofs are
|
||||
determined from the element with maximal attribute. */
|
||||
void ProjectDiscCoefficient(VectorCoefficient &coeff) override;
|
||||
|
||||
void ProjectDiscCoefficient(Coefficient &coeff, AvgType type) override;
|
||||
|
||||
void ProjectDiscCoefficient(VectorCoefficient &vcoeff, AvgType type) override;
|
||||
@@ -280,8 +281,7 @@ public:
|
||||
using GridFunction::ProjectBdrCoefficient;
|
||||
|
||||
void ProjectBdrCoefficient(VectorCoefficient &vcoeff,
|
||||
const Array<int> &attr) override
|
||||
{ ProjectBdrCoefficient(NULL, &vcoeff, attr); }
|
||||
const Array<int> &attr) override;
|
||||
|
||||
void ProjectBdrCoefficient(Coefficient *coeff[],
|
||||
const Array<int> &attr) override
|
||||
|
||||
+11
-5
@@ -321,12 +321,17 @@ void ParL2FaceRestriction::DoubleValuedConformingMult(
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
const int threshold = ndofs;
|
||||
const int nsdofs = pfes.GetFaceNbrVSize();
|
||||
const int nsdofs = pfes.GetFaceNbrVSize() / vd;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_x_shared = Reshape(face_nbr_data.Read(),
|
||||
t?vd:nsdofs, t?nsdofs:vd);
|
||||
const int ne_shared = nsdofs / elem_dofs;
|
||||
const int nedof = elem_dofs;
|
||||
// Note: the shape of face_nbr_data, as determined by
|
||||
// ParFiniteElementSpace::ExchangeFaceNbrData, is (elem_dofs, vdim,
|
||||
// ne_shared), independent of the ordering (byNODES or byVDIM) of the finite
|
||||
// element space.
|
||||
auto d_x_shared = Reshape(face_nbr_data.Read(), elem_dofs, vd, ne_shared);
|
||||
auto d_y = Reshape(y.Write(), nface_dofs, vd, 2, nf);
|
||||
mfem::forall(nfdofs, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
@@ -346,8 +351,9 @@ void ParL2FaceRestriction::DoubleValuedConformingMult(
|
||||
}
|
||||
else if (idx2>=threshold) // shared boundary
|
||||
{
|
||||
d_y(dof, c, 1, face) = d_x_shared(t?c:(idx2-threshold),
|
||||
t?(idx2-threshold):c);
|
||||
const int e_shared = (idx2 - threshold) / nedof;
|
||||
const int i_shared = (idx2 - threshold) % nedof;
|
||||
d_y(dof, c, 1, face) = d_x_shared(i_shared,c,e_shared);
|
||||
}
|
||||
else // true boundary
|
||||
{
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "eval_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// @cond Suppress_Doxygen_warnings
|
||||
|
||||
QuadratureInterpolator::TensorEvalTransposeKernelType
|
||||
QuadratureInterpolator::TensorEvalTransposeKernels::Fallback(
|
||||
int DIM, QVectorLayout Q_LAYOUT, int, int, int)
|
||||
{
|
||||
using namespace internal::quadrature_interpolator;
|
||||
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
if (DIM == 1) { return ValuesTranspose1D<QVectorLayout::byNODES>; }
|
||||
else if (DIM == 2) { return ValuesTranspose2D<QVectorLayout::byNODES>; }
|
||||
else if (DIM == 3) { return ValuesTranspose3D<QVectorLayout::byNODES>; }
|
||||
}
|
||||
else
|
||||
{
|
||||
if (DIM == 1) { return ValuesTranspose1D<QVectorLayout::byVDIM>; }
|
||||
else if (DIM == 2) { return ValuesTranspose2D<QVectorLayout::byVDIM>; }
|
||||
else if (DIM == 3) { return ValuesTranspose3D<QVectorLayout::byVDIM>; }
|
||||
}
|
||||
MFEM_ABORT("Invalid dimension");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// @endcond
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,300 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
#include "../kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT>
|
||||
static void ValuesTranspose1D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const auto b = Reshape(b_, q1d, d1d);
|
||||
const auto qd = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, q1d, vdim, NE) :
|
||||
Reshape(q_, vdim, q1d, NE);
|
||||
auto e = Reshape(e_, d1d, vdim, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
for (int d = 0; d < d1d; d++)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int q = 0; q < q1d; q++)
|
||||
{
|
||||
const real_t qval = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
qd(c, q, el) : qd(q, c, el);
|
||||
u += b(q, d) * qval;
|
||||
}
|
||||
e(d, c, el) += u;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1>
|
||||
static void ValuesTranspose2D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, VDIM, NE) :
|
||||
Reshape(q_, VDIM, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D_batch(NE, D1D, D1D, NBZ, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1*MD1];
|
||||
MFEM_SHARED real_t sm0[NBZ][MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[NBZ][MDQ*MDQ];
|
||||
|
||||
kernels::internal::LoadB<MD1,MQ1>(D1D,Q1D,b,sB);
|
||||
|
||||
ConstDeviceMatrix B(sB, D1D, Q1D);
|
||||
DeviceMatrix QQ(sm0[tidz], MQ1, MQ1);
|
||||
DeviceMatrix DQ(sm1[tidz], MD1, MQ1);
|
||||
DeviceMatrix DD(sm0[tidz], MD1, MD1);
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Load Q data
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
QQ(qx,qy) = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
q(c,qx,qy,el) : q(qx,qy,c,el);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in y: QQ -> DQ (apply B^T in y-direction)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * QQ(qx,qy);
|
||||
}
|
||||
DQ(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in x: DQ -> DD (apply B^T in x-direction)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DQ(dy,qx);
|
||||
}
|
||||
DD(dx,dy) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Store result
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,c,el) += DD(dx,dy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0>
|
||||
static void ValuesTranspose3D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, Q1D, VDIM, NE) :
|
||||
Reshape(q_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_3D(NE, D1D, D1D, D1D, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1*MD1];
|
||||
MFEM_SHARED real_t sm0[MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[MDQ*MDQ*MDQ];
|
||||
|
||||
kernels::internal::LoadB<MD1,MQ1>(D1D,Q1D,b,sB);
|
||||
|
||||
ConstDeviceMatrix B(sB, D1D, Q1D);
|
||||
DeviceCube QQQ(sm0, MQ1, MQ1, MQ1);
|
||||
DeviceCube DQQ(sm1, MD1, MQ1, MQ1);
|
||||
DeviceCube DDQ(sm0, MD1, MD1, MQ1);
|
||||
DeviceCube DDD(sm1, MD1, MD1, MD1);
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Load Q data
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
QQQ(qx,qy,qz) = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
q(c,qx,qy,qz,el) : q(qx,qy,qz,c,el);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in z
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += B(dz,qz) * QQQ(qx,qy,qz);
|
||||
}
|
||||
DQQ(dz,qx,qy) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in y
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * DQQ(dz,qx,qy);
|
||||
}
|
||||
DDQ(dz,dy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in x
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DDQ(dz,dy,qx);
|
||||
}
|
||||
DDD(dx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,dz,c,el) += DDD(dx,dy,dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, QVectorLayout Q_LAYOUT,
|
||||
int VDIM, int D1D, int Q1D, int NBZ>
|
||||
QuadratureInterpolator::TensorEvalTransposeKernelType
|
||||
QuadratureInterpolator::TensorEvalTransposeKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::quadrature_interpolator::ValuesTranspose1D<Q_LAYOUT>; }
|
||||
else if (DIM == 2) { return internal::quadrature_interpolator::ValuesTranspose2D<Q_LAYOUT, VDIM, D1D, Q1D, NBZ>; }
|
||||
else if (DIM == 3) { return internal::quadrature_interpolator::ValuesTranspose3D<Q_LAYOUT, VDIM, D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,61 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "eval_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
void InitEvalTransposeByVDimKernels()
|
||||
{
|
||||
using k = QuadratureInterpolator::TensorEvalTransposeKernels;
|
||||
constexpr auto L = QVectorLayout::byVDIM;
|
||||
|
||||
// 2D
|
||||
k::Specialization<2,L,1,2,4>::Opt<8>::Add();
|
||||
k::Specialization<2,L,1,3,6>::Opt<4>::Add();
|
||||
k::Specialization<2,L,1,4,8>::Opt<2>::Add();
|
||||
|
||||
k::Specialization<2,L,2,2,4>::Opt<8>::Add();
|
||||
k::Specialization<2,L,2,3,4>::Opt<8>::Add();
|
||||
k::Specialization<2,L,2,3,6>::Opt<4>::Add();
|
||||
k::Specialization<2,L,2,4,6>::Opt<2>::Add();
|
||||
k::Specialization<2,L,2,4,8>::Opt<2>::Add();
|
||||
|
||||
// 3D
|
||||
k::Specialization<3,L,1,2,4>::Opt<1>::Add();
|
||||
k::Specialization<3,L,1,3,6>::Opt<1>::Add();
|
||||
k::Specialization<3,L,1,4,8>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,2,4>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,3,6>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,4,8>::Opt<1>::Add();
|
||||
|
||||
k::Specialization<3,L,3,2,2>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,3,3>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,4,4>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,5,5>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,6,6>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,7,7>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,8,8>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,9,9>::Opt<1>::Add();
|
||||
|
||||
k::Specialization<3,L,3,4,6>::Opt<1>::Add();
|
||||
k::Specialization<3,L,3,3,4>::Opt<1>::Add();
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,62 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "grad_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// @cond Suppress_Doxygen_warnings
|
||||
|
||||
QuadratureInterpolator::GradTransposeKernelType
|
||||
QuadratureInterpolator::GradTransposeKernels::Fallback(
|
||||
int DIM, QVectorLayout Q_LAYOUT, bool GRAD_PHYS, int, int, int)
|
||||
{
|
||||
using namespace internal::quadrature_interpolator;
|
||||
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
if (DIM == 1) { return DerivativesTranspose1D<QVectorLayout::byNODES, true>; }
|
||||
else if (DIM == 2) { return DerivativesTranspose2D<QVectorLayout::byNODES, true>; }
|
||||
else if (DIM == 3) { return DerivativesTranspose3D<QVectorLayout::byNODES, true>; }
|
||||
}
|
||||
else
|
||||
{
|
||||
if (DIM == 1) { return DerivativesTranspose1D<QVectorLayout::byNODES, false>; }
|
||||
else if (DIM == 2) { return DerivativesTranspose2D<QVectorLayout::byNODES, false>; }
|
||||
else if (DIM == 3) { return DerivativesTranspose3D<QVectorLayout::byNODES, false>; }
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
if (DIM == 1) { return DerivativesTranspose1D<QVectorLayout::byVDIM, true>; }
|
||||
else if (DIM == 2) { return DerivativesTranspose2D<QVectorLayout::byVDIM, true>; }
|
||||
else if (DIM == 3) { return DerivativesTranspose3D<QVectorLayout::byVDIM, true>; }
|
||||
}
|
||||
else
|
||||
{
|
||||
if (DIM == 1) { return DerivativesTranspose1D<QVectorLayout::byVDIM, false>; }
|
||||
else if (DIM == 2) { return DerivativesTranspose2D<QVectorLayout::byVDIM, false>; }
|
||||
else if (DIM == 3) { return DerivativesTranspose3D<QVectorLayout::byVDIM, false>; }
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Invalid dimension");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// @endcond
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,737 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
#include "../kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
// Transpose gradient operation: integrate against shape function derivatives
|
||||
// This is the adjoint of the Derivatives operation
|
||||
|
||||
template<QVectorLayout Q_LAYOUT, bool GRAD_PHYS>
|
||||
static void DerivativesTranspose1D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *g_,
|
||||
const real_t *j_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int sdim,
|
||||
const int vdim,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(b_);
|
||||
const int SDIM = GRAD_PHYS ? sdim : 1;
|
||||
const auto g = Reshape(g_, q1d, d1d);
|
||||
const auto j = Reshape(j_, q1d, SDIM, NE);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, q1d, vdim, SDIM, NE):
|
||||
Reshape(q_, vdim, SDIM, q1d, NE);
|
||||
auto e = Reshape(e_, d1d, vdim, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
for (int d = 0; d < d1d; d++)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
// Load gradient from q-vector
|
||||
real_t dq[3] = {0.0, 0.0, 0.0};
|
||||
for (int s = 0; s < SDIM; ++s)
|
||||
{
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { dq[s] = q(c, s, qx, el); }
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES) { dq[s] = q(qx, c, s, el); }
|
||||
}
|
||||
|
||||
// Apply inverse Jacobian transpose (adjoint of physical gradient)
|
||||
real_t du = dq[0];
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
if (SDIM == 1) { du = dq[0] / j(qx, 0, el); }
|
||||
else if (SDIM == 2)
|
||||
{
|
||||
const real_t Jloc[2] = {j(qx,0,el), j(qx,1,el)};
|
||||
real_t Jinv[3];
|
||||
kernels::CalcLeftInverse<2,1>(Jloc, Jinv);
|
||||
du = Jinv[0]*dq[0] + Jinv[1]*dq[1];
|
||||
}
|
||||
else // SDIM == 3
|
||||
{
|
||||
const real_t Jloc[3] = {j(qx,0,el), j(qx,1,el), j(qx,2,el)};
|
||||
real_t Jinv[3];
|
||||
kernels::CalcLeftInverse<3,1>(Jloc, Jinv);
|
||||
du = Jinv[0]*dq[0] + Jinv[1]*dq[1] + Jinv[2]*dq[2];
|
||||
}
|
||||
}
|
||||
|
||||
// Accumulate contribution (transpose of G matrix)
|
||||
u += g(qx, d) * du;
|
||||
}
|
||||
e(d, c, el) += u;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT, bool GRAD_PHYS,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1>
|
||||
static void DerivativesTranspose2D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *g_,
|
||||
const real_t *j_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int sdim = 2,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
const int SDIM = GRAD_PHYS ? sdim : 2;
|
||||
static constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, SDIM, 2, NE);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, VDIM, SDIM, NE):
|
||||
Reshape(q_, VDIM, SDIM, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D_batch(NE, D1D, D1D, NBZ, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED real_t BG[2][MQ1*MD1];
|
||||
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
DeviceMatrix B(BG[0], D1D, Q1D);
|
||||
DeviceMatrix G(BG[1], D1D, Q1D);
|
||||
|
||||
MFEM_SHARED real_t sm0[NBZ][MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[NBZ][MDQ*MDQ];
|
||||
|
||||
DeviceMatrix QQ(sm0[tidz], MQ1, MQ1);
|
||||
DeviceMatrix DQ0(sm1[tidz], MD1, MQ1);
|
||||
DeviceMatrix DQ1(sm1[tidz], MD1, MQ1); // Reuse sm1 after DQ0 is done
|
||||
DeviceMatrix DD(sm0[tidz], MD1, MD1); // Reuse sm0 after QQ is done
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Load Q data and apply inverse Jacobian
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
// Load gradient components
|
||||
real_t dq[3] = {0.0, 0.0, 0.0};
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { dq[d] = q(c, d, qx, qy, el); }
|
||||
else { dq[d] = q(qx, qy, c, d, el); }
|
||||
}
|
||||
|
||||
// Apply inverse Jacobian transpose (adjoint of physical gradient)
|
||||
real_t du[2] = {dq[0], dq[1]};
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
if (SDIM == 2)
|
||||
{
|
||||
real_t Jloc[4], Jinv[4];
|
||||
Jloc[0] = j(qx,qy,0,0,el);
|
||||
Jloc[1] = j(qx,qy,1,0,el);
|
||||
Jloc[2] = j(qx,qy,0,1,el);
|
||||
Jloc[3] = j(qx,qy,1,1,el);
|
||||
kernels::CalcInverse<2>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[2]*dq[1];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[3]*dq[1];
|
||||
du[0] = U;
|
||||
du[1] = V;
|
||||
}
|
||||
else // SDIM == 3
|
||||
{
|
||||
real_t Jloc[6], Jinv[6];
|
||||
Jloc[0] = j(qx,qy,0,0,el);
|
||||
Jloc[1] = j(qx,qy,1,0,el);
|
||||
Jloc[2] = j(qx,qy,2,0,el);
|
||||
Jloc[3] = j(qx,qy,0,1,el);
|
||||
Jloc[4] = j(qx,qy,1,1,el);
|
||||
Jloc[5] = j(qx,qy,2,1,el);
|
||||
kernels::CalcLeftInverse<3,2>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[2]*dq[1] + Jinv[4]*dq[2];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[3]*dq[1] + Jinv[5]*dq[2];
|
||||
du[0] = U;
|
||||
du[1] = V;
|
||||
}
|
||||
}
|
||||
QQ(qx, qy) = du[0]; // Store du/dx component
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in y-direction: QQ -> DQ0
|
||||
// (Transpose of d/dx which uses DQ1(dy,qx)*B(dy,qy))
|
||||
// Must produce DQ0(dy,qx) to match forward's DQ1 indexing
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * QQ(qx,qy);
|
||||
}
|
||||
DQ0(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply G^T in x-direction: DQ0 -> DD
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += G(dx,qx) * DQ0(dy,qx);
|
||||
}
|
||||
DD(dx,dy) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Accumulate to output
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,c,el) += DD(dx,dy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Now process du/dy component
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
// Load gradient components
|
||||
real_t dq[3] = {0.0, 0.0, 0.0};
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { dq[d] = q(c, d, qx, qy, el); }
|
||||
else { dq[d] = q(qx, qy, c, d, el); }
|
||||
}
|
||||
|
||||
// Apply inverse Jacobian transpose
|
||||
real_t du[2] = {dq[0], dq[1]};
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
if (SDIM == 2)
|
||||
{
|
||||
real_t Jloc[4], Jinv[4];
|
||||
Jloc[0] = j(qx,qy,0,0,el);
|
||||
Jloc[1] = j(qx,qy,1,0,el);
|
||||
Jloc[2] = j(qx,qy,0,1,el);
|
||||
Jloc[3] = j(qx,qy,1,1,el);
|
||||
kernels::CalcInverse<2>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[2]*dq[1];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[3]*dq[1];
|
||||
du[0] = U;
|
||||
du[1] = V;
|
||||
}
|
||||
else // SDIM == 3
|
||||
{
|
||||
real_t Jloc[6], Jinv[6];
|
||||
Jloc[0] = j(qx,qy,0,0,el);
|
||||
Jloc[1] = j(qx,qy,1,0,el);
|
||||
Jloc[2] = j(qx,qy,2,0,el);
|
||||
Jloc[3] = j(qx,qy,0,1,el);
|
||||
Jloc[4] = j(qx,qy,1,1,el);
|
||||
Jloc[5] = j(qx,qy,2,1,el);
|
||||
kernels::CalcLeftInverse<3,2>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[2]*dq[1] + Jinv[4]*dq[2];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[3]*dq[1] + Jinv[5]*dq[2];
|
||||
du[0] = U;
|
||||
du[1] = V;
|
||||
}
|
||||
}
|
||||
QQ(qx, qy) = du[1]; // Store du/dy component
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply G^T in y-direction: QQ -> DQ1
|
||||
// (Transpose of d/dy which uses DQ0(dy,qx)*G(dy,qy))
|
||||
// Must produce DQ1(dy,qx) to match forward's DQ0 indexing
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += G(dy,qy) * QQ(qx,qy);
|
||||
}
|
||||
DQ1(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in x-direction: DQ1 -> DD
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DQ1(dy,qx);
|
||||
}
|
||||
DD(dx,dy) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Accumulate to output
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,c,el) += DD(dx,dy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT, bool GRAD_PHYS,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0>
|
||||
static void DerivativesTranspose3D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *g_,
|
||||
const real_t *j_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int sdim = 3,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, Q1D, 3, 3, NE);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, Q1D, VDIM, 3, NE):
|
||||
Reshape(q_, VDIM, 3, Q1D, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
|
||||
MFEM_SHARED real_t BG[2][MQ1*MD1];
|
||||
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
DeviceMatrix B(BG[0], D1D, Q1D);
|
||||
DeviceMatrix G(BG[1], D1D, Q1D);
|
||||
|
||||
MFEM_SHARED real_t sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED real_t sm1[3][MQ1*MQ1*MQ1];
|
||||
DeviceCube QQQ(sm0[0], MQ1, MQ1, MQ1);
|
||||
DeviceCube DQQ(sm1[0], MD1, MQ1, MQ1);
|
||||
DeviceCube DDQ(sm0[0], MD1, MD1, MQ1);
|
||||
DeviceCube DDD(sm1[0], MD1, MD1, MD1);
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Process du/dx component
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t dq[3];
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
dq[0] = q(c,0,qx,qy,qz,el);
|
||||
dq[1] = q(c,1,qx,qy,qz,el);
|
||||
dq[2] = q(c,2,qx,qy,qz,el);
|
||||
}
|
||||
else
|
||||
{
|
||||
dq[0] = q(qx,qy,qz,c,0,el);
|
||||
dq[1] = q(qx,qy,qz,c,1,el);
|
||||
dq[2] = q(qx,qy,qz,c,2,el);
|
||||
}
|
||||
|
||||
real_t du[3] = {dq[0], dq[1], dq[2]};
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
real_t Jloc[9], Jinv[9];
|
||||
for (int col = 0; col < 3; col++)
|
||||
{
|
||||
for (int row = 0; row < 3; row++)
|
||||
{
|
||||
Jloc[row+3*col] = j(qx,qy,qz,row,col,el);
|
||||
}
|
||||
}
|
||||
kernels::CalcInverse<3>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[3]*dq[1] + Jinv[6]*dq[2];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[4]*dq[1] + Jinv[7]*dq[2];
|
||||
const real_t W = Jinv[2]*dq[0] + Jinv[5]*dq[1] + Jinv[8]*dq[2];
|
||||
du[0] = U; du[1] = V; du[2] = W;
|
||||
}
|
||||
QQQ(qx,qy,qz) = du[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply G^T in x: QQQ -> DQQ (transpose of G⊗B⊗B)
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += G(dx,qx) * QQQ(qx,qy,qz);
|
||||
}
|
||||
DQQ(dx,qy,qz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in y: DQQ -> DDQ
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * DQQ(dx,qy,qz);
|
||||
}
|
||||
DDQ(dx,dy,qz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in z: DDQ -> DDD
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += B(dz,qz) * DDQ(dx,dy,qz);
|
||||
}
|
||||
DDD(dx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Accumulate result
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,dz,c,el) += DDD(dx,dy,dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Process du/dy component
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t dq[3];
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
dq[0] = q(c,0,qx,qy,qz,el);
|
||||
dq[1] = q(c,1,qx,qy,qz,el);
|
||||
dq[2] = q(c,2,qx,qy,qz,el);
|
||||
}
|
||||
else
|
||||
{
|
||||
dq[0] = q(qx,qy,qz,c,0,el);
|
||||
dq[1] = q(qx,qy,qz,c,1,el);
|
||||
dq[2] = q(qx,qy,qz,c,2,el);
|
||||
}
|
||||
|
||||
real_t du[3] = {dq[0], dq[1], dq[2]};
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
real_t Jloc[9], Jinv[9];
|
||||
for (int col = 0; col < 3; col++)
|
||||
{
|
||||
for (int row = 0; row < 3; row++)
|
||||
{
|
||||
Jloc[row+3*col] = j(qx,qy,qz,row,col,el);
|
||||
}
|
||||
}
|
||||
kernels::CalcInverse<3>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[3]*dq[1] + Jinv[6]*dq[2];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[4]*dq[1] + Jinv[7]*dq[2];
|
||||
const real_t W = Jinv[2]*dq[0] + Jinv[5]*dq[1] + Jinv[8]*dq[2];
|
||||
du[0] = U; du[1] = V; du[2] = W;
|
||||
}
|
||||
QQQ(qx,qy,qz) = du[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in x: QQQ -> DQQ (transpose of B⊗G⊗B)
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * QQQ(qx,qy,qz);
|
||||
}
|
||||
DQQ(dx,qy,qz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply G^T in y: DQQ -> DDQ
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += G(dy,qy) * DQQ(dx,qy,qz);
|
||||
}
|
||||
DDQ(dx,dy,qz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in z: DDQ -> DDD
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += B(dz,qz) * DDQ(dx,dy,qz);
|
||||
}
|
||||
DDD(dx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Accumulate result
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,dz,c,el) += DDD(dx,dy,dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Process du/dz component
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t dq[3];
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
dq[0] = q(c,0,qx,qy,qz,el);
|
||||
dq[1] = q(c,1,qx,qy,qz,el);
|
||||
dq[2] = q(c,2,qx,qy,qz,el);
|
||||
}
|
||||
else
|
||||
{
|
||||
dq[0] = q(qx,qy,qz,c,0,el);
|
||||
dq[1] = q(qx,qy,qz,c,1,el);
|
||||
dq[2] = q(qx,qy,qz,c,2,el);
|
||||
}
|
||||
|
||||
real_t du[3] = {dq[0], dq[1], dq[2]};
|
||||
if (GRAD_PHYS)
|
||||
{
|
||||
real_t Jloc[9], Jinv[9];
|
||||
for (int col = 0; col < 3; col++)
|
||||
{
|
||||
for (int row = 0; row < 3; row++)
|
||||
{
|
||||
Jloc[row+3*col] = j(qx,qy,qz,row,col,el);
|
||||
}
|
||||
}
|
||||
kernels::CalcInverse<3>(Jloc, Jinv);
|
||||
const real_t U = Jinv[0]*dq[0] + Jinv[3]*dq[1] + Jinv[6]*dq[2];
|
||||
const real_t V = Jinv[1]*dq[0] + Jinv[4]*dq[1] + Jinv[7]*dq[2];
|
||||
const real_t W = Jinv[2]*dq[0] + Jinv[5]*dq[1] + Jinv[8]*dq[2];
|
||||
du[0] = U; du[1] = V; du[2] = W;
|
||||
}
|
||||
QQQ(qx,qy,qz) = du[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply G^T in z: QQQ -> DQQ (transpose of B⊗B⊗G)
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += G(dz,qz) * QQQ(qx,qy,qz);
|
||||
}
|
||||
DQQ(qx,qy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in y: DQQ -> DDQ
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * DQQ(qx,qy,dz);
|
||||
}
|
||||
DDQ(qx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Apply B^T in x: DDQ -> DDD
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DDQ(qx,dy,dz);
|
||||
}
|
||||
DDD(dx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Accumulate result
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,dz,c,el) += DDD(dx,dy,dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, QVectorLayout Q_LAYOUT, bool GRAD_PHYS, int VDIM, int D1D,
|
||||
int Q1D, int NBZ>
|
||||
QuadratureInterpolator::GradTransposeKernelType
|
||||
QuadratureInterpolator::GradTransposeKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::quadrature_interpolator::DerivativesTranspose1D<Q_LAYOUT, GRAD_PHYS>; }
|
||||
else if (DIM == 2) { return internal::quadrature_interpolator::DerivativesTranspose2D<Q_LAYOUT, GRAD_PHYS, VDIM, D1D, Q1D, NBZ>; }
|
||||
else if (DIM == 3) { return internal::quadrature_interpolator::DerivativesTranspose3D<Q_LAYOUT, GRAD_PHYS, VDIM, D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,50 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "grad_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
template <bool P>
|
||||
void InitGradTransposeByNodesKernels()
|
||||
{
|
||||
using k = QuadratureInterpolator::GradTransposeKernels;
|
||||
constexpr auto L = QVectorLayout::byNODES;
|
||||
// 2D
|
||||
k::Specialization<2,L,P,1,3,4>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,1,4,6>::template Opt<4>::Add();
|
||||
k::Specialization<2,L,P,1,5,8>::template Opt<2>::Add();
|
||||
|
||||
k::Specialization<2,L,P,2,3,3>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,2,3,4>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,2,4,6>::template Opt<4>::Add();
|
||||
k::Specialization<2,L,P,2,5,8>::template Opt<2>::Add();
|
||||
// 3D
|
||||
k::Specialization<3,L,P,1,3,4>::Add();
|
||||
k::Specialization<3,L,P,1,4,6>::Add();
|
||||
k::Specialization<3,L,P,1,5,8>::Add();
|
||||
k::Specialization<3,L,P,3,3,4>::Add();
|
||||
k::Specialization<3,L,P,3,4,6>::Add();
|
||||
k::Specialization<3,L,P,3,5,8>::Add();
|
||||
}
|
||||
|
||||
template void InitGradTransposeByNodesKernels<false>();
|
||||
template void InitGradTransposeByNodesKernels<true>();
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,50 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "grad_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
template <bool P>
|
||||
void InitGradTransposeByVDimKernels()
|
||||
{
|
||||
using k = QuadratureInterpolator::GradTransposeKernels;
|
||||
constexpr auto L = QVectorLayout::byVDIM;
|
||||
// 2D
|
||||
k::Specialization<2,L,P,1,3,4>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,1,4,6>::template Opt<4>::Add();
|
||||
k::Specialization<2,L,P,1,5,8>::template Opt<2>::Add();
|
||||
|
||||
k::Specialization<2,L,P,2,3,3>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,2,3,4>::template Opt<8>::Add();
|
||||
k::Specialization<2,L,P,2,4,6>::template Opt<4>::Add();
|
||||
k::Specialization<2,L,P,2,5,8>::template Opt<2>::Add();
|
||||
// 3D
|
||||
k::Specialization<3,L,P,1,3,4>::Add();
|
||||
k::Specialization<3,L,P,1,4,6>::Add();
|
||||
k::Specialization<3,L,P,1,5,8>::Add();
|
||||
k::Specialization<3,L,P,3,3,4>::Add();
|
||||
k::Specialization<3,L,P,3,4,6>::Add();
|
||||
k::Specialization<3,L,P,3,5,8>::Add();
|
||||
}
|
||||
|
||||
template void InitGradTransposeByVDimKernels<false>();
|
||||
template void InitGradTransposeByVDimKernels<true>();
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
} // namespace mfem
|
||||
@@ -11,6 +11,7 @@
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "qinterp/grad.hpp"
|
||||
#include "qinterp/grad_transpose.hpp"
|
||||
#include "qinterp/eval.hpp"
|
||||
#include "qspace.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
@@ -30,7 +31,10 @@ void InitEvalKernels();
|
||||
void InitDetKernels();
|
||||
template <bool P> void InitGradByNodesKernels();
|
||||
template <bool P> void InitGradByVDimKernels();
|
||||
template <bool P> void InitGradTransposeByNodesKernels();
|
||||
template <bool P> void InitGradTransposeByVDimKernels();
|
||||
void InitTensorEvalHDivKernels();
|
||||
void InitEvalTransposeByVDimKernels();
|
||||
struct Kernels
|
||||
{
|
||||
Kernels()
|
||||
@@ -45,12 +49,19 @@ struct Kernels
|
||||
// Phys grad kernels
|
||||
InitGradByNodesKernels<true>();
|
||||
InitGradByVDimKernels<true>();
|
||||
// Non-phys grad transpose kernels
|
||||
InitGradTransposeByNodesKernels<false>();
|
||||
InitGradTransposeByVDimKernels<false>();
|
||||
// Phys grad transpose kernels
|
||||
InitGradTransposeByNodesKernels<true>();
|
||||
InitGradTransposeByVDimKernels<true>();
|
||||
// Determinants
|
||||
InitDetKernels();
|
||||
// Non-tensor
|
||||
InitEvalKernels();
|
||||
// Tensor (quad,hex) H(div)
|
||||
InitTensorEvalHDivKernels();
|
||||
InitEvalTransposeByVDimKernels();
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -646,16 +657,59 @@ void QuadratureInterpolator::MultHDiv(const Vector &e_vec,
|
||||
MFEM_CONTRACT_VAR(q_div);
|
||||
}
|
||||
|
||||
void QuadratureInterpolator::MultTranspose(unsigned eval_flags,
|
||||
const Vector &q_val,
|
||||
const Vector &q_der,
|
||||
Vector &e_vec) const
|
||||
void QuadratureInterpolator::AddMultTranspose(unsigned eval_flags,
|
||||
const Vector &q_val,
|
||||
const Vector &q_der,
|
||||
Vector &e_vec) const
|
||||
{
|
||||
MFEM_CONTRACT_VAR(eval_flags);
|
||||
MFEM_CONTRACT_VAR(q_val);
|
||||
MFEM_CONTRACT_VAR(q_der);
|
||||
MFEM_CONTRACT_VAR(e_vec);
|
||||
MFEM_ABORT("this method is not implemented yet");
|
||||
const int ne = fespace->GetNE();
|
||||
if (ne == 0) { return; }
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int sdim = fespace->GetMesh()->SpaceDimension();
|
||||
|
||||
const bool use_tensor_eval =
|
||||
use_tensor_products &&
|
||||
dynamic_cast<const TensorBasisElement*>(fe) != nullptr;
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad::Mode mode =
|
||||
use_tensor_eval ? DofToQuad::TENSOR : DofToQuad::FULL;
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
|
||||
const int dim = maps.FE->GetDim();
|
||||
const int nd = maps.ndof;
|
||||
const int nq = maps.nqpt;
|
||||
|
||||
const GeometricFactors *geom = nullptr;
|
||||
if (eval_flags & PHYSICAL_DERIVATIVES)
|
||||
{
|
||||
const int jacobians = GeometricFactors::JACOBIANS;
|
||||
geom = fespace->GetMesh()->GetGeometricFactors(*ir, jacobians);
|
||||
}
|
||||
|
||||
if (use_tensor_eval)
|
||||
{
|
||||
if (eval_flags & (VALUES | PHYSICAL_VALUES))
|
||||
{
|
||||
TensorEvalTransposeKernels::Run(dim, q_layout, vdim, nd, nq, ne,
|
||||
maps.B.Read(), q_val.Read(),
|
||||
e_vec.ReadWrite(), vdim, nd, nq);
|
||||
}
|
||||
if (eval_flags & (DERIVATIVES | PHYSICAL_DERIVATIVES))
|
||||
{
|
||||
const bool phys = (eval_flags & PHYSICAL_DERIVATIVES);
|
||||
const real_t *J = phys ? geom->J.Read() : nullptr;
|
||||
const int s_dim = phys ? sdim : dim;
|
||||
GradTransposeKernels::Run(dim, q_layout, phys, vdim, nd, nq, ne,
|
||||
maps.B.Read(), maps.G.Read(), J,
|
||||
q_der.Read(), e_vec.ReadWrite(),
|
||||
s_dim, vdim, nd, nq);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Non-tensor MultTranspose not yet implemented");
|
||||
}
|
||||
}
|
||||
|
||||
void QuadratureInterpolator::Values(const Vector &e_vec,
|
||||
|
||||
@@ -152,8 +152,8 @@ public:
|
||||
void Determinants(const Vector &e_vec, Vector &q_det) const;
|
||||
|
||||
/// Perform the transpose operation of Mult(). (TODO)
|
||||
void MultTranspose(unsigned eval_flags, const Vector &q_val,
|
||||
const Vector &q_der, Vector &e_vec) const;
|
||||
void AddMultTranspose(unsigned eval_flags, const Vector &q_val,
|
||||
const Vector &q_der, Vector &e_vec) const;
|
||||
|
||||
/// @brief Returns true if the given finite element space is supported by
|
||||
/// QuadratureInterpolator.
|
||||
@@ -178,6 +178,13 @@ public:
|
||||
using TensorEvalHDivKernelType =
|
||||
void(*)(const int, const real_t *, const real_t *, const real_t *,
|
||||
const real_t *, real_t *, const int, const int);
|
||||
using TensorEvalTransposeKernelType = void(*)(const int, const real_t *,
|
||||
const real_t *,
|
||||
real_t *, const int, const int, const int);
|
||||
using GradTransposeKernelType = void(*)(const int, const real_t *,
|
||||
const real_t *, const real_t *,
|
||||
const real_t *, real_t *,
|
||||
const int, const int, const int, const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(TensorEvalKernels, TensorEvalKernelType,
|
||||
(int, QVectorLayout, int, int, int), (int));
|
||||
@@ -189,6 +196,10 @@ public:
|
||||
(int, QVectorLayout, bool, int, int), (int));
|
||||
MFEM_REGISTER_KERNELS(TensorEvalHDivKernels, TensorEvalHDivKernelType,
|
||||
(int, QVectorLayout, unsigned, int, int));
|
||||
MFEM_REGISTER_KERNELS(TensorEvalTransposeKernels, TensorEvalTransposeKernelType,
|
||||
(int, QVectorLayout, int, int, int), (int));
|
||||
MFEM_REGISTER_KERNELS(GradTransposeKernels, GradTransposeKernelType,
|
||||
(int, QVectorLayout, bool, int, int, int), (int));
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
+3
-6
@@ -1398,20 +1398,17 @@ void L2FaceRestriction::PermuteAndSetSharedFaceDofsScatterIndices2(
|
||||
const int dim = fes.GetMesh()->Dimension();
|
||||
const int dof1d = fes.GetTypicalFE()->GetOrder()+1;
|
||||
fes.GetTypicalFE()->GetFaceMap(face_id2, face_map);
|
||||
Array<int> face_nbr_dofs;
|
||||
const ParFiniteElementSpace &pfes =
|
||||
static_cast<const ParFiniteElementSpace&>(this->fes);
|
||||
pfes.GetFaceNbrElementVDofs(elem_index, face_nbr_dofs);
|
||||
|
||||
for (int face_dof_elem1 = 0; face_dof_elem1 < face_dofs; ++face_dof_elem1)
|
||||
{
|
||||
const int face_dof_elem2 = PermuteFaceL2(dim, face_id1, face_id2,
|
||||
orientation, dof1d, face_dof_elem1);
|
||||
const int volume_dof_elem2 = face_map[face_dof_elem2];
|
||||
const int global_dof_elem2 = face_nbr_dofs[volume_dof_elem2];
|
||||
// Encode the volume DOF index and element index
|
||||
const int global_dof_elem2 = elem_index*elem_dofs + volume_dof_elem2;
|
||||
const int restriction_dof_elem2 = face_dofs*face_index + face_dof_elem1;
|
||||
// Trick to differentiate dof location inter/shared
|
||||
scatter_indices2[restriction_dof_elem2] = ndofs+global_dof_elem2;
|
||||
scatter_indices2[restriction_dof_elem2] = ndofs + global_dof_elem2;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
+24
-17
@@ -14,6 +14,7 @@
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "array.hpp"
|
||||
#include "text.hpp"
|
||||
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
@@ -247,7 +248,8 @@ inline void ArraysByName<T>::Print(std::ostream &os, int width) const
|
||||
os << data.size() << '\n';
|
||||
for (auto const &it : data)
|
||||
{
|
||||
os << '"' << it.first << '"' << '\n' << it.second.Size() << '\n';
|
||||
// Note: The method Load() can read any string formatted with std::quoted.
|
||||
os << std::quoted(it.first) << '\n' << it.second.Size() << '\n';
|
||||
it.second.Print(os, width > 0 ? width : it.second.Size());
|
||||
}
|
||||
}
|
||||
@@ -258,31 +260,36 @@ void ArraysByName<T>::Load(std::istream &in)
|
||||
int NumArrays;
|
||||
in >> NumArrays;
|
||||
|
||||
std::string ArrayLine, ArrayName;
|
||||
for (int i=0; i < NumArrays; i++)
|
||||
for (int i = 0; i < NumArrays; i++)
|
||||
{
|
||||
in >> std::ws;
|
||||
getline(in, ArrayLine);
|
||||
|
||||
std::size_t q0 = ArrayLine.find('"');
|
||||
std::size_t q1 = ArrayLine.rfind('"');
|
||||
|
||||
if (q0 != std::string::npos && q1 > q0)
|
||||
// Read the name:
|
||||
// - If the stream 'in' starts with " then parse it with the function
|
||||
// parse_quoted_string() from text.hpp. In this case, the name can be
|
||||
// empty. Note: this case allows for reading any string formatted using
|
||||
// std::quoted, e.g. as in the method Print().
|
||||
// - If the name does not start with " then the name ends with the first
|
||||
// white space character (and the white space character is not included
|
||||
// in the name). Since white space characters are skipped before reading
|
||||
// the name, there will be at least one non-white-space character in the
|
||||
// name in this case.
|
||||
std::string ArrayName;
|
||||
if (in.peek() == '"')
|
||||
{
|
||||
// Locate set name between first and last double quote
|
||||
ArrayName = ArrayLine.substr(q0+1,q1-q0-1);
|
||||
if (parse_quoted_string(ArrayName, in) != 0)
|
||||
{
|
||||
MFEM_ABORT("error parsing input!");
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// If no double quotes found locate set name using white space
|
||||
q1 = ArrayLine.find(' ');
|
||||
ArrayName = ArrayLine.substr(0,q1-1);
|
||||
in >> ArrayName;
|
||||
MFEM_VERIFY(in.good(), "error parsing input!");
|
||||
}
|
||||
|
||||
// Ignore the remainder of the line which may contain explanatory comments
|
||||
data[ArrayName].Load(in, 0);
|
||||
// Read the array
|
||||
data[ArrayName].Load(in);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -27,6 +27,14 @@
|
||||
#endif
|
||||
#include "hip.hpp"
|
||||
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
#define Gpu(...) Cu##__VA_ARGS__
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
#define Gpu(...) Hip##__VA_ARGS__
|
||||
#else
|
||||
#define Gpu(...) __VA_ARGS__
|
||||
#endif
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
#include "occa.hpp"
|
||||
#endif
|
||||
@@ -48,6 +56,7 @@ constexpr bool mfem_use_gpu = false;
|
||||
#define MFEM_DEVICE
|
||||
#define MFEM_HOST
|
||||
#define MFEM_LAMBDA
|
||||
#define MFEM_CONSTANT
|
||||
// #define MFEM_HOST_DEVICE // defined in config/config.hpp
|
||||
// MFEM_DEVICE_SYNC is made available for debugging purposes
|
||||
#define MFEM_DEVICE_SYNC
|
||||
@@ -65,6 +74,13 @@ constexpr bool mfem_use_gpu = false;
|
||||
#define MFEM_THREAD_SIZE(k) 1
|
||||
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
|
||||
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) MFEM_FOREACH_THREAD(i,k,N)
|
||||
|
||||
inline const void* MemcpyToSymbol(const void *d_sym, const void *h_src,
|
||||
size_t bytes)
|
||||
{
|
||||
memcpy(const_cast<void *>(d_sym), h_src, bytes);
|
||||
return d_sym;
|
||||
}
|
||||
#endif
|
||||
|
||||
// 'double' and 'float' atomicAdd implementation for previous versions of CUDA
|
||||
|
||||
@@ -175,6 +175,17 @@ void* CuMemcpyDtoHAsync(void *dst, const void *src, size_t bytes)
|
||||
return dst;
|
||||
}
|
||||
|
||||
const void* CuMemcpyToSymbol(const void *d_sym, const void *h_src,
|
||||
size_t bytes)
|
||||
{
|
||||
#ifdef MFEM_USE_CUDA
|
||||
MFEM_GPU_CHECK(cudaMemcpyToSymbol(d_sym, h_src, bytes));
|
||||
return d_sym;
|
||||
#endif
|
||||
MFEM_ABORT("CUDA has no shadow host copy of device symbols");
|
||||
return memcpy(const_cast<void*>(d_sym), h_src, bytes);
|
||||
}
|
||||
|
||||
void CuCheckLastError()
|
||||
{
|
||||
#ifdef MFEM_USE_CUDA
|
||||
|
||||
@@ -25,6 +25,8 @@ constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_LAMBDA __host__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_CONSTANT __constant__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
#define MFEM_DEVICE_SYNC MFEM_GPU_CHECK(cudaDeviceSynchronize())
|
||||
#define MFEM_STREAM_SYNC MFEM_GPU_CHECK(cudaStreamSynchronize(0))
|
||||
@@ -94,6 +96,10 @@ void* CuMemcpyDtoH(void *h_dst, const void *d_src, size_t bytes);
|
||||
/// Copies memory from Device to Host
|
||||
void* CuMemcpyDtoHAsync(void *h_dst, const void *d_src, size_t bytes);
|
||||
|
||||
/// Copies data to the given symbol on the device.
|
||||
const void* CuMemcpyToSymbol(const void *d_sym, const void *h_src,
|
||||
size_t bytes);
|
||||
|
||||
/// Check the error code returned by cudaGetLastError(), aborting on error.
|
||||
void CuCheckLastError();
|
||||
|
||||
|
||||
+4
-4
@@ -726,16 +726,16 @@ std::string Device::GetUUID(const int device_id)
|
||||
MFEM_GPU_CHECK(cudaGetDeviceProperties(&prop, device_id));
|
||||
for (int i = 0; i < 16; ++i)
|
||||
{
|
||||
res << std::setfill('0') << std::setw(2) << std::hex
|
||||
<< static_cast<unsigned>(prop.uuid.bytes[i]);
|
||||
const unsigned b = static_cast<unsigned char>(prop.uuid.bytes[i]);
|
||||
res << std::setfill('0') << std::setw(2) << std::hex << b;
|
||||
}
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
hipUUID uuid;
|
||||
MFEM_GPU_CHECK(hipDeviceGetUuid(&uuid, device_id));
|
||||
for (int i = 0; i < 16; ++i)
|
||||
{
|
||||
res << std::setfill('0') << std::setw(2) << std::hex
|
||||
<< static_cast<unsigned>(uuid.bytes[i]);
|
||||
const unsigned b = static_cast<unsigned char>(uuid.bytes[i]);
|
||||
res << std::setfill('0') << std::setw(2) << std::hex << b;
|
||||
}
|
||||
#endif
|
||||
return res.str();
|
||||
|
||||
@@ -1090,6 +1090,12 @@ inline void forall_2D_batch(int N, int X, int Y, int BZ, lambda &&body)
|
||||
ForallWrap<2>(true, N, body, X, Y, BZ);
|
||||
}
|
||||
|
||||
template<int MAX_THREADS_PER_BLOCK, typename lambda>
|
||||
inline void forall_2D_batch(int N, int X, int Y, int BZ, lambda &&body)
|
||||
{
|
||||
ForallWrap<2, MAX_THREADS_PER_BLOCK>(true, N, body, X, Y, BZ);
|
||||
}
|
||||
|
||||
template<typename lambda>
|
||||
inline void forall_3D(int N, int X, int Y, int Z, lambda &&body)
|
||||
{
|
||||
|
||||
@@ -175,6 +175,15 @@ void* HipMemcpyDtoHAsync(void *dst, const void *src, size_t bytes)
|
||||
return dst;
|
||||
}
|
||||
|
||||
const void* HipMemcpyToSymbol(const void *d_sym, const void *h_src,
|
||||
size_t bytes)
|
||||
{
|
||||
#ifdef MFEM_USE_HIP
|
||||
MFEM_GPU_CHECK(hipMemcpyToSymbol(d_sym, h_src, bytes));
|
||||
#endif
|
||||
return memcpy(const_cast<void*>(d_sym), h_src, bytes);
|
||||
}
|
||||
|
||||
void HipCheckLastError()
|
||||
{
|
||||
#ifdef MFEM_USE_HIP
|
||||
|
||||
+6
-1
@@ -21,8 +21,9 @@
|
||||
#if defined(MFEM_USE_HIP) && defined(__HIP__)
|
||||
#define MFEM_USE_CUDA_OR_HIP
|
||||
constexpr bool mfem_use_gpu = true;
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_HOST __host__
|
||||
#define MFEM_DEVICE __device__
|
||||
#define MFEM_CONSTANT __constant__
|
||||
#define MFEM_LAMBDA __host__ __device__
|
||||
#define MFEM_LAUNCH_BOUNDS __launch_bounds__
|
||||
// #define MFEM_HOST_DEVICE __host__ __device__ // defined in config/config.hpp
|
||||
@@ -96,6 +97,10 @@ void* HipMemcpyDtoH(void *h_dst, const void *d_src, size_t bytes);
|
||||
/// Copies memory from Device to Host
|
||||
void* HipMemcpyDtoHAsync(void *h_dst, const void *d_src, size_t bytes);
|
||||
|
||||
/// Copies data to the given symbol on the device.
|
||||
const void* HipMemcpyToSymbol(const void *d_sym, const void *h_src,
|
||||
size_t bytes);
|
||||
|
||||
/// Check the error code returned by hipGetLastError(), aborting on error.
|
||||
void HipCheckLastError();
|
||||
|
||||
|
||||
Symlink
+1
@@ -0,0 +1 @@
|
||||
../../stash/debug/nvtx.hpp
|
||||
@@ -50,6 +50,48 @@ inline void filter_dos(std::string &line)
|
||||
}
|
||||
}
|
||||
|
||||
/** @brief Read a string formatted using std::quoted. Return nonzero on error.
|
||||
|
||||
The stream @a in must begin with @a delim. After clearing @a result and
|
||||
extracting the opening @a delim, characters are extracted from @a in and
|
||||
processed as follows:
|
||||
- if the character is @a delim, return 0;
|
||||
- if the character is different from @a escape, it is appended to @a result;
|
||||
- if the character is @a escape, the next character from @a in is extracted
|
||||
and if it is one of @a delim or @a escape, it is appended to @a result;
|
||||
otherwise, both @a escape and the character after it are appended to
|
||||
@a result; note that the latter case is not possible if the input was
|
||||
formatted with std::quoted with the same @a delim and @a escape
|
||||
characters.
|
||||
|
||||
If the stream @a in does not begin with @a delim, error code 1 is returned.
|
||||
If reading the stream fails, error code 2 is returned. On success, zero is
|
||||
returned and the closing @a delim character is the last character extracted
|
||||
from @a in. */
|
||||
inline int parse_quoted_string(std::string &result, std::istream &in,
|
||||
char delim = '"', char escape = '\\')
|
||||
{
|
||||
using tt = std::string::traits_type; // std::char_traits<char>
|
||||
auto equal = [](tt::int_type c1, tt::char_type c2) -> bool
|
||||
{
|
||||
return tt::eq_int_type(c1, tt::to_int_type(c2));
|
||||
};
|
||||
result.clear();
|
||||
if (!equal(in.peek(), delim)) { return 1; }
|
||||
in.get(); // extract delim
|
||||
for (auto c = in.get(); !equal(c, delim); c = in.get())
|
||||
{
|
||||
if (equal(c, escape))
|
||||
{
|
||||
c = in.get();
|
||||
if (!equal(c, escape) && !equal(c, delim)) { result += escape; }
|
||||
}
|
||||
if (!in) { return 2; }
|
||||
result += tt::to_char_type(c);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Convert an integer to a 0-padded string with the given number of @a digits
|
||||
inline std::string to_padded_string(int i, int digits)
|
||||
{
|
||||
|
||||
@@ -27,6 +27,7 @@ list(APPEND SRCS
|
||||
handle.cpp
|
||||
matrix.cpp
|
||||
mma.cpp
|
||||
multivector.cpp
|
||||
ode.cpp
|
||||
operator.cpp
|
||||
ordering.cpp
|
||||
@@ -63,6 +64,7 @@ list(APPEND HDRS
|
||||
linalg.hpp
|
||||
matrix.hpp
|
||||
mma.hpp
|
||||
multivector.hpp
|
||||
ode.hpp
|
||||
operator.hpp
|
||||
ordering.hpp
|
||||
|
||||
@@ -317,6 +317,9 @@ void HypreParVector::WrapHypreParVector(hypre_ParVector *y, bool owner)
|
||||
|
||||
Vector * HypreParVector::GlobalVector() const
|
||||
{
|
||||
MFEM_VERIFY(size > 0,
|
||||
"GlobalVector method can only be called on vectors wherein each "
|
||||
"process owns one or more entries");
|
||||
hypre_Vector *hv = hypre_ParVectorToVectorAll(*this);
|
||||
Vector *v = new Vector(hv->data, internal::to_int(hv->size));
|
||||
v->MakeDataOwner();
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
// Linear algebra header file
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "multivector.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "matrix.hpp"
|
||||
#include "sparsemat.hpp"
|
||||
@@ -37,6 +38,7 @@
|
||||
#include "batched/gpu_blas.hpp"
|
||||
#include "batched/solver.hpp"
|
||||
#include "tensor.hpp"
|
||||
#include "tensor_arrays.hpp"
|
||||
#include "filteredsolver.hpp"
|
||||
#include "ordering.hpp"
|
||||
#include "particlevector.hpp"
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "multivector.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MultiVector::MultiVector(const Array<int> &vector_sizes)
|
||||
{
|
||||
SetSizes(vector_sizes);
|
||||
}
|
||||
|
||||
MultiVector::MultiVector(const Array<int> &vector_sizes, MemoryType mt)
|
||||
{
|
||||
SetSizes(vector_sizes, mt);
|
||||
}
|
||||
|
||||
MultiVector::MultiVector(Vector &base, const Array<int> &vector_sizes)
|
||||
{
|
||||
MakeRef(base, vector_sizes);
|
||||
}
|
||||
|
||||
void MultiVector::SetSizes(const Array<int> &vector_sizes)
|
||||
{
|
||||
blocks.resize(vector_sizes.Size());
|
||||
for (int i = 0; i < vector_sizes.Size(); i++)
|
||||
{
|
||||
operator[](i).SetSize(vector_sizes[i]);
|
||||
}
|
||||
}
|
||||
|
||||
void MultiVector::SetSizes(const Array<int> &vector_sizes, MemoryType mt)
|
||||
{
|
||||
blocks.resize(vector_sizes.Size());
|
||||
for (int i = 0; i < vector_sizes.Size(); i++)
|
||||
{
|
||||
operator[](i).SetSize(vector_sizes[i], mt);
|
||||
}
|
||||
}
|
||||
|
||||
void MultiVector::MakeRef(Vector &base, const Array<int> &vector_sizes)
|
||||
{
|
||||
blocks.resize(vector_sizes.Size());
|
||||
for (int offset = 0, i = 0; i < vector_sizes.Size(); i++)
|
||||
{
|
||||
blocks[i].emplace<0>(base, offset, vector_sizes[i]);
|
||||
offset += vector_sizes[i];
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,198 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_MULTIVECTOR_HPP
|
||||
#define MFEM_MULTIVECTOR_HPP
|
||||
|
||||
#include "../general/array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include <vector>
|
||||
#include <array>
|
||||
#include <variant>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Class representing an array of Vectors with generally different sizes.
|
||||
/** This class is similar to BlockVector with the following two main
|
||||
differences:
|
||||
- the data for the individual Vector blocks does not need to be part of one
|
||||
big contiguous memory allocation;
|
||||
- this class does not inherit from class Vector (as a consequence of the
|
||||
first bullet).
|
||||
|
||||
Internally, each Vector block is represented as either:
|
||||
- (default) a Vector object constructed and owned by this class; this
|
||||
object, in turn, as any Vector object, can own its Memory allocation or
|
||||
refer to a sub-Memory of another Memory object; or
|
||||
- a pointer to an externally allocated Vector or classes derived from
|
||||
Vector. */
|
||||
class MultiVector
|
||||
{
|
||||
private:
|
||||
std::vector<std::variant<Vector,Vector*>> blocks;
|
||||
|
||||
public:
|
||||
/// Create an empty MultiVector with zero blocks.
|
||||
MultiVector() = default;
|
||||
|
||||
/** @brief Create a MultiVector with @a num_blocks blocks. The individual
|
||||
Vector blocks are default initialized, i.e. they all have size zero. */
|
||||
MultiVector(int num_blocks)
|
||||
: blocks(num_blocks) { }
|
||||
|
||||
/** @brief Construct a MultiVector with number of blocks and individual block
|
||||
Vector sizes given by @a vector_sizes.
|
||||
|
||||
@note The memory of the individual Vector blocks is NOT initialized. */
|
||||
MultiVector(const Array<int> &vector_sizes);
|
||||
|
||||
/** @brief Construct a MultiVector with number of blocks and individual block
|
||||
Vector sizes given by @a vector_sizes. All Vector blocks use the
|
||||
MemoryType @a mt.
|
||||
|
||||
@note The memory of the individual Vector blocks is NOT initialized. */
|
||||
MultiVector(const Array<int> &vector_sizes, MemoryType mt);
|
||||
|
||||
/** @brief Construct a MultiVector referencing data within a given monolithic
|
||||
Vector @a base.
|
||||
|
||||
With this constructor, the Memory flags of @a base and of the individual
|
||||
Vector blocks may need to be explicitly synchronized when data is moved
|
||||
between host and device. */
|
||||
MultiVector(Vector &base, const Array<int> &vector_sizes);
|
||||
|
||||
/** @brief Construct a MultiVector referencing multiple Vectors given as
|
||||
arguments.
|
||||
|
||||
The VectorTypes reference arguments are expected to be static_cast-able
|
||||
to (Vector &) which is the case if the types are derived from Vector,
|
||||
e.g. HypreParVector, GridFunction, etc.
|
||||
|
||||
With this constructor, operations on individual Vector blocks are
|
||||
performed directly on the objects @a vs. In particular, there is no need
|
||||
to synchronize the Memory flags of @a vs and the ones of the individual
|
||||
Vector blocks when data is moved between host and device. */
|
||||
template <typename... VectorTypes,
|
||||
std::enable_if_t<
|
||||
std::conjunction_v<
|
||||
std::is_convertible<VectorTypes&,Vector&>...>, bool> = true>
|
||||
MultiVector(VectorTypes &...vs) { MakeRef(vs...); }
|
||||
|
||||
/// Return the number of Vectors in the MultiVector.
|
||||
int NumBlocks() const { return blocks.size(); }
|
||||
|
||||
/** @brief Set the number of Vectors in the MultiVector. Existing Vector
|
||||
blocks will remain unmodified. New Vector blocks will be default
|
||||
initialized, i.e. they all have size zero. */
|
||||
void SetNumBlocks(int num_blocks) { blocks.resize(num_blocks); }
|
||||
|
||||
/// Read-write access to the i-th Vector.
|
||||
inline Vector &operator[](int i);
|
||||
|
||||
/// Read-only access to the i-th Vector.
|
||||
inline const Vector &operator[](int i) const;
|
||||
|
||||
/** @brief Update the MultiVector according to the given @a vector_sizes.
|
||||
|
||||
This method can be used to add or remove blocks. The individual Vector
|
||||
sizes are updated using the method Vector::SetSize(int). */
|
||||
void SetSizes(const Array<int> &vector_sizes);
|
||||
|
||||
/** @brief Update the MultiVector according to the given @a vector_sizes and
|
||||
MemoryType @a mt.
|
||||
|
||||
This method can be used to add or remove blocks. The individual Vector
|
||||
sizes and MemoryType are updated using the method
|
||||
Vector::SetSize(int, MemoryType). */
|
||||
void SetSizes(const Array<int> &vector_sizes, MemoryType mt);
|
||||
|
||||
/** @brief Update the MultiVector to reference data within a given monolithic
|
||||
Vector @a base.
|
||||
|
||||
After calling this method, the Memory flags of @a base and of the
|
||||
individual Vector blocks may need to be explicitly synchronized when data
|
||||
is moved between host and device.*/
|
||||
void MakeRef(Vector &base, const Array<int> &vector_sizes);
|
||||
|
||||
/** @brief Update the @a i-th MultiVector block to reference data within the
|
||||
given monolithic Vector @a base at the given @a offset and with the given
|
||||
@a size.
|
||||
|
||||
After calling this method, the Memory flags of @a base and of the @a i-th
|
||||
Vector block may need to be explicitly synchronized when data is moved
|
||||
between host and device.*/
|
||||
inline void MakeRef(int i, Vector &base, int offset, int size)
|
||||
{
|
||||
blocks[i].emplace<0>(base, offset, size);
|
||||
}
|
||||
|
||||
/** @brief Update the MultiVector to reference multiple Vectors given as
|
||||
arguments.
|
||||
|
||||
The VectorTypes reference arguments are expected to be static_cast-able
|
||||
to (Vector &) which is the case if the types are derived from Vector,
|
||||
e.g. HypreParVector, GridFunction, etc.
|
||||
|
||||
After calling this method, operations on individual Vector blocks are
|
||||
performed directly on the objects @a vs. In particular, there is no need
|
||||
to synchronize the Memory flags of @a vs and the ones of the individual
|
||||
Vector blocks when data is moved between host and device. */
|
||||
template <typename... VectorTypes,
|
||||
std::enable_if_t<
|
||||
std::conjunction_v<
|
||||
std::is_convertible<VectorTypes&,Vector&>...>, bool> = true>
|
||||
inline void MakeRef(VectorTypes &...vs);
|
||||
|
||||
/** @brief Update the @a i-th MultiVector block to reference the given
|
||||
Vector @a v.
|
||||
|
||||
After calling this method, operations on the @a i-th Vector block are
|
||||
performed directly on the Vector @a v. In particular, there is no need
|
||||
to synchronize the Memory flags of @a v and the ones of the @a i-th
|
||||
Vector blocks when data is moved between host and device. */
|
||||
inline void MakeRef(int i, Vector &v) { blocks[i] = &v; }
|
||||
};
|
||||
|
||||
// Inline and template methods
|
||||
|
||||
inline Vector &MultiVector::operator[](int i)
|
||||
{
|
||||
auto &bi = blocks[i];
|
||||
return (bi.index() == 0) ? std::get<0>(bi) : *std::get<1>(bi);
|
||||
}
|
||||
|
||||
inline const Vector &MultiVector::operator[](int i) const
|
||||
{
|
||||
auto &bi = blocks[i];
|
||||
return (bi.index() == 0) ? std::get<0>(bi) : *std::get<1>(bi);
|
||||
}
|
||||
|
||||
template <typename... VectorTypes,
|
||||
std::enable_if_t<
|
||||
std::conjunction_v<
|
||||
std::is_convertible<VectorTypes&,Vector&>...>, bool>>
|
||||
inline void MultiVector::MakeRef(VectorTypes &...vs)
|
||||
{
|
||||
blocks.resize(sizeof...(vs));
|
||||
if constexpr (sizeof...(vs) > 0)
|
||||
{
|
||||
const std::array vs_p{&static_cast<Vector&>(vs)...};
|
||||
for (std::size_t i = 0; i < sizeof...(vs); i++)
|
||||
{
|
||||
blocks[i] = vs_p[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_MULTIVECTOR_HPP
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user