Compare commits
671
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5a954e2b48 | ||
|
|
867c0a0e7e | ||
|
|
270c4d5175 | ||
|
|
0779fbbc72 | ||
|
|
7d80dfd93d | ||
|
|
a4638d6d61 | ||
|
|
5cd7f88da0 | ||
|
|
cc6a34b575 | ||
|
|
40ae2ffb7b | ||
|
|
b500757b68 | ||
|
|
445a8ba801 | ||
|
|
519e9185df | ||
|
|
2d98f8c8ae | ||
|
|
fa9fdc6576 | ||
|
|
e25f778f7b | ||
|
|
b5baebd0b3 | ||
|
|
c9ecc5b466 | ||
|
|
f38de5059e | ||
|
|
0ff0e1c76b | ||
|
|
983096e0a4 | ||
|
|
d4bd939578 | ||
|
|
91aa10b7eb | ||
|
|
b527d78a2c | ||
|
|
fcca0eaa71 | ||
|
|
0bfa06f87d | ||
|
|
3aed584bf7 | ||
|
|
c4d944cfad | ||
|
|
c05114555f | ||
|
|
dfc15d77e5 | ||
|
|
25e5ac7db0 | ||
|
|
d0533903f3 | ||
|
|
48e258a60c | ||
|
|
e1b9c355e6 | ||
|
|
c3db6ed873 | ||
|
|
56b3e1ecc6 | ||
|
|
5385e090f7 | ||
|
|
ecb7767b70 | ||
|
|
42bf39cd8c | ||
|
|
716f92d636 | ||
|
|
efdc17ce07 | ||
|
|
ffab9d82fa | ||
|
|
891eaf2829 | ||
|
|
053804f759 | ||
|
|
a8af93c4ec | ||
|
|
00a1f5ed64 | ||
|
|
a76b0dbc0e | ||
|
|
59ce107b33 | ||
|
|
ce1036ccc8 | ||
|
|
b6861ed26f | ||
|
|
8522cceb89 | ||
|
|
dec9509a5d | ||
|
|
5f02c6f64b | ||
|
|
8271da1517 | ||
|
|
05c0941cbb | ||
|
|
8e565163ec | ||
|
|
8a69b52f48 | ||
|
|
51a0f8accb | ||
|
|
8777bc0810 | ||
|
|
596b75282d | ||
|
|
de97775bd0 | ||
|
|
2ed23cec01 | ||
|
|
11fbd1f65c | ||
|
|
d3556dc2ac | ||
|
|
ad453255a5 | ||
|
|
2143ed5ca8 | ||
|
|
2aa8372cdc | ||
|
|
80c7331a29 | ||
|
|
9e31083745 | ||
|
|
fc8272216b | ||
|
|
0176c7664c | ||
|
|
e1c0705785 | ||
|
|
8efa320470 | ||
|
|
a4d114ad6e | ||
|
|
1e5609f6c9 | ||
|
|
c21d90589b | ||
|
|
bef7183ce7 | ||
|
|
c6ea42a681 | ||
|
|
f6eda1fce6 | ||
|
|
a1bcd7443d | ||
|
|
51955e88ce | ||
|
|
61bb7755a3 | ||
|
|
a7fa61464c | ||
|
|
93329a7ba0 | ||
|
|
9c4d742db8 | ||
|
|
16ac6c40a3 | ||
|
|
9a69d92756 | ||
|
|
fb93e21a0c | ||
|
|
ee2484383c | ||
|
|
1863dafeca | ||
|
|
a4c7828ac2 | ||
|
|
0a02f5737a | ||
|
|
154a1840d0 | ||
|
|
13e83cbd14 | ||
|
|
19d60f95ec | ||
|
|
f64405ebb6 | ||
|
|
92f2057650 | ||
|
|
861d318162 | ||
|
|
f3558cb78c | ||
|
|
3c5acae825 | ||
|
|
9a749f8936 | ||
|
|
c095d055a1 | ||
|
|
d29e914146 | ||
|
|
a2491a0e43 | ||
|
|
4475a8c6d4 | ||
|
|
d5ddb83c9d | ||
|
|
b5e99d8d83 | ||
|
|
91f58c293d | ||
|
|
4836e7cb53 | ||
|
|
6451e64637 | ||
|
|
b7ac1963bb | ||
|
|
869b0e48a6 | ||
|
|
baefb786fb | ||
|
|
99523b3a97 | ||
|
|
f67c1e5beb | ||
|
|
50af02454c | ||
|
|
aa94eb11b7 | ||
|
|
b73cc75635 | ||
|
|
75180606f8 | ||
|
|
3cac4326e4 | ||
|
|
da5e8b5844 | ||
|
|
081a90d1fa | ||
|
|
7b5f050bdc | ||
|
|
d06a3eec7f | ||
|
|
637fa7cc35 | ||
|
|
3f332d50b5 | ||
|
|
782e468325 | ||
|
|
0c59b93ecf | ||
|
|
5a0e89cf31 | ||
|
|
4c3bfce583 | ||
|
|
59d6820ff3 | ||
|
|
9ec379f176 | ||
|
|
2766e56f2d | ||
|
|
72cc503eb9 | ||
|
|
0c781120a2 | ||
|
|
449c9f5903 | ||
|
|
5aff935c98 | ||
|
|
6f793edfb7 | ||
|
|
6455a0c1fc | ||
|
|
b19c3c7e01 | ||
|
|
fb7c8af59e | ||
|
|
d56298ba17 | ||
|
|
b9d67dec34 | ||
|
|
5b664e393d | ||
|
|
c582282084 | ||
|
|
944ece4090 | ||
|
|
f225d0c3ef | ||
|
|
a9c98c2e3a | ||
|
|
ac8c2948e7 | ||
|
|
f75b4c10d2 | ||
|
|
f6ea1ea0da | ||
|
|
ecf6b0b44f | ||
|
|
ef6510b42e | ||
|
|
27ce32de07 | ||
|
|
341f6ec683 | ||
|
|
8c88e1e26a | ||
|
|
cabdb42bb4 | ||
|
|
e4a69c6245 | ||
|
|
5a52c676e1 | ||
|
|
5be9693235 | ||
|
|
5421abc4c4 | ||
|
|
f2d3fb45dc | ||
|
|
e447f090cc | ||
|
|
239c83a742 | ||
|
|
c315028a09 | ||
|
|
99b66f4f45 | ||
|
|
c3096d08bf | ||
|
|
67b906ea3b | ||
|
|
e7be50eb91 | ||
|
|
2dd3915ad5 | ||
|
|
29b572819a | ||
|
|
0a0acfda66 | ||
|
|
9167fc0c58 | ||
|
|
e48ffe3579 | ||
|
|
a7697c1db1 | ||
|
|
36c1cce132 | ||
|
|
7dbce44472 | ||
|
|
c67bd21219 | ||
|
|
38f257cf87 | ||
|
|
a0022e0330 | ||
|
|
38f5b93520 | ||
|
|
a078dfd59e | ||
|
|
3a2f286e2d | ||
|
|
66818f5525 | ||
|
|
b810a5e540 | ||
|
|
e63e421343 | ||
|
|
7031f7cb80 | ||
|
|
afaced7cd6 | ||
|
|
d671ca4712 | ||
|
|
1095aee346 | ||
|
|
176958144b | ||
|
|
9d191edf06 | ||
|
|
86609f139b | ||
|
|
420fcba457 | ||
|
|
ba8764300c | ||
|
|
510477a204 | ||
|
|
95acb1f85f | ||
|
|
0b9ae2202c | ||
|
|
60641259b2 | ||
|
|
99e2b5023d | ||
|
|
1c7261db07 | ||
|
|
e1b678664a | ||
|
|
4c59f6cfe6 | ||
|
|
d3e43a6423 | ||
|
|
8f06539b6e | ||
|
|
1f6d115d78 | ||
|
|
cee913cc04 | ||
|
|
686c8416c2 | ||
|
|
30a8eb5ccf | ||
|
|
bd5d928084 | ||
|
|
5e62d62c3d | ||
|
|
078e59a33b | ||
|
|
5faba43544 | ||
|
|
4b61294dc2 | ||
|
|
fced53cd29 | ||
|
|
f47447d92d | ||
|
|
a523710117 | ||
|
|
20805d87b4 | ||
|
|
0119d25dfc | ||
|
|
da4e1b5137 | ||
|
|
22c15267f9 | ||
|
|
6c83fec2da | ||
|
|
54b0a83ffd | ||
|
|
bc845844ea | ||
|
|
b5a1660c2d | ||
|
|
20ba3f3d0c | ||
|
|
07cd99fc3d | ||
|
|
f6eb88574f | ||
|
|
2631ba93ca | ||
|
|
5ea36c8fd6 | ||
|
|
b07fc2bb8e | ||
|
|
53b1b8f9a9 | ||
|
|
5ffc2ef502 | ||
|
|
35ceed5193 | ||
|
|
ce52e1f51f | ||
|
|
16a21b366e | ||
|
|
d540fa5a12 | ||
|
|
1221dea58e | ||
|
|
a5b9a7948f | ||
|
|
386e7e8c6a | ||
|
|
a709bdb9ee | ||
|
|
46c01f196a | ||
|
|
26077ab9aa | ||
|
|
240955c2cb | ||
|
|
da4a8e3412 | ||
|
|
59cef5f9e3 | ||
|
|
05be944a86 | ||
|
|
258bd917ad | ||
|
|
b9cf853dd3 | ||
|
|
915967925c | ||
|
|
87bdb90bcc | ||
|
|
848b54d13a | ||
|
|
88bc3b5833 | ||
|
|
bdd36c8982 | ||
|
|
c103cfa84a | ||
|
|
62214f61ae | ||
|
|
40bcad05c4 | ||
|
|
6c1c98e4fb | ||
|
|
9b93f1c1e1 | ||
|
|
be03c9703f | ||
|
|
b885fdc50f | ||
|
|
3888cba7c4 | ||
|
|
932b30e163 | ||
|
|
e4c0069ad9 | ||
|
|
395e4b0d0e | ||
|
|
a7988aa845 | ||
|
|
4ec768c82b | ||
|
|
9fe53c2403 | ||
|
|
e32ea54e00 | ||
|
|
630a75440f | ||
|
|
3ef3c8e6b4 | ||
|
|
00b9678b1d | ||
|
|
fe01ebf36c | ||
|
|
905de04020 | ||
|
|
145efc313d | ||
|
|
26b2aa5cea | ||
|
|
476c148949 | ||
|
|
faa3e22816 | ||
|
|
5e15a2e29b | ||
|
|
d168f5b489 | ||
|
|
7a5d13e86a | ||
|
|
2e0d1697b9 | ||
|
|
87168f51ab | ||
|
|
2558e32ee0 | ||
|
|
7521efeff9 | ||
|
|
8ed259be31 | ||
|
|
67025d49ff | ||
|
|
de1dea610e | ||
|
|
9f3f5c0372 | ||
|
|
3b23fd4941 | ||
|
|
d6a084322b | ||
|
|
d2a38dd3d4 | ||
|
|
17aab43082 | ||
|
|
e5c3cf4ee3 | ||
|
|
b09cc88bab | ||
|
|
620904124e | ||
|
|
6e55cfdc39 | ||
|
|
bef23c5770 | ||
|
|
a6fcf162a6 | ||
|
|
29caf08098 | ||
|
|
ba9af41877 | ||
|
|
2762b9dbfc | ||
|
|
d75a153df8 | ||
|
|
76d225439a | ||
|
|
5ee3f03902 | ||
|
|
d3a1144d10 | ||
|
|
20e38f3b10 | ||
|
|
9205efab48 | ||
|
|
1ccc27226a | ||
|
|
b20f61b3b8 | ||
|
|
1d0b49e5dd | ||
|
|
a9dcb20e84 | ||
|
|
68f6ce14a6 | ||
|
|
78905d471c | ||
|
|
61806ff1f7 | ||
|
|
0d3195e69b | ||
|
|
e217864f16 | ||
|
|
de8aacddce | ||
|
|
7b1656e19f | ||
|
|
c228538c17 | ||
|
|
a8f5fac0bb | ||
|
|
206eb51618 | ||
|
|
02226f934b | ||
|
|
3d0ba2251a | ||
|
|
b400ee6741 | ||
|
|
f40335f9e7 | ||
|
|
f0de33e33b | ||
|
|
b170c6ae54 | ||
|
|
c4b4ad3224 | ||
|
|
cecd75aff6 | ||
|
|
f37a596173 | ||
|
|
c982aa2448 | ||
|
|
94d1238637 | ||
|
|
d218d38af3 | ||
|
|
04dd962b6d | ||
|
|
383914db9a | ||
|
|
f77d238a5d | ||
|
|
84996ce32f | ||
|
|
5fa7ab3602 | ||
|
|
48cb5996b7 | ||
|
|
8e36285a98 | ||
|
|
27db27b088 | ||
|
|
8339ee0fe9 | ||
|
|
f2b64de28f | ||
|
|
3415b0f3d4 | ||
|
|
9f18d7e044 | ||
|
|
90820b76cf | ||
|
|
3adb2add4c | ||
|
|
5979dd1cce | ||
|
|
be999694b0 | ||
|
|
5d3b9ea727 | ||
|
|
a4d01470d8 | ||
|
|
a055c7ec63 | ||
|
|
50b197e754 | ||
|
|
69dc2b5142 | ||
|
|
fc26f0a773 | ||
|
|
6ea799e385 | ||
|
|
4e83a1604c | ||
|
|
dd0d879e7b | ||
|
|
d160b88706 | ||
|
|
007d5e3e43 | ||
|
|
fd78d48dfb | ||
|
|
f1561e47d1 | ||
|
|
f1174bfbf5 | ||
|
|
c1d1df3d80 | ||
|
|
abf5fedc5b | ||
|
|
4195e4ea2f | ||
|
|
d183f43c96 | ||
|
|
a545b94ad7 | ||
|
|
c0d32918b0 | ||
|
|
03937d0dc2 | ||
|
|
94828dbdd0 | ||
|
|
63e24c2a25 | ||
|
|
c202f9244a | ||
|
|
76cf9a2d00 | ||
|
|
4f796b3708 | ||
|
|
12eefe3c41 | ||
|
|
8e33891c07 | ||
|
|
62fbabe3a5 | ||
|
|
7b4df2d374 | ||
|
|
2d7460bde1 | ||
|
|
65acd08d38 | ||
|
|
0acc85d962 | ||
|
|
3c45d59813 | ||
|
|
63acbeb8c0 | ||
|
|
9bf6819f7a | ||
|
|
bed2cc5735 | ||
|
|
4e35641c28 | ||
|
|
8594867ab6 | ||
|
|
6e0fbbd3bf | ||
|
|
d8fd6d95c0 | ||
|
|
a2b34ab650 | ||
|
|
f478f687ff | ||
|
|
0bf998510f | ||
|
|
2399e5e294 | ||
|
|
205b694162 | ||
|
|
f2b661ca05 | ||
|
|
c8ef971f03 | ||
|
|
10aa1133b2 | ||
|
|
05c3a2c83f | ||
|
|
dbaff07ae9 | ||
|
|
d916299a49 | ||
|
|
10840ac6b4 | ||
|
|
d194a26542 | ||
|
|
a3a368dfd9 | ||
|
|
e8f39487b9 | ||
|
|
ec6e2a7c74 | ||
|
|
b437016f6e | ||
|
|
5170bbe010 | ||
|
|
f8fcdf6a97 | ||
|
|
fdfc019cc1 | ||
|
|
d10c908b38 | ||
|
|
cb1f54ed2b | ||
|
|
73aceea741 | ||
|
|
89f85cee21 | ||
|
|
0e9a9d9f7c | ||
|
|
c25be44dd6 | ||
|
|
00da00b93a | ||
|
|
4eca673111 | ||
|
|
4507a02249 | ||
|
|
9a18da5aaa | ||
|
|
4e31111827 | ||
|
|
0cbf5b5562 | ||
|
|
d81bbc1214 | ||
|
|
41255f308e | ||
|
|
96be0cafdf | ||
|
|
55fc2f806c | ||
|
|
3eac6fe764 | ||
|
|
f1ed582828 | ||
|
|
07e0d7cd4f | ||
|
|
c68cc62143 | ||
|
|
2c02b41d71 | ||
|
|
d71d1005f9 | ||
|
|
b8f1071168 | ||
|
|
d88529d632 | ||
|
|
7bd028b7fe | ||
|
|
339f20ea7f | ||
|
|
336d80e93a | ||
|
|
b64a189215 | ||
|
|
fc76ff8b2f | ||
|
|
e0a65ffaaf | ||
|
|
61c7187a86 | ||
|
|
a6bad19b8f | ||
|
|
174d991451 | ||
|
|
4fbefc6987 | ||
|
|
8dd75d2548 | ||
|
|
b0c2ec505f | ||
|
|
a03095d84d | ||
|
|
9ee63d6521 | ||
|
|
9917fa8306 | ||
|
|
d0324074c1 | ||
|
|
983d0f4361 | ||
|
|
eb95b46fad | ||
|
|
30016c83b8 | ||
|
|
cf5d93604e | ||
|
|
a0778c759e | ||
|
|
a922c1f2c3 | ||
|
|
17600f59f2 | ||
|
|
6cfec88f47 | ||
|
|
f11009d674 | ||
|
|
5df384f6ef | ||
|
|
daa555b68d | ||
|
|
76cfafe700 | ||
|
|
ada42c9fd8 | ||
|
|
0b802d8fce | ||
|
|
58976c6f41 | ||
|
|
9cdb604796 | ||
|
|
c55e3fa7d2 | ||
|
|
7b47ee4cf5 | ||
|
|
0114739e5e | ||
|
|
708b642be2 | ||
|
|
d78357494d | ||
|
|
9398b1a6e0 | ||
|
|
b23a087f42 | ||
|
|
72203ebb5c | ||
|
|
86b2ad3a61 | ||
|
|
75be9250a9 | ||
|
|
35ba876ecf | ||
|
|
2886c4b211 | ||
|
|
9455ed83b2 | ||
|
|
cc46bfdfcc | ||
|
|
010e96b6f0 | ||
|
|
211470966c | ||
|
|
be887d05a4 | ||
|
|
edf551c40a | ||
|
|
df386413a9 | ||
|
|
fa7fbdf36b | ||
|
|
30f1ad7c2c | ||
|
|
11debd6bf8 | ||
|
|
44a783993e | ||
|
|
2bbb317831 | ||
|
|
36835e62e0 | ||
|
|
70867ab87b | ||
|
|
4b59c90b08 | ||
|
|
b538e9d34c | ||
|
|
505661eb5f | ||
|
|
a3e229ef82 | ||
|
|
5e9d32a0f7 | ||
|
|
724866141a | ||
|
|
f17c9eef12 | ||
|
|
00b6dcdd37 | ||
|
|
d9913262df | ||
|
|
69c1d0d15b | ||
|
|
5f7d138aef | ||
|
|
4659602e63 | ||
|
|
1771a26426 | ||
|
|
b42a383201 | ||
|
|
19f444489f | ||
|
|
deae7e4997 | ||
|
|
6142015168 | ||
|
|
305a7f1e02 | ||
|
|
0bff07026c | ||
|
|
32c3e5dfd8 | ||
|
|
ee78679277 | ||
|
|
185f8b099d | ||
|
|
89a0f18b30 | ||
|
|
4cbc345ba6 | ||
|
|
abe3843712 | ||
|
|
3c2d1e4814 | ||
|
|
a8c4ed3c79 | ||
|
|
0aa73fa285 | ||
|
|
f1652b5ba0 | ||
|
|
fed0baf6b6 | ||
|
|
49a8ee5b54 | ||
|
|
c9569a0629 | ||
|
|
a4e83b5836 | ||
|
|
8b285047bd | ||
|
|
f999be0372 | ||
|
|
e639cfd75b | ||
|
|
2753692303 | ||
|
|
2722e979fc | ||
|
|
144c0ce106 | ||
|
|
b64172e12e | ||
|
|
c0072fcb84 | ||
|
|
37cd34a0e1 | ||
|
|
005be84af4 | ||
|
|
0ef682c156 | ||
|
|
3faa838774 | ||
|
|
f10c2793c9 | ||
|
|
9f4b83362a | ||
|
|
8ef423ca65 | ||
|
|
aff5173656 | ||
|
|
2164f07f01 | ||
|
|
fc0037da31 | ||
|
|
795a1a29a3 | ||
|
|
476d9c5111 | ||
|
|
64625f7333 | ||
|
|
4ae8c1bc2b | ||
|
|
4ae97f1834 | ||
|
|
816e8e4bea | ||
|
|
c14d630f14 | ||
|
|
899433f79f | ||
|
|
ff9294e5b0 | ||
|
|
16db883427 | ||
|
|
1834268091 | ||
|
|
947e25f769 | ||
|
|
9fbe90527a | ||
|
|
10c637837c | ||
|
|
076f1450e5 | ||
|
|
bfa672c644 | ||
|
|
30ac5d8d38 | ||
|
|
5a2b05cc35 | ||
|
|
afe388a573 | ||
|
|
50a9cce9f4 | ||
|
|
8ee2fdbfd9 | ||
|
|
d6b4594150 | ||
|
|
771e263db8 | ||
|
|
ee13fec158 | ||
|
|
665ba30f65 | ||
|
|
0162a6727d | ||
|
|
d6de2c1a1d | ||
|
|
c4b4cfdc2f | ||
|
|
3435475ae0 | ||
|
|
31fae9d195 | ||
|
|
6ce5fe8fac | ||
|
|
e43b58fa02 | ||
|
|
1739cebfdb | ||
|
|
27c8412439 | ||
|
|
4b9299188a | ||
|
|
5aaef22cc0 | ||
|
|
f9df36a6de | ||
|
|
b3a75a9295 | ||
|
|
b8c90d24f1 | ||
|
|
e0b2ba5e54 | ||
|
|
5cdc7499ca | ||
|
|
ee859d044d | ||
|
|
cf8d1ddd10 | ||
|
|
a7ca72c59f | ||
|
|
8267c7ea64 | ||
|
|
d2cf8fef73 | ||
|
|
2df3fc8ddb | ||
|
|
987f1636aa | ||
|
|
15faf0d225 | ||
|
|
a8eba594d2 | ||
|
|
1783050f9a | ||
|
|
52291cadbf | ||
|
|
1865b430b0 | ||
|
|
d8734b4b18 | ||
|
|
eae1fa217b | ||
|
|
59ee0d93c6 | ||
|
|
d8164fd158 | ||
|
|
23ce5f08e7 | ||
|
|
e059253549 | ||
|
|
7dc8b0d9fe | ||
|
|
3f236b406f | ||
|
|
47dcdd8159 | ||
|
|
d93d36d6d3 | ||
|
|
c89807da1e | ||
|
|
53de3bde2c | ||
|
|
56414350e2 | ||
|
|
6bcad07a5c | ||
|
|
d56493c2c5 | ||
|
|
42f2594430 | ||
|
|
5ba3e4de97 | ||
|
|
02938c9cce | ||
|
|
65313cd7d3 | ||
|
|
5246f9dbc3 | ||
|
|
4d352bf726 | ||
|
|
eed3bc067f | ||
|
|
7bd256e17c | ||
|
|
7dbad4da3e | ||
|
|
fb85c34ca4 | ||
|
|
6401ca5847 | ||
|
|
496e240837 | ||
|
|
016ebe62cc | ||
|
|
28ab39cf96 | ||
|
|
ec67fe536e | ||
|
|
eff793d6fa | ||
|
|
e5e7280a17 | ||
|
|
773dc57712 | ||
|
|
98a043b195 | ||
|
|
c2a3c83099 | ||
|
|
3ea90aff27 | ||
|
|
23a362d6ad | ||
|
|
9b90a7980b | ||
|
|
cf2c43b5c7 | ||
|
|
2cb6a6e899 | ||
|
|
3ec6292520 | ||
|
|
7011d623d3 | ||
|
|
d4326eddd3 | ||
|
|
39dcdb18e2 | ||
|
|
fb0abff1c2 | ||
|
|
8a615b8742 | ||
|
|
f24d9d8c0d | ||
|
|
85cf7b41d5 | ||
|
|
dff07dd1e3 | ||
|
|
f17c25caf0 | ||
|
|
929c7baf16 | ||
|
|
dfa845a91c | ||
|
|
7fe9733e4d | ||
|
|
1b3c326784 | ||
|
|
0fd42364ca | ||
|
|
7e2c9641c2 | ||
|
|
a0d18d4d52 | ||
|
|
7adea0556c | ||
|
|
1e8efef66b | ||
|
|
f14747eead | ||
|
|
d1151c09a3 | ||
|
|
a9501ed65f | ||
|
|
d16deb42f5 | ||
|
|
b8f88a6560 | ||
|
|
1653781a9d | ||
|
|
fcff34045e | ||
|
|
ae675a05ef | ||
|
|
7f370e8193 | ||
|
|
870732a5aa | ||
|
|
2bf4de6db4 | ||
|
|
41d15a3a0a | ||
|
|
daca70bf80 | ||
|
|
0e02aa947a | ||
|
|
44ca1fbf8f | ||
|
|
27b84d79b8 |
@@ -94,6 +94,16 @@ inputs:
|
||||
description: If true, do not set any CXXFLAGS or LDFLAGS.
|
||||
default: false
|
||||
|
||||
# Unfortunately, "uses:" fields cannot have references to variables like
|
||||
# ${{env.MFEM_ACTIONS_VERSION}}, so the branch/tag name has to be hard coded.
|
||||
# Therefore, in the future, when updating the version of the
|
||||
# mfem/github-actions to use, we'll have to replace:
|
||||
# - all definitions of MFEM_ACTIONS_VERSION and
|
||||
# - all "uses:" fields that refer to mfem/github-actions.
|
||||
MFEM_ACTIONS_VERSION:
|
||||
description: Version (branch or tag) of the mfem/github-actions to use.
|
||||
default: v2.7
|
||||
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
@@ -118,6 +128,7 @@ runs:
|
||||
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
|
||||
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
|
||||
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
|
||||
echo MFEM_ACTIONS_VERSION=${{inputs.MFEM_ACTIONS_VERSION}} >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- name: Env (dir)
|
||||
|
||||
@@ -53,7 +53,7 @@ runs:
|
||||
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- uses: mfem/github-actions/build-mfem@v2.5
|
||||
- uses: mfem/github-actions/build-mfem@v2.7
|
||||
if: ${{steps.debug.outputs.cache-hit != 'true'}}
|
||||
env:
|
||||
CXXFLAGS: ${{env.CXXFLAGS}}
|
||||
@@ -82,7 +82,7 @@ runs:
|
||||
run: find . -type f -name '*.o' -delete
|
||||
shell: bash
|
||||
|
||||
- uses: actions/upload-artifact@v4
|
||||
- uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: build-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build
|
||||
|
||||
@@ -12,6 +12,11 @@
|
||||
name: 'Install MPI'
|
||||
description: 'Installs MPI and set up its environment variables'
|
||||
|
||||
inputs:
|
||||
NO_FLAGS:
|
||||
description: If true, do not set any CXXFLAGS or LDFLAGS.
|
||||
default: false
|
||||
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
@@ -27,6 +32,7 @@ runs:
|
||||
shell: bash
|
||||
|
||||
- name: Env (bis)
|
||||
if: ${{ inputs.NO_FLAGS != 'true' }}
|
||||
run: |
|
||||
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
|
||||
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
|
||||
|
||||
@@ -49,7 +49,7 @@ runs:
|
||||
par: ${{inputs.par}}
|
||||
sanitizer: ${{inputs.sanitizer}}
|
||||
|
||||
- uses: actions/download-artifact@v4
|
||||
- uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: build-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build
|
||||
|
||||
@@ -37,14 +37,14 @@ runs:
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{env.MFEM_ACTIONS_VERSION}}
|
||||
|
||||
- uses: actions/cache/restore@v5 # Cache for Metis
|
||||
if: ${{inputs.par == 'true'}}
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
|
||||
|
||||
- name: Hypre/Metis links
|
||||
if: ${{inputs.par == 'true'}}
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# MFEM Pull Request Review Agent Guide
|
||||
|
||||
## Purpose and scope
|
||||
Review MFEM PRs for correctness, maintainability, performance, portability, test coverage, and MFEM consistency. Use the diff and PR context; reference source files, tests, and CI results when available. Follow `CONTRIBUTING.md`, especially Developer Guidelines, PR rules, checklist, and testing.
|
||||
|
||||
## Critical review pillars
|
||||
- Correctness and numerical behavior
|
||||
- API and user-facing impact
|
||||
- Performance implications
|
||||
- Maintainability and portability
|
||||
|
||||
## Review workflow
|
||||
1. Read the PR description, linked issues, and intended behavior.
|
||||
2. Inspect the diff before commenting.
|
||||
3. Identify affected MFEM components, examples, tests, build or docs changes, and downstream APIs.
|
||||
4. Analyze the code against the critical review pillars.
|
||||
5. Compare the change against nearby code and MFEM patterns; flag unmotivated deviations.
|
||||
6. Check whether tests and documentation were updated appropriately.
|
||||
7. Review CI results and suggest actions.
|
||||
8. Produce a structured review with prioritized findings.
|
||||
9. Always limit conclusions to available evidence.
|
||||
|
||||
## MFEM-specific review checklist
|
||||
- Component-aware scope: identify the touched subsystem (FEM, solvers, preconditioners, linear algebra, mesh, examples, miniapps, build, or docs) and assess its impact against the review pillars.
|
||||
- Numerical and algorithmic behavior: assess issues in convergence, stability, tolerances, precision, iteration limits, and failure handling. If clear opportunities exist to improve the algorithmic approach, call them out with expected impact.
|
||||
- API and user-facing impact: assess backward compatibility, user-visible behavior and default changes, migration impact, deprecations, and whether documentation clearly explains user-facing API changes.
|
||||
- Data structure and memory semantics: assess ownership, lifetime, aliasing, container behavior, and device-host synchronization.
|
||||
- Parallel and serial behavior: assess whether the change preserves equivalent semantics in serial and parallel modes where applicable; if logic is currently mode-specific, check whether extension to the other mode is straightforward (clear abstractions, no hard-wired assumptions), document constraints, and call out expected behavior differences explicitly.
|
||||
- Backend and portability impact: assess likely cross-backend risks in CPU, CUDA, HIP, OCCA, RAJA, partial assembly, fallback paths, compiler compatibility, and platform assumptions.
|
||||
- Build, dependency, and configuration impact: assess CMake or make changes, optional dependency behavior, and feature-flag interactions.
|
||||
- Tests and docs alignment: check available regression or unit coverage evidence for changed behavior, and ensure docs are updated for new flags, APIs, options, or behavior changes.
|
||||
- MFEM developer-guideline fit: keep code lean, simple, general, logically separated, and portable; suggest C++17 improvements when they clearly improve safety, clarity, or maintainability.
|
||||
- New source files, examples, or miniapps: if a PR adds source/header files, verify they are properly wired into the relevant `makefile` and `CMakeLists.txt`, referenced in docs where applicable (including `doc/CodeDocumentation.dox`), and added to top-level `.gitignore` only when generated artifacts require it.
|
||||
- Changelog: verify `CHANGELOG` is updated if the PR introduces significant new features or user-facing changes.
|
||||
- MFEM conventions: use `real_t`; use `mfem::out`/`mfem::err` instead of `std::cout`/`std::cerr` in library code; flag large/binary files; if AI assistance is apparent but undisclosed, suggest following `CONTRIBUTING.md`.
|
||||
- Edge cases: if the PR touches complex or error-prone areas, suggest additional tests for edge cases, failure modes, and parallel behavior.
|
||||
|
||||
## Commenting guidelines
|
||||
- Keep comments concise, actionable, and grounded in the diff.
|
||||
- Focus on correctness, behavior changes, and user impact over style nits.
|
||||
- Be professional, concise, collaborative, technically precise, and avoid unsupported assumptions.
|
||||
|
||||
@@ -13,7 +13,7 @@ Note that some of these scripts use the shared MFEM GitHub Actions from the exte
|
||||
|
||||
<https://github.com/mfem/github-actions>
|
||||
|
||||
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
|
||||
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch (or tag) in the above from which the action is taken.
|
||||
|
||||
The current CI workflows are:
|
||||
|
||||
|
||||
@@ -40,6 +40,7 @@ env:
|
||||
METIS_ARCHIVE_MAC: metis-4.0.3-mac.tgz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
MFEM_TOP_DIR: mfem
|
||||
MFEM_ACTIONS_VERSION: v2.7
|
||||
|
||||
# Note for future improvements:
|
||||
#
|
||||
@@ -170,20 +171,6 @@ jobs:
|
||||
env
|
||||
shell: bash
|
||||
|
||||
# For info on Xcode see:
|
||||
# - https://github.com/actions/runner-images/issues/12541
|
||||
# - https://github.com/actions/runner-images/blob/releases/macos-15-arm64/20250811/images/macos/macos-15-arm64-Readme.md#xcode
|
||||
- name: Xcode version setup (MacOS)
|
||||
if: matrix.os == 'macos-latest'
|
||||
run: |
|
||||
XCODE_PATH="/Applications/Xcode_16.4.app"
|
||||
echo "> sudo xcode-select -s ${XCODE_PATH}"
|
||||
sudo xcode-select -s ${XCODE_PATH}
|
||||
echo "> g++ -v"
|
||||
g++ -v
|
||||
echo "> clang++ -v"
|
||||
clang++ -v
|
||||
|
||||
# Only get MPI if defined for the job.
|
||||
# TODO: It would be nice to have only one step, e.g. with a dedicated
|
||||
# action, but I (@adrienbernede) don't see how at the moment.
|
||||
@@ -228,11 +215,11 @@ jobs:
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: get hypre
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -242,7 +229,7 @@ jobs:
|
||||
|
||||
- name: get hypre (Windows)
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -258,11 +245,11 @@ jobs:
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: install metis
|
||||
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
with:
|
||||
archive: ${{ matrix.os != 'macos-latest' && env.METIS_ARCHIVE || env.METIS_ARCHIVE_MAC }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
@@ -304,7 +291,7 @@ jobs:
|
||||
|
||||
# MFEM build and test
|
||||
- name: build
|
||||
uses: mfem/github-actions/build-mfem@v2.5
|
||||
uses: mfem/github-actions/build-mfem@v2.7
|
||||
env:
|
||||
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
|
||||
with:
|
||||
@@ -375,8 +362,10 @@ jobs:
|
||||
# Code coverage (process and upload reports)
|
||||
- name: codecov
|
||||
if: matrix.codecov == 'YES'
|
||||
uses: mfem/github-actions/upload-coverage@v2.5
|
||||
uses: mfem/github-actions/upload-coverage@v2.7
|
||||
with:
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
|
||||
project_dir: ${{ env.MFEM_TOP_DIR }}
|
||||
directories: "fem general linalg mesh"
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
|
||||
@@ -32,6 +32,7 @@ env:
|
||||
METIS_ARCHIVE: metis-4.0.3.tar.gz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
COVERAGE_ENV: mfem-coverage
|
||||
MFEM_ACTIONS_VERSION: v2.7
|
||||
|
||||
jobs:
|
||||
gitignore:
|
||||
@@ -53,33 +54,34 @@ jobs:
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
target: int32
|
||||
precision: fp64
|
||||
|
||||
- name: Cache Metis Install
|
||||
id: metis-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
|
||||
- name: Install Metis
|
||||
if: steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build-mfem
|
||||
uses: mfem/github-actions/build-mfem@v2.5
|
||||
uses: mfem/github-actions/build-mfem@v2.7
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: opt
|
||||
|
||||
@@ -19,18 +19,22 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
- name: Setup
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: ./.github/actions/sanitize/mpi
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Build
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
with:
|
||||
archive: ${{env.HYPRE_TGZ}}
|
||||
dir: ${{env.HYPRE_DIR}}
|
||||
|
||||
@@ -19,18 +19,22 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
|
||||
- name: Setup
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: ./.github/actions/sanitize/mpi
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Build
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
with:
|
||||
archive: ${{env.METIS_TGZ}}
|
||||
dir: ${{env.METIS_DIR}}
|
||||
|
||||
@@ -146,7 +146,7 @@ jobs:
|
||||
if: ${{steps.restore.outputs.cache-hit != 'true'}}
|
||||
working-directory: mfem/build/tests/unit
|
||||
run: find . -type f -name '*.o' -delete
|
||||
- uses: actions/upload-artifact@v4
|
||||
- uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build/tests/unit/${{env.unit_tests}}
|
||||
@@ -172,7 +172,7 @@ jobs:
|
||||
par: ${{inputs.par}}
|
||||
sanitizer: ${{inputs.sanitizer}}
|
||||
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
|
||||
- uses: actions/download-artifact@v4
|
||||
- uses: actions/download-artifact@v8
|
||||
if: ${{steps.restore.outputs.cache-hit != 'true'}}
|
||||
with:
|
||||
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
|
||||
+2
-2
@@ -451,8 +451,8 @@ miniapps/plasma/pic/*.csv
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
tests/unit/punit_tests
|
||||
tests/unit/cunit_tests
|
||||
tests/unit/pcunit_tests
|
||||
tests/unit/gpu_unit_tests
|
||||
tests/unit/pgpu_unit_tests
|
||||
tests/unit/sedov_tests_*
|
||||
tests/unit/psedov_tests_*
|
||||
tests/unit/tmop_pa_tests_*
|
||||
|
||||
@@ -85,3 +85,8 @@ opt_par_gcc_10_pumi:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +pumi"
|
||||
|
||||
opt_par_gcc_10_gslib:
|
||||
extends: .mfem_job_on_dane
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +gslib"
|
||||
|
||||
@@ -63,3 +63,8 @@ opt_mpi_cuda_hypre_cuda_gcc:
|
||||
extends: .mfem_job_on_matrix
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
|
||||
|
||||
opt_mpi_cuda_gcc_gslib:
|
||||
extends: .mfem_job_on_matrix
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +mpi +cuda +gslib cuda_arch=90 ^hypre+cuda"
|
||||
|
||||
@@ -32,9 +32,9 @@ mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
|
||||
|
||||
# run
|
||||
if [[ "${MACHINE_NAME}" == "dane" ]]; then
|
||||
salloc --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
srun --nodes=1 -t 60 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
elif [[ ${MACHINE_NAME} == "corona" ]]; then
|
||||
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
srun --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
|
||||
else
|
||||
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
|
||||
exit 1
|
||||
|
||||
@@ -8,43 +8,83 @@
|
||||
https://mfem.org
|
||||
|
||||
|
||||
Version 4.10 (development)
|
||||
==========================
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Replaced legacy simplex quadrature rules with symmetric positive-weight
|
||||
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
|
||||
rules guarantee all-positive weights and interior quadrature points,
|
||||
improving numerical stability. Higher orders fall back to Grundmann-Moller.
|
||||
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
|
||||
2015.
|
||||
Tet rules (d=1-13): Witherden & Vincent (ibid).
|
||||
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
|
||||
2022.
|
||||
|
||||
|
||||
Version 4.9.1 (development)
|
||||
===========================
|
||||
|
||||
- Added policy for AI-assisted contribution to CONTRIBUTING.md.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Improved the gridfunction projection routines. Projections work for Scalar,
|
||||
Vector and VectorFE, also NURBS versions. Optionally different types of
|
||||
projections can be selected, default behaviour has not changed.
|
||||
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
|
||||
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
|
||||
integrators.
|
||||
|
||||
- Added methods to estimate function extremum using piecewise linear bounds +
|
||||
- Replaced legacy simplex quadrature rules with symmetric positive weight rules
|
||||
for triangles (orders 0-25) and tetrahedra (orders 0-20). These rules
|
||||
guarantee all-positive weights and interior quadrature points, improving
|
||||
numerical stability. Higher orders fall back to Grundmann-Moller.
|
||||
* Triangle rules: Witherden and Vincent, DOI: 10.1016/j.camwa.2015.03.017
|
||||
* Tet rules (d=1-13): Witherden and Vincent (same as above)
|
||||
* Tet rules (d=14-20): Chuluunbaatar et al., DOI: 10.1016/j.camwa.2022.08.016
|
||||
|
||||
- Added support for general 1D Gauss-Jacobi quadrature rules and Stroud conical
|
||||
quadrature rules on triangles and tetrahedra.
|
||||
|
||||
- Improved the GridFunction projection routines. Projections work for Scalar,
|
||||
Vector and VectorFE, also NURBS versions. Optionally different types of
|
||||
projections can be selected, default behavior has not changed.
|
||||
|
||||
- Added GridFunction projection methods for trace spaces, i.e., project
|
||||
coefficients on the mesh skeleton.
|
||||
|
||||
- Added methods to estimate function extremum using piecewise linear bounds plus
|
||||
recursive subdivision.
|
||||
|
||||
- Extend FindPointsGSLIB to support surface meshes.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
|
||||
bounds on the determinant of the mesh transformation Jacobian.
|
||||
|
||||
- Added PA support for TMOP's adaptive limiting functionality. Multiple
|
||||
GridFunctions and Coefficients can be combined to form a composite term.
|
||||
|
||||
- Improved support for 1D NURBS meshes with variable order, including using
|
||||
the patches construct for 1D NURBS meshes.
|
||||
|
||||
- Added the option to include material interfaces (faces separating elements
|
||||
with different element attributes) as additional boundary elements, for
|
||||
parallel visualization, e.g. with GLVis. This is supported by both the Print
|
||||
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added support for trace spaces in PRefinementTransferOperator. This is used in
|
||||
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
|
||||
DPG miniapps).
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Added NVIDIA cuDSS library interface. Implementation examples have been
|
||||
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
|
||||
details. Supported versions >= 0.6.0.
|
||||
|
||||
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
|
||||
capability.
|
||||
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
|
||||
leverage the ParticleSet capability.
|
||||
|
||||
- Added (Complex)PRefinementMultigrid solver option in the DPG miniapps.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Fixed signed DOF handling in ParGridFunction reading (read constructor) and
|
||||
saving via SaveAsOne(). Simplified the process of applying the DOF signs by
|
||||
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
|
||||
method will return immediately if no sign flips are needed.
|
||||
|
||||
|
||||
Version 4.9, released on Dec 11, 2025
|
||||
|
||||
+10
-18
@@ -433,6 +433,15 @@ if (MFEM_USE_STRUMPACK)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# cuDSS can only be enabled in CUDA
|
||||
if (MFEM_USE_CUDSS)
|
||||
if (MFEM_USE_CUDA)
|
||||
find_package(CUDSS REQUIRED)
|
||||
else()
|
||||
message(FATAL_ERROR " *** cuDSS requires that CUDA be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# GnuTLS
|
||||
if (MFEM_USE_GNUTLS)
|
||||
find_package(_GnuTLS REQUIRED)
|
||||
@@ -592,13 +601,6 @@ if (MFEM_USE_ENZYME)
|
||||
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_PROTEUS)
|
||||
enable_language(C)
|
||||
find_package(proteus REQUIRED PATHS "${PROTEUS_DIR}")
|
||||
message(STATUS "${PROTEUS_DIR}/include")
|
||||
include_directories("${PROTEUS_DIR}/include")
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
if (APPLE)
|
||||
@@ -638,7 +640,7 @@ find_package(Threads REQUIRED)
|
||||
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
|
||||
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
|
||||
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CUDSS CALIPER CODIPACK
|
||||
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
|
||||
ALGOIM ENZYME CUDA::cudart)
|
||||
|
||||
@@ -735,16 +737,6 @@ mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
target_compile_features(mfem PUBLIC cxx_std_${CMAKE_CXX_STANDARD})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES} ${TPL_TARGETS})
|
||||
|
||||
if (MFEM_USE_PROTEUS)
|
||||
add_library(ClangProteusFlags INTERFACE IMPORTED)
|
||||
set_target_properties(ClangProteusFlags PROPERTIES
|
||||
INTERFACE_COMPILE_OPTIONS "-fpass-plugin=$<TARGET_FILE:ProteusPass>"
|
||||
)
|
||||
target_link_libraries(mfem PUBLIC ClangProteusFlags)
|
||||
target_link_libraries(mfem PUBLIC proteus)
|
||||
endif()
|
||||
|
||||
if (TPL_TARGETS)
|
||||
add_dependencies(mfem ${TPL_TARGETS})
|
||||
endif()
|
||||
|
||||
+73
-65
@@ -3,12 +3,13 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml?query=branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
</p>
|
||||
|
||||
|
||||
@@ -24,6 +25,14 @@ must be made under this license.
|
||||
Note also that MFEM has a [Code of Conduct](CODE_OF_CONDUCT.md). By participating
|
||||
in the MFEM community, you agree to abide by its rules.
|
||||
|
||||
## AI Policy
|
||||
- Use of AI code generation in MFEM is allowed but must be disclosed, e.g. by
|
||||
selecting the `AI-assisted` label on the PR.
|
||||
- By submitting a PR, the author acknowledges that they have reviewed and
|
||||
understand the changes they are proposing.
|
||||
- PR authors are still responsible for correctness, licensing, and attribution
|
||||
of all changes.
|
||||
|
||||
If you plan on contributing to MFEM, consider reviewing the
|
||||
[issue tracker](https://github.com/mfem/mfem/issues) first to check if a thread
|
||||
already exists for your desired feature or the bug you ran into. Use a pull
|
||||
@@ -76,7 +85,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
follow the [MFEM PR Rules](#mfem-pr-rules).
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
the `ready-for-review` label.
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning two
|
||||
- PRs are treated similarly to journal submission, with an "editor" assigning two
|
||||
reviewers to evaluate the changes.
|
||||
- The reviewers have 3 weeks to evaluate the PR and work with the author to
|
||||
fix issues and implement improvements.
|
||||
@@ -117,7 +126,7 @@ The MFEM source code has the following structure:
|
||||
│ ├── petsc
|
||||
│ ├── pumi
|
||||
│ ├── sundials
|
||||
| └── superlu
|
||||
│ └── superlu
|
||||
├── fem
|
||||
│ ├── ceed
|
||||
│ ├── dfem
|
||||
@@ -129,10 +138,6 @@ The MFEM source code has the following structure:
|
||||
│ ├── moonolith
|
||||
│ ├── qinterp
|
||||
│ └── tmop
|
||||
│ | ├── assemble
|
||||
│ | ├── metrics
|
||||
│ | ├── mult
|
||||
│ | └── tools
|
||||
├── general
|
||||
├── linalg
|
||||
│ ├── batched
|
||||
@@ -145,11 +150,10 @@ The MFEM source code has the following structure:
|
||||
│ ├── common
|
||||
│ ├── contact
|
||||
│ ├── dfem
|
||||
│ ├── diag-smoothers
|
||||
│ ├── dpg
|
||||
│ ├── electromagnetics
|
||||
│ ├── fluids
|
||||
│ │ ├── navier
|
||||
│ │ └── schrodinger-flow
|
||||
│ ├── gslib
|
||||
│ ├── hdiv-linear-solver
|
||||
│ ├── hooke
|
||||
@@ -159,6 +163,7 @@ The MFEM source code has the following structure:
|
||||
│ ├── nurbs
|
||||
│ ├── parelag
|
||||
│ ├── performance
|
||||
│ ├── plasma
|
||||
│ ├── shifted
|
||||
│ ├── solvers
|
||||
│ ├── spde
|
||||
@@ -189,15 +194,15 @@ respectively.
|
||||
|
||||
- The main finite element classes are:
|
||||
+ [`FiniteElement`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElementCollection.html)
|
||||
+ [`FiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1FiniteElementSpace.html)
|
||||
+ [`GridFunction`](https://docs.mfem.org/html/classmfem_1_1GridFunction.html)
|
||||
+ [`BilinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html)
|
||||
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
|
||||
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
|
||||
|
||||
- The main linear algebra classes and sources are
|
||||
+ [`Operator`](https://docs.mfem.org/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html)
|
||||
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
|
||||
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1Vector.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
|
||||
+ [`DenseMatrix`](https://docs.mfem.org/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://docs.mfem.org/html/classmfem_1_1SparseMatrix.html)
|
||||
+ Sparse [smoothers](https://docs.mfem.org/html/sparsesmoothers_8hpp.html) and linear [solvers](https://docs.mfem.org/html/solvers_8hpp.html)
|
||||
|
||||
@@ -209,8 +214,8 @@ shared geometric entities between different tasks. The parallel source files
|
||||
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
|
||||
- The main parallel classes are
|
||||
+ [`ParMesh`](https://docs.mfem.org/html/solvers_8hpp.html)
|
||||
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParNCMesh.html)
|
||||
+ [`ParFiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1ParFiniteElementSpace.html)
|
||||
+ [`ParGridFunction`](https://docs.mfem.org/html/classmfem_1_1ParGridFunction.html)
|
||||
+ [`ParBilinearForm`](https://docs.mfem.org/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://docs.mfem.org/html/classmfem_1_1ParLinearForm.html)
|
||||
@@ -220,14 +225,14 @@ have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
#### GPU and general device support
|
||||
|
||||
GPU and multi-core CPU support is based on device kernels supporting different
|
||||
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
|
||||
backends (CUDA, HIP, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
|
||||
device/host memory manager.
|
||||
|
||||
- The main device-relevant classes and sources are:
|
||||
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
|
||||
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
|
||||
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
|
||||
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
|
||||
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html), [`hip.hpp`](https://docs.mfem.org/html/hip_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
|
||||
|
||||
#### Utilities, building and documentation
|
||||
- The `general/` directory contains C++ classes that serve as utilities for
|
||||
@@ -241,8 +246,8 @@ device/host memory manager.
|
||||
- `examples` and `miniapps` respectively gather simple and more fully-featured
|
||||
demonstrations of the usage on MFEM. They both rely on `data/` for the
|
||||
collection of meshes.
|
||||
- The `tests/` directory contains a unit test suite and will later contain more
|
||||
tests that run example codes.
|
||||
- The `tests/` directory contains a unit test suite, additional tests, and
|
||||
benchmarks.
|
||||
|
||||
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
|
||||
website.
|
||||
@@ -276,8 +281,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
the top of https://github.com/mfem.
|
||||
- Consider making your membership public by going to https://github.com/orgs/mfem/people
|
||||
and clicking on the organization visibility drop box next to your name.
|
||||
- Project discussions and announcements will be posted at
|
||||
https://github.com/orgs/mfem/teams/everyone.
|
||||
- Project discussions and announcements will be posted at https://github.com/orgs/mfem/discussions,
|
||||
tagging the `@mfem/everyone` team when appropriate.
|
||||
|
||||
#### Structure
|
||||
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
|
||||
@@ -337,11 +342,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- Well-designed simple code is frequently more general and powerful.
|
||||
- Lean code base is easier to understand by new collaborators.
|
||||
- New features should be added only if they are necessary or generally useful.
|
||||
- Introduction of language constructions not currently used in MFEM should be
|
||||
- Introduction of language constructs not currently used in MFEM should be
|
||||
justified and generally avoided (to maintain portability to various systems
|
||||
and compilers, including early access hardware).
|
||||
- We prefer basic C++ and the C++03 standard, to keep the code readable by
|
||||
a large audience and to make sure it compiles anywhere.
|
||||
- We prefer basic C++. Use C++17 features judiciously, prioritizing readability,
|
||||
consistency with existing MFEM code, and portability to different systems,
|
||||
compilers and device backends.
|
||||
|
||||
- *Keep the code general and reasonably efficient*
|
||||
- The main goal is fast prototyping for research and application development.
|
||||
@@ -384,7 +390,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- When your branch is ready for other developers to review / comment on
|
||||
the code, create a pull request towards `mfem:master`.
|
||||
|
||||
- Pull request typically have titles like:
|
||||
- Pull requests typically have titles like:
|
||||
|
||||
`Description [new-feature-dev]`
|
||||
|
||||
@@ -405,12 +411,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
|
||||
team will add reviewers as appropriate.
|
||||
|
||||
- List outstanding TODO items in the description, see PR #222 for an example.
|
||||
- List outstanding TODO items in the description.
|
||||
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
the `ready-for-review` label.
|
||||
or request the `ready-for-review` label.
|
||||
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning
|
||||
- PRs are treated similarly to journal submission, with an "editor" assigning
|
||||
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
|
||||
the PR and work with the author to implement improvements and fix issues.
|
||||
|
||||
@@ -436,7 +442,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
checks in GitHub Actions enforce MFEM-specific rules which are explained in
|
||||
the error messages and the `tests/scripts` directory.
|
||||
|
||||
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
|
||||
- Also note that the tests `branch-history` and `repo-check` found in GitHub
|
||||
Actions can be triggered automatically before each push using git hooks. See
|
||||
the [git hooks README](config/githooks/README.md) for a detailed explanation.
|
||||
|
||||
@@ -493,15 +499,15 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
|
||||
|
||||
3. To ensure the quality of the PR by making sure that the code adheres to the [Developer Guidelines](#developer-guidelines), e.g. all methods, data members, and functions have documentation, including data ownership and lifetime, new examples/miniapps have a corresponding PR in mfem/web, major features have `CHANGELOG` entries, etc.
|
||||
|
||||
3. To seek help from the editors in case of difficulties.
|
||||
4. To seek help from the editors in case of difficulties.
|
||||
|
||||
4. To complete the review in a timely manner: 3 weeks from assignment.
|
||||
5. To complete the review in a timely manner: 3 weeks from assignment.
|
||||
|
||||
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
|
||||
6. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
|
||||
|
||||
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
|
||||
7. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
|
||||
|
||||
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
|
||||
8. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
|
||||
|
||||
#### Responsibilities of Authors
|
||||
|
||||
@@ -527,30 +533,30 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Code builds.
|
||||
- [ ] Code passes `make style`.
|
||||
- [ ] Update `CHANGELOG`:
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Update `INSTALL`:
|
||||
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
|
||||
- [ ] Have the version ranges for any required or optional libraries changed?
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*
|
||||
- [ ] Has a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
|
||||
- [ ] Have the version ranges for any required or optional libraries changed?
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*
|
||||
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] Update `.gitignore`:
|
||||
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] New examples:
|
||||
- [ ] All sample runs at the top of the example source file work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] All sample runs at the top of the example source file work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
|
||||
- [ ] Add any files generated by it to the `clean` target.
|
||||
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
|
||||
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
|
||||
- [ ] List the new example in `doc/CodeDocumentation.dox`.
|
||||
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] If new examples directory (e.g. `examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
@@ -567,13 +573,13 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
|
||||
- [ ] Consider adding a new test for the new miniapp.
|
||||
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
|
||||
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New capability:
|
||||
- [ ] All new public, protected, and private classes, methods, data members, and functions have full Doxygen-style documentation in source comments. Documentation should include descriptions of member data, function arguments and return values, template parameters, and prerequisites for calling new functions.
|
||||
- [ ] Pointer arguments and return values must specify whether ownership is being transferred or lent with the call.
|
||||
@@ -675,7 +681,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
- [ ] Update URL shortlinks:
|
||||
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
|
||||
- [ ] Add the new shortlinks to the MFEM package in `spack`.
|
||||
- [ ] Update website in `mfem/web` repo:
|
||||
- Update version and shortlinks in `src/index.md` and `src/download.md`.
|
||||
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
|
||||
@@ -727,22 +733,24 @@ commit or push, see the [README](config/githooks/README.md) in the `config/githo
|
||||
directory.
|
||||
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
### GitHub Actions smoke tests
|
||||
|
||||
We use GitHub Actions to drive the default tests on the `master` and `next`
|
||||
branches. See the `.github/workflows` files and the logs at
|
||||
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
|
||||
|
||||
Testing using GitHub Actions should be kept lightweight, as there is a time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
GitHub Actions testing should be kept lightweight, as there is a time
|
||||
constraint on jobs. The current workflows cover Linux, macOS, and Windows
|
||||
configurations.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `next` branch are currently scheduled to run each night.
|
||||
|
||||
### Additional Windows smoke test
|
||||
|
||||
### Windows smoke test
|
||||
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor` file and the
|
||||
build logs at
|
||||
We also use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor.yml` file
|
||||
and the build logs at
|
||||
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
|
||||
|
||||
CMake is used to generate the MSVC Project files and drive the build. A release
|
||||
|
||||
@@ -38,14 +38,13 @@ the option MFEM_USE_METIS.
|
||||
MFEM also includes support for devices such as GPUs, and programming models such
|
||||
as CUDA, HIP, OCCA, OpenMP and RAJA.
|
||||
|
||||
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
|
||||
a newer compiler, e.g. GCC version 4.9 or higher.
|
||||
- Starting with version 4.9, MFEM requires a C++17 compiler.
|
||||
|
||||
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
|
||||
https://developer.nvidia.com/cuda-toolkit
|
||||
|
||||
- HIP support requires an AMD GPU and an installation of the ROCm software stack
|
||||
https://rocmdocs.amd.com
|
||||
https://rocm.docs.amd.com
|
||||
|
||||
- OCCA support requires the OCCA library
|
||||
https://libocca.org
|
||||
@@ -83,9 +82,9 @@ Serial build:
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
make parallel -j 4
|
||||
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
|
||||
CUDA build:
|
||||
make cuda -j 4
|
||||
@@ -115,14 +114,14 @@ Serial build:
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
|
||||
make -j 4
|
||||
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
|
||||
Parallel build with fetching of hypre and METIS:
|
||||
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
|
||||
make -j 4
|
||||
|
||||
@@ -134,7 +133,8 @@ CUDA build:
|
||||
|
||||
HIP build:
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
|
||||
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 \
|
||||
-DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
|
||||
make -j 4
|
||||
|
||||
Example codes (serial/parallel, depending on the build):
|
||||
@@ -269,6 +269,7 @@ Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
|
||||
HIP_CXX - The HIP compiler, e.g. 'hipcc'
|
||||
|
||||
Compiler options:
|
||||
OPTIM_FLAGS - Options for optimized build
|
||||
@@ -395,6 +396,11 @@ MFEM_USE_STRUMPACK = YES/NO
|
||||
classes. When enabled, this option uses the STRUMPACK_* library options, see
|
||||
below.
|
||||
|
||||
MFEM_USE_CUDSS = YES/NO
|
||||
Enable MFEM functionality based on the cuDSS library. When using cuDSS, CUDA
|
||||
support must be also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
When enabled, this option uses the CUDSS_* library options, see below.
|
||||
|
||||
MFEM_USE_GINKGO = YES/NO
|
||||
Enable MFEM functionality based on the Ginkgo library, which provides
|
||||
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
|
||||
@@ -554,13 +560,13 @@ MFEM_USE_RAJA = YES/NO
|
||||
MFEM_USE_OCCA = YES/NO
|
||||
Enables support for the OCCA library in MFEM. OCCA is an open-source library
|
||||
which aims to make it easy to program different types of devices (e.g. CPU,
|
||||
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
|
||||
GPU, FPGA) by providing a unified API for interacting with JIT-compiled
|
||||
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
|
||||
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
|
||||
MFEM_USE_GSLIB = YES/NO
|
||||
Enables MFEM functionality based on the GSLIB library, and specifically its
|
||||
FindPoints component, which provides a robust algorithms to evaluate finite
|
||||
FindPoints component, which provides robust algorithms to evaluate finite
|
||||
element functions in a collection of points in physical space. When enabled,
|
||||
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
|
||||
|
||||
@@ -719,9 +725,18 @@ The specific libraries and their options are:
|
||||
Options: STRUMPACK_OPT, STRUMPACK_LIB.
|
||||
Versions: STRUMPACK >= 3.0.0.
|
||||
|
||||
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
|
||||
C++ compiler that supports the C++-17 standard. For additional requirements
|
||||
and dependencies of specific modules, see the Ginkgo webpage below.
|
||||
- CUDSS (optional), used when MFEM_USE_CUDSS = YES. Note that CUDSS requires
|
||||
CUDA 12.x toolkit and the cuDSS libraries. The supported communication backend
|
||||
is OpenMPI 4.x (default), and OpenMPI 4.x or a later version must be pre-built.
|
||||
The source files in the cuDSS tarball provide guidance for developing custom
|
||||
MPI implementations.
|
||||
URL: https://developer.nvidia.com/cudss
|
||||
https://docs.nvidia.com/cuda/cudss/advanced_features.html#communication-layer-library-in-cudss
|
||||
Options: CUDSS_OPT, CUDSS_LIB.
|
||||
Versions: cuDSS >= 0.6.0.
|
||||
|
||||
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Ginkgo may have additional
|
||||
requirements and module-specific dependencies; see the webpage below.
|
||||
URL: https://ginkgo-project.github.io
|
||||
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or
|
||||
Debug).
|
||||
@@ -793,7 +808,7 @@ The specific libraries and their options are:
|
||||
Options: CONDUIT_OPT, CONDUIT_LIB.
|
||||
Versions: Conduit >= 0.3.1.
|
||||
|
||||
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
|
||||
- ADIOS2 (optional), used when MFEM_USE_ADIOS2 = YES.
|
||||
URL: https://adios2.readthedocs.io/
|
||||
Versions: ADIOS >= 2.5.0.
|
||||
|
||||
@@ -869,7 +884,7 @@ The specific libraries and their options are:
|
||||
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
|
||||
Versions: RAJA >= 2022.10.3.
|
||||
|
||||
- Moonolith (optional), use when MFEM_USE_MOONOLITH = YES.
|
||||
- Moonolith (optional), used when MFEM_USE_MOONOLITH = YES.
|
||||
URL: https://bitbucket.org/zulianp/par_moonolith
|
||||
Options: MOONOLITH_DIR
|
||||
Versions: MOONOLITH >= 1.1.0.
|
||||
@@ -957,7 +972,7 @@ CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
|
||||
To use a specific generator use the "-G <generator>" option of cmake:
|
||||
|
||||
cmake <mfem-source-dir> -G "Xcode"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 17 2022"
|
||||
cmake <mfem-source-dir> -G "MinGW Makefiles"
|
||||
|
||||
With CMake it is possible to build MFEM as a shared library using the standard
|
||||
@@ -1202,7 +1217,7 @@ larger problems, there are two options:
|
||||
Specific options for HIP
|
||||
========================
|
||||
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
|
||||
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
|
||||
ROCm install, as well as having `$ROCM_PATH/bin` in `PATH`.
|
||||
|
||||
Specific options for RAJA+HIP+MPI
|
||||
=================================
|
||||
|
||||
@@ -28,6 +28,7 @@ license files. These software products and their licenses are as follows:
|
||||
* AmgXWrapper (linalg/amgxsolver.{hpp,cpp}) -- MIT license
|
||||
* Catch++ (tests/unit/catch.hpp) -- Boost 1.0 license
|
||||
* Gecko (general/gecko.{cpp,hpp}) -- BSD 3-clause license
|
||||
* gslib (fem/gslib.{cpp,hpp}, mesh/bb_grid_map.{cpp,hpp}) -- BSD 3-clause license
|
||||
* Picojson (fem/picojson.h) -- Custom 2-clause license
|
||||
* TinyXML2 (general/tinyxml2.{cpp,h}) -- zlib license
|
||||
* Zstr (general/zstr.hpp) -- MIT license
|
||||
|
||||
@@ -35,6 +35,7 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
|
||||
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
|
||||
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
|
||||
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
|
||||
set(MFEM_USE_CUDSS @MFEM_USE_CUDSS@)
|
||||
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
|
||||
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
|
||||
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
|
||||
@@ -109,6 +110,10 @@ if (MFEM_USE_RAJA)
|
||||
find_dependency(RAJA)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CUDSS)
|
||||
find_dependency(cudss)
|
||||
endif (MFEM_USE_CUDSS)
|
||||
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_dependency(umpire)
|
||||
endif()
|
||||
|
||||
@@ -108,6 +108,15 @@
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable MFEM functionality based on the cuDSS library.
|
||||
#cmakedefine MFEM_USE_CUDSS
|
||||
|
||||
// CUDSS communication layer library path
|
||||
#cmakedefine MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
|
||||
|
||||
// CUDSS threading layer library path
|
||||
#cmakedefine MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
|
||||
|
||||
// Enable functionality based on the Ginkgo library.
|
||||
#cmakedefine MFEM_USE_GINKGO
|
||||
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
if (NOT cudss_DIR AND CUDSS_DIR)
|
||||
set(cudss_DIR ${CUDSS_DIR}/lib/cmake/cudss)
|
||||
endif()
|
||||
message(STATUS "Looking for CUDSS ...")
|
||||
message(STATUS " in CUDSS_DIR = ${CUDSS_DIR}")
|
||||
message(STATUS " cudss_DIR = ${cudss_DIR}")
|
||||
find_package(cudss)
|
||||
set(CUDSS_FOUND ${cudss_FOUND})
|
||||
set(CUDSS_LIBRARIES "cudss")
|
||||
if (CUDSS_FOUND)
|
||||
message(STATUS
|
||||
"Found CUDSS target: ${CUDSS_LIBRARIES} (version: ${cudss_VERSION})")
|
||||
else()
|
||||
set(msg STATUS)
|
||||
if (CUDSS_FIND_REQUIRED)
|
||||
set(msg FATAL_ERROR)
|
||||
endif()
|
||||
message(${msg}
|
||||
"CUDSS not found. Please set CUDSS_DIR to the install prefix.")
|
||||
endif()
|
||||
|
||||
if(CUDSS_FOUND AND TARGET cudss)
|
||||
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION)
|
||||
if(NOT CUDSS_LIBRARY_LOCATION)
|
||||
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION_RELEASE)
|
||||
endif()
|
||||
if(CUDSS_LIBRARY_LOCATION)
|
||||
get_filename_component(CUDSS_LIBRARY_DIR "${CUDSS_LIBRARY_LOCATION}" DIRECTORY)
|
||||
else()
|
||||
message(WARNING "Could not determine the location of the cuDSS library.")
|
||||
endif()
|
||||
else()
|
||||
message(WARNING "cuDSS target not available; cannot determine library directory.")
|
||||
endif()
|
||||
|
||||
# Set the full name of the cuDSS threading library if OpenMP is enabled.
|
||||
# The threading layer library (libcudss_mtlayer_gomp.so) is located under the
|
||||
# cuDSS library directory by default.
|
||||
if (MFEM_USE_OPENMP)
|
||||
find_file(
|
||||
CUDSS_THREADING_LIB
|
||||
NAMES libcudss_mtlayer_gomp.so
|
||||
PATHS ${CUDSS_LIBRARY_DIR}
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if (NOT DEFINED MFEM_CUDSS_THREADING_LIB AND CUDSS_THREADING_LIB)
|
||||
set(MFEM_CUDSS_THREADING_LIB "${CUDSS_THREADING_LIB}")
|
||||
endif()
|
||||
message(STATUS "CUDSS threading layer library: ${MFEM_CUDSS_THREADING_LIB}")
|
||||
endif()
|
||||
|
||||
# Set the full name of the cuDSS communication library if MFEM use OpenMPI.
|
||||
# The communication layer library (libcudss_commlayer_mpi.so) is located under the
|
||||
# cuDSS library directory by default.
|
||||
# The communication layer library is used pre-built communication layers for OpenMPI
|
||||
# by default.
|
||||
if (MFEM_USE_MPI)
|
||||
find_file(
|
||||
CUDSS_COMM_LIB
|
||||
NAMES libcudss_commlayer_openmpi.so
|
||||
PATHS ${CUDSS_LIBRARY_DIR}
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if (NOT DEFINED MFEM_CUDSS_COMM_LIB AND CUDSS_COMM_LIB)
|
||||
set(MFEM_CUDSS_COMM_LIB "${CUDSS_COMM_LIB}")
|
||||
endif()
|
||||
message(STATUS "CUDSS communication layer library: ${MFEM_CUDSS_COMM_LIB}")
|
||||
endif()
|
||||
+4
-16
@@ -157,22 +157,10 @@ constexpr real_t operator""_r(unsigned long long v)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
#ifdef NVTX_DBG_HPP
|
||||
#include NVTX_DBG_HPP
|
||||
#else
|
||||
#define db1(...)
|
||||
#define dbg(...)
|
||||
#define dbl(...)
|
||||
#define dba(...)
|
||||
#define dbc(...)
|
||||
#define NVTX_MARK_FUNCTION
|
||||
#define NVTX_MARK_BEGIN(...)
|
||||
#define NVTX_INI(...)
|
||||
#define NVTX_END(...)
|
||||
#define NVTX_MARK_INI(...)
|
||||
#define NVTX_MARK_END(...)
|
||||
#define NVTX_MARK(...)
|
||||
#define NVTX(...)
|
||||
#ifndef MFEM_USE_CUDA
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
#error Building with cuDSS (MFEM_USE_CUDSS=YES) requires CUDA (MFEM_USE_CUDA=YES)
|
||||
#endif
|
||||
#endif // MFEM_USE_CUDSS not defined
|
||||
|
||||
#endif // MFEM_CONFIG_HPP
|
||||
|
||||
@@ -108,6 +108,15 @@
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
// #define MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable MFEM functionality based on the cuDSS library.
|
||||
// #define MFEM_USE_CUDSS
|
||||
|
||||
// CUDSS communication layer library path
|
||||
// #define MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
|
||||
|
||||
// CUDSS threading layer library path
|
||||
// #define MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
|
||||
|
||||
// Enable MFEM features based on the Ginkgo library.
|
||||
// #define MFEM_USE_GINKGO
|
||||
|
||||
|
||||
@@ -36,6 +36,9 @@ MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
|
||||
MFEM_USE_SUPERLU5 = @MFEM_USE_SUPERLU5@
|
||||
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
|
||||
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
|
||||
MFEM_USE_CUDSS = @MFEM_USE_CUDSS@
|
||||
MFEM_CUDSS_COMM_LIB = @MFEM_CUDSS_COMM_LIB@
|
||||
MFEM_CUDSS_THREADING_LIB = @MFEM_CUDSS_THREADING_LIB@
|
||||
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
|
||||
MFEM_USE_AMGX = @MFEM_USE_AMGX@
|
||||
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
|
||||
|
||||
@@ -38,6 +38,7 @@ option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
|
||||
option(MFEM_USE_SUPERLU5 "Use the old SuperLU_DIST 5.1 version" OFF)
|
||||
option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
|
||||
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
|
||||
option(MFEM_USE_CUDSS "Enable cuDSS usage" OFF)
|
||||
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
|
||||
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
|
||||
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
|
||||
|
||||
+15
-1
@@ -153,6 +153,7 @@ MFEM_USE_SUPERLU = NO
|
||||
MFEM_USE_SUPERLU5 = NO
|
||||
MFEM_USE_MUMPS = NO
|
||||
MFEM_USE_STRUMPACK = NO
|
||||
MFEM_USE_CUDSS = NO
|
||||
MFEM_USE_GINKGO = NO
|
||||
MFEM_USE_AMGX = NO
|
||||
MFEM_USE_MAGMA = NO
|
||||
@@ -368,6 +369,19 @@ STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
|
||||
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
|
||||
$(SCOTCH_LIB) $(SCALAPACK_LIB)
|
||||
|
||||
# CUDSS library configuration
|
||||
CUDSS_DIR = @MFEM_DIR@/../cudss
|
||||
CUDSS_INCLUDE_DIR = $(CUDSS_DIR)/include
|
||||
CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
|
||||
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
|
||||
CUDSS_LIB = \
|
||||
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
|
||||
# The cuDSS communication and threading libraries.
|
||||
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
|
||||
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
|
||||
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
|
||||
$(subst @MFEM_DIR@,$(MFEM_DIR),$(CUDSS_LIBRARY_DIR)/libcudss_mtlayer_gomp.so))))
|
||||
|
||||
# Ginkgo library configuration
|
||||
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
|
||||
GINKGO_SEARCH_DIR = $(subst @MFEM_DIR@,$(MFEM_DIR),$(GINKGO_DIR))
|
||||
@@ -621,7 +635,7 @@ PARELAG_LIB = -L$(PARELAG_DIR)/build/src -lParELAG
|
||||
AXOM_DIR = @MFEM_DIR@/../axom
|
||||
TRIBOL_DIR = @MFEM_DIR@/../tribol
|
||||
TRIBOL_OPT = -I$(TRIBOL_DIR)/include -I$(AXOM_DIR)/include
|
||||
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
|
||||
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -ltribol_shared -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
|
||||
-laxom_slam -laxom_slic -laxom_core
|
||||
|
||||
# Enzyme configuration
|
||||
|
||||
@@ -47,7 +47,6 @@ list(APPEND ALL_EXE_SRCS
|
||||
ex39.cpp
|
||||
ex40.cpp
|
||||
ex41.cpp
|
||||
jitplayground.cpp
|
||||
)
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
@@ -216,7 +215,7 @@ if (MFEM_ENABLE_TESTING)
|
||||
add_test(NAME ex1p_ceed_np=${MFEM_MPI_NP}
|
||||
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
|
||||
${MPIEXEC_PREFLAGS}
|
||||
$<TARGET_FILE:ex1p> "-no-vis" "-d ceed-cpu" "-pa" "-a"
|
||||
$<TARGET_FILE:ex1p> "-no-vis" "-d" "ceed-cpu" "-pa" "-a"
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -64,7 +64,7 @@ PARALLEL_NAME := Parallel AMGX example
|
||||
$(MFEM_LIB_FILE):
|
||||
$(error The MFEM library is not build)
|
||||
|
||||
clean: clean-build
|
||||
clean: clean-build clean-exec
|
||||
|
||||
clean-build:
|
||||
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
|
||||
|
||||
@@ -64,12 +64,12 @@ ex1p-test-par: ex1p
|
||||
$(MFEM_LIB_FILE):
|
||||
$(error The MFEM library is not built)
|
||||
|
||||
clean: clean-build clean-exec $(SUBDIRS_CLEAN)
|
||||
clean: clean-build clean-exec
|
||||
|
||||
clean-build:
|
||||
rm -f *.o *~ $(SEQ_EXAMPLES) $(PAR_EXAMPLES)
|
||||
rm -rf *.dSYM *.TVD.*breakpoints
|
||||
|
||||
clean-exec:
|
||||
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh
|
||||
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.*
|
||||
@rm -f refined.mesh mesh.*
|
||||
@rm -f sol.*
|
||||
|
||||
+36
-23
@@ -50,6 +50,10 @@
|
||||
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cpu
|
||||
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
|
||||
//
|
||||
// Device simplices sample runs:
|
||||
// ex1 -pa -d gpu -m ../data/inline-tet.mesh
|
||||
// ex1 -pa -d gpu -m ../data/inline-tri.mesh
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
@@ -138,25 +142,25 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 5. Define a finite element space on the mesh. Here we use continuous
|
||||
// Lagrange finite elements of the specified order. If order < 1, we
|
||||
// instead use an isoparametric/isogeometric space.
|
||||
// Lagrange finite elements of the specified order.
|
||||
// - If order < 1, we instead use an isoparametric/isogeometric space.
|
||||
// - If the mesh is simplicial and partial assembly is requested,
|
||||
// we use the positive basis, which supports device execution.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
auto basis_type = (pa && mesh.IsSimplexMesh()) ?
|
||||
BasisType::Positive : BasisType::GaussLobatto;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
fec = new H1_FECollection(order, dim, basis_type);
|
||||
}
|
||||
else if (mesh.GetNodes())
|
||||
{
|
||||
fec = mesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
fec = new H1_FECollection(order = 1, dim, basis_type);
|
||||
}
|
||||
FiniteElementSpace fespace(&mesh, fec);
|
||||
cout << "Number of finite element unknowns: "
|
||||
@@ -224,17 +228,29 @@ int main(int argc, char *argv[])
|
||||
// 11. Solve the linear system A X = B.
|
||||
if (!pa)
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
|
||||
#else
|
||||
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
if (Device::Allows(Backend::CUDA_MASK))
|
||||
{
|
||||
// Use cuDSS to solve the system.
|
||||
CuDSSSolver cudss_solver;
|
||||
cudss_solver.SetOperator(*A);
|
||||
cudss_solver.Mult(B, X);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
|
||||
#else
|
||||
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -273,17 +289,14 @@ int main(int argc, char *argv[])
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << x << flush;
|
||||
}
|
||||
|
||||
// 15. Free the used memory.
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
if (order > 0) { delete fec; }
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
+61
-35
@@ -42,7 +42,11 @@
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/square-mixed.mesh
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/fichera-mixed.mesh
|
||||
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cpu -m ../data/beam-tet.mesh
|
||||
//
|
||||
// Device simplices sample runs:
|
||||
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tet.mesh
|
||||
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tri.mesh
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
@@ -83,6 +87,9 @@ int main(int argc, char *argv[])
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = true;
|
||||
bool algebraic_ceed = false;
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
bool cudss_solver = false;
|
||||
#endif
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -102,6 +109,10 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&algebraic_ceed, "-a", "--algebraic",
|
||||
"-no-a", "--no-algebraic",
|
||||
"Use algebraic Ceed solver");
|
||||
#endif
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
args.AddOption(&cudss_solver, "-cudss", "--cudss-solver", "-no-cudss",
|
||||
"--no-cudss-solver", "Use the cuDSS Solver.");
|
||||
#endif
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
@@ -158,19 +169,20 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 7. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// use continuous Lagrange finite elements of the specified order. If
|
||||
// order < 1, we instead use an isoparametric/isogeometric space.
|
||||
// use continuous Lagrange finite elements of the specified order.
|
||||
// - If order < 1, we instead use an isoparametric/isogeometric space.
|
||||
// - If the mesh is simplicial and partial assembly is requested,
|
||||
// we use the positive basis, which supports device execution.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
auto basis_type = (pa && pmesh.IsSimplexMesh()) ?
|
||||
BasisType::Positive : BasisType::GaussLobatto;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
fec = new H1_FECollection(order, dim, basis_type);
|
||||
}
|
||||
else if (pmesh.GetNodes())
|
||||
{
|
||||
fec = pmesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
@@ -178,8 +190,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
fec = new H1_FECollection(order = 1, dim, basis_type);
|
||||
}
|
||||
ParFiniteElementSpace fespace(&pmesh, fec);
|
||||
HYPRE_BigInt size = fespace.GlobalTrueVSize();
|
||||
@@ -248,33 +259,51 @@ int main(int argc, char *argv[])
|
||||
// 13. Solve the linear system A X = B.
|
||||
// * With full assembly, use the BoomerAMG preconditioner from hypre.
|
||||
// * With partial assembly, use Jacobi smoothing, for now.
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
if (!pa && (Device::Allows(Backend::CUDA_MASK) && cudss_solver))
|
||||
{
|
||||
if (UsesTensorBasis(fespace))
|
||||
{
|
||||
if (algebraic_ceed)
|
||||
{
|
||||
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
}
|
||||
}
|
||||
// Solve using a direct solver with cuDSS
|
||||
CuDSSSolver cudss_solver(MPI_COMM_WORLD);
|
||||
cudss_solver.SetMatrixSymType(
|
||||
CuDSSSolver::SYMMETRIC_POSITIVE_DEFINITE);
|
||||
cudss_solver.SetMatrixViewType(CuDSSSolver::UPPER);
|
||||
cudss_solver.SetOperator(*A);
|
||||
cudss_solver.Mult(B, X);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
prec = new HypreBoomerAMG;
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
{
|
||||
if (UsesTensorBasis(fespace))
|
||||
{
|
||||
if (algebraic_ceed)
|
||||
{
|
||||
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new HypreBoomerAMG;
|
||||
}
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
if (prec)
|
||||
{
|
||||
cg.SetPreconditioner(*prec);
|
||||
}
|
||||
cg.SetOperator(*A);
|
||||
cg.Mult(B, X);
|
||||
delete prec;
|
||||
}
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
if (prec) { cg.SetPreconditioner(*prec); }
|
||||
cg.SetOperator(*A);
|
||||
cg.Mult(B, X);
|
||||
delete prec;
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
@@ -308,10 +337,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
if (order > 0) { delete fec; }
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -95,6 +95,15 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
if (amg_elast && !static_cond && reorder_space)
|
||||
{
|
||||
if (myid == 0)
|
||||
cerr << "\nThe AMG elasticity solver requires ordering byVDIM! "
|
||||
<< "Ignoring the specified option -nodes/--by-nodes.\n"
|
||||
<< endl;
|
||||
reorder_space = false;
|
||||
}
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
|
||||
@@ -76,4 +76,4 @@ clean-build:
|
||||
rm -rf *.dSYM *.TVD.*breakpoints
|
||||
|
||||
clean-exec:
|
||||
@rm -f refined.mesh sol.gf
|
||||
@rm -f refined.mesh sol.gf mesh.* sol.*
|
||||
|
||||
@@ -1,536 +0,0 @@
|
||||
#include <mfem.hpp>
|
||||
|
||||
#include "../fem/dfem/util.hpp"
|
||||
|
||||
#include <proteus/CppJitModule.h>
|
||||
|
||||
#include "jitplayground.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <cctype>
|
||||
#include <cmath>
|
||||
#include <fstream>
|
||||
#include <initializer_list>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <type_traits>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace util
|
||||
{
|
||||
constexpr std::string_view Dirname(std::string_view path)
|
||||
{
|
||||
const size_t last_sep = path.find_last_of("/\\");
|
||||
if (last_sep == std::string_view::npos) { return {}; }
|
||||
return path.substr(0, last_sep);
|
||||
}
|
||||
|
||||
constexpr std::string_view thisFileDir = Dirname(__FILE__);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static std::string TypeNameString()
|
||||
{
|
||||
return std::string(mfem::future::get_type_name<T>());
|
||||
}
|
||||
|
||||
template <typename Tuple, size_t... Is>
|
||||
static auto ParamTypeStringsImpl(std::index_sequence<Is...>)
|
||||
{
|
||||
return std::array<std::string, sizeof...(Is)>
|
||||
{
|
||||
TypeNameString<std::remove_reference_t<decltype(mfem::future::get<Is>(std::declval<Tuple&>()))>>()...
|
||||
};
|
||||
}
|
||||
|
||||
template <typename Tuple>
|
||||
static auto ParamTypeStrings()
|
||||
{
|
||||
return ParamTypeStringsImpl<Tuple>(
|
||||
std::make_index_sequence<mfem::future::tuple_size<Tuple>::value> {});
|
||||
}
|
||||
|
||||
static std::string_view Trim(std::string_view s)
|
||||
{
|
||||
size_t begin = 0;
|
||||
while (begin < s.size() && std::isspace(static_cast<unsigned char>(s[begin])))
|
||||
{
|
||||
++begin;
|
||||
}
|
||||
size_t end = s.size();
|
||||
while (end > begin &&
|
||||
std::isspace(static_cast<unsigned char>(s[end - 1])))
|
||||
{
|
||||
--end;
|
||||
}
|
||||
return s.substr(begin, end - begin);
|
||||
}
|
||||
|
||||
static bool IsValidIdentifier(std::string_view s)
|
||||
{
|
||||
if (s.empty()) { return false; }
|
||||
const unsigned char c0 = static_cast<unsigned char>(s[0]);
|
||||
if (!(std::isalpha(c0) || c0 == '_')) { return false; }
|
||||
for (size_t i = 1; i < s.size(); ++i)
|
||||
{
|
||||
const unsigned char c = static_cast<unsigned char>(s[i]);
|
||||
if (!(std::isalnum(c) || c == '_')) { return false; }
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool ParseJitDirective(std::string_view line,
|
||||
std::string &type,
|
||||
std::string &var,
|
||||
std::string &kind)
|
||||
{
|
||||
const size_t jit_pos = line.find("$JIT");
|
||||
if (jit_pos == std::string_view::npos) { return false; }
|
||||
|
||||
const size_t open = line.find('[', jit_pos);
|
||||
const size_t close = line.find(']', jit_pos);
|
||||
MFEM_VERIFY(open != std::string_view::npos &&
|
||||
close != std::string_view::npos &&
|
||||
close > open,
|
||||
"malformed $JIT directive (expected brackets): " << line);
|
||||
|
||||
const std::string_view payload = line.substr(open + 1, close - open - 1);
|
||||
const size_t comma1 = payload.find(',');
|
||||
const size_t comma2 = (comma1 == std::string_view::npos)
|
||||
? std::string_view::npos
|
||||
: payload.find(',', comma1 + 1);
|
||||
MFEM_VERIFY(comma1 != std::string_view::npos &&
|
||||
comma2 != std::string_view::npos,
|
||||
"malformed $JIT directive (expected 3 comma-separated fields): "
|
||||
<< line);
|
||||
|
||||
const std::string_view f0 = Trim(payload.substr(0, comma1));
|
||||
const std::string_view f1 = Trim(payload.substr(comma1 + 1,
|
||||
comma2 - comma1 - 1));
|
||||
const std::string_view f2 = Trim(payload.substr(comma2 + 1));
|
||||
MFEM_VERIFY(!f0.empty() && !f1.empty() && !f2.empty(),
|
||||
"malformed $JIT directive (empty field): " << line);
|
||||
|
||||
type.assign(f0);
|
||||
var.assign(f1);
|
||||
kind.assign(f2);
|
||||
return true;
|
||||
}
|
||||
|
||||
static std::string ReadFileOrEmpty(const std::string &fn)
|
||||
{
|
||||
std::ifstream file(fn);
|
||||
if (!file.is_open())
|
||||
{
|
||||
std::cerr << "could not open file " << fn << "\n";
|
||||
return {};
|
||||
}
|
||||
std::stringstream buffer;
|
||||
buffer << file.rdbuf();
|
||||
return buffer.str();
|
||||
}
|
||||
|
||||
static std::vector<std::string> ExtractJitVarNames(const std::string
|
||||
&kernel_code)
|
||||
{
|
||||
std::stringstream ss(kernel_code);
|
||||
std::string line;
|
||||
std::vector<std::string> var_names;
|
||||
std::unordered_set<std::string> seen_vars;
|
||||
|
||||
while (std::getline(ss, line))
|
||||
{
|
||||
std::string type, var, kind;
|
||||
if (ParseJitDirective(line, type, var, kind))
|
||||
{
|
||||
MFEM_VERIFY(IsValidIdentifier(var),
|
||||
"$JIT variable must be a valid identifier: " << var);
|
||||
MFEM_VERIFY(seen_vars.insert(var).second,
|
||||
"duplicate $JIT variable name: " << var);
|
||||
var_names.push_back(var);
|
||||
}
|
||||
}
|
||||
return var_names;
|
||||
}
|
||||
|
||||
static std::string RewriteKernelForJit(std::string kernel_code,
|
||||
const std::vector<std::string> &jit_values)
|
||||
{
|
||||
std::stringstream ss(kernel_code);
|
||||
std::string line;
|
||||
|
||||
std::string out;
|
||||
out.reserve(kernel_code.size() + 128);
|
||||
|
||||
bool have_pending = false;
|
||||
size_t pending_index = 0;
|
||||
std::string pending_type;
|
||||
std::string pending_var;
|
||||
std::unordered_set<std::string> seen_vars;
|
||||
|
||||
while (std::getline(ss, line))
|
||||
{
|
||||
line.push_back('\n');
|
||||
|
||||
if (have_pending)
|
||||
{
|
||||
MFEM_VERIFY(pending_index < jit_values.size(),
|
||||
"not enough JIT values provided");
|
||||
const size_t indent_end = line.find_first_not_of(" \t");
|
||||
const std::string indent =
|
||||
(indent_end == std::string::npos) ? std::string() :
|
||||
line.substr(0, indent_end);
|
||||
out += indent + "const " + pending_type + " " + pending_var + " = " +
|
||||
jit_values[pending_index] + ";\n";
|
||||
have_pending = false;
|
||||
++pending_index;
|
||||
continue;
|
||||
}
|
||||
|
||||
std::string type, var, kind;
|
||||
if (ParseJitDirective(line, type, var, kind))
|
||||
{
|
||||
MFEM_VERIFY(IsValidIdentifier(var),
|
||||
"$JIT variable must be a valid identifier: " << var);
|
||||
MFEM_VERIFY(kind == "generic",
|
||||
"unsupported $JIT kind: " << kind);
|
||||
MFEM_VERIFY(seen_vars.insert(var).second,
|
||||
"duplicate $JIT variable name: " << var);
|
||||
|
||||
pending_type = std::move(type);
|
||||
pending_var = std::move(var);
|
||||
have_pending = true;
|
||||
continue; // drop directive line
|
||||
}
|
||||
|
||||
out += line;
|
||||
}
|
||||
|
||||
MFEM_VERIFY(!have_pending,
|
||||
"$JIT directive must annotate a following line");
|
||||
MFEM_VERIFY(jit_values.size() == pending_index,
|
||||
"JIT value count must match number of $JIT directives");
|
||||
return out;
|
||||
}
|
||||
|
||||
static std::string GeneratedOutputPath(std::string_view original_path)
|
||||
{
|
||||
const size_t last_sep = original_path.find_last_of("/\\");
|
||||
const size_t dot = original_path.find_last_of('.');
|
||||
const bool dot_in_filename =
|
||||
(dot != std::string_view::npos) &&
|
||||
(last_sep == std::string_view::npos || dot > last_sep);
|
||||
|
||||
const std::string_view base =
|
||||
dot_in_filename ? original_path.substr(0, dot) : original_path;
|
||||
return std::string(base) + "_generated.hpp";
|
||||
}
|
||||
|
||||
static void WriteFileOrWarn(const std::string &path,
|
||||
const std::string &contents)
|
||||
{
|
||||
std::ofstream out(path);
|
||||
if (!out.is_open())
|
||||
{
|
||||
std::cerr << "could not write generated file " << path << "\n";
|
||||
return;
|
||||
}
|
||||
out << contents;
|
||||
}
|
||||
|
||||
class JitQFunction
|
||||
{
|
||||
public:
|
||||
template <typename ImplT, size_t N>
|
||||
JitQFunction(ImplT, const std::string &fn,
|
||||
const std::array<bool, N> &activity_map)
|
||||
{
|
||||
using qf_signature = typename
|
||||
mfem::future::get_function_signature<
|
||||
decltype(&ImplT::operator())>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
constexpr size_t nparams = mfem::future::tuple_size<qf_param_ts>::value;
|
||||
static_assert(N == nparams, "activity_map size must match qfunc arity");
|
||||
|
||||
this->fn = fn;
|
||||
this->nparams = nparams;
|
||||
this->activity_map.reserve(N);
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
this->activity_map.push_back(activity_map[i]);
|
||||
}
|
||||
{
|
||||
const auto param_types_arr = ParamTypeStrings<qf_param_ts>();
|
||||
this->param_types.assign(param_types_arr.begin(), param_types_arr.end());
|
||||
}
|
||||
this->return_type = TypeNameString<typename qf_signature::return_t>();
|
||||
this->return_is_void = std::is_same_v<typename qf_signature::return_t, void>;
|
||||
this->impl_type_name = TypeNameString<ImplT>();
|
||||
this->jit_var_names = ExtractJitVarNames(ReadFileOrEmpty(fn));
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run(std::string_view name,
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
auto ordered_values = MatchJitValues(jit_values);
|
||||
auto &mod = GetOrCreateModule(ordered_values);
|
||||
auto &instance = mod.instantiate(std::string(name), std::string());
|
||||
return instance.template run<ReturnT>(std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run_primal(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
return run<ReturnT>(qfunc_name, jit_values,
|
||||
std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
template <typename ReturnT, typename... Args>
|
||||
ReturnT run_derivative(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>> jit_values,
|
||||
Args&&... args)
|
||||
{
|
||||
return run<ReturnT>(qfunc_name + "_fwddiff", jit_values,
|
||||
std::forward<Args>(args)...);
|
||||
}
|
||||
|
||||
private:
|
||||
std::vector<std::string_view> MatchJitValues(
|
||||
std::initializer_list<std::pair<std::string_view, std::string_view>>
|
||||
named_values) const
|
||||
{
|
||||
std::unordered_map<std::string_view, std::string_view> value_map;
|
||||
for (const auto &[name, value] : named_values)
|
||||
{
|
||||
value_map[name] = value;
|
||||
}
|
||||
|
||||
std::vector<std::string_view> ordered_values;
|
||||
ordered_values.reserve(jit_var_names.size());
|
||||
for (const auto &var_name : jit_var_names)
|
||||
{
|
||||
auto it = value_map.find(var_name);
|
||||
MFEM_VERIFY(it != value_map.end(),
|
||||
"missing JIT value for variable: " << var_name);
|
||||
ordered_values.push_back(it->second);
|
||||
}
|
||||
|
||||
MFEM_VERIFY(ordered_values.size() == named_values.size(),
|
||||
"provided " << named_values.size() << " JIT values but expected "
|
||||
<< jit_var_names.size());
|
||||
return ordered_values;
|
||||
}
|
||||
|
||||
|
||||
std::string BuildModuleCode(const std::vector<std::string> &jit_values) const
|
||||
{
|
||||
std::string module_code =
|
||||
RewriteKernelForJit(ReadFileOrEmpty(fn), jit_values);
|
||||
module_code += "\n\n";
|
||||
module_code += "// --- generated ---\n";
|
||||
module_code +=
|
||||
"template <typename return_type, typename... Args>\n"
|
||||
"return_type __enzyme_fwddiff(Args...);\n"
|
||||
"\n"
|
||||
"extern int enzyme_const;\n"
|
||||
"extern int enzyme_dup;\n"
|
||||
"\n";
|
||||
|
||||
// Generate a primal wrapper with the requested symbol name, so the kernel
|
||||
// header can just define the qfunc as a functor.
|
||||
//
|
||||
// Note: Proteus instantiates entrypoints via `qfunc_wrapper<>(...)` even
|
||||
// when there are no user template args, so keep the wrapper itself a
|
||||
// template (with a default parameter) while still doing literal `$JIT`
|
||||
// replacements in the kernel code.
|
||||
module_code += "template <typename = void>\n";
|
||||
module_code += return_type + " " +
|
||||
std::string(qfunc_name) + "(";
|
||||
bool first = true;
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (!first) { module_code += ", "; }
|
||||
first = false;
|
||||
module_code += param_types[i] + " Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += ")\n";
|
||||
module_code += "{\n";
|
||||
module_code += " " + impl_type_name + " qf;\n";
|
||||
if (return_is_void)
|
||||
{
|
||||
module_code += " ";
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " return ";
|
||||
}
|
||||
module_code += "qf(";
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (i) { module_code += ", "; }
|
||||
module_code += "Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += ");\n";
|
||||
module_code += "}\n\n";
|
||||
|
||||
module_code += "template <typename = void>\n";
|
||||
module_code += return_type + " " +
|
||||
std::string(qfunc_name) + "_fwddiff(";
|
||||
|
||||
first = true;
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (!first) { module_code += ", "; }
|
||||
first = false;
|
||||
module_code += param_types[i] + " Arg" + std::to_string(i);
|
||||
if (activity_map[i])
|
||||
{
|
||||
module_code += ", " + param_types[i] + " dArg" + std::to_string(i);
|
||||
}
|
||||
}
|
||||
module_code += ")\n";
|
||||
module_code += "{\n";
|
||||
if (return_is_void)
|
||||
{
|
||||
module_code += " __enzyme_fwddiff<void>(\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " return __enzyme_fwddiff<" +
|
||||
return_type + ">(\n";
|
||||
}
|
||||
module_code += " (void*)" + std::string(qfunc_name) + "<>";
|
||||
module_code += ",\n";
|
||||
for (size_t i = 0; i < nparams; ++i)
|
||||
{
|
||||
if (activity_map[i])
|
||||
{
|
||||
module_code += " enzyme_dup, Arg" + std::to_string(i) +
|
||||
", dArg" + std::to_string(i);
|
||||
}
|
||||
else
|
||||
{
|
||||
module_code += " enzyme_const, Arg" + std::to_string(i);
|
||||
}
|
||||
module_code += (i + 1 == nparams) ? ");\n" : ",\n";
|
||||
}
|
||||
module_code += "}\n";
|
||||
|
||||
WriteFileOrWarn(GeneratedOutputPath(fn), module_code);
|
||||
return module_code;
|
||||
}
|
||||
|
||||
proteus::CppJitModule &GetOrCreateModule(
|
||||
const std::vector<std::string_view> &jit_values)
|
||||
{
|
||||
std::string key;
|
||||
for (const auto &val : jit_values)
|
||||
{
|
||||
if (!key.empty()) { key += ","; }
|
||||
key += val;
|
||||
}
|
||||
|
||||
auto it = modules.find(key);
|
||||
if (it != modules.end())
|
||||
{
|
||||
return *it->second;
|
||||
}
|
||||
|
||||
std::vector<std::string> values(jit_values.begin(), jit_values.end());
|
||||
std::string code = BuildModuleCode(values);
|
||||
auto mod = std::make_unique<proteus::CppJitModule>("host", code,
|
||||
DefaultExtraArgs());
|
||||
auto [inserted, ok] = modules.emplace(key, std::move(mod));
|
||||
MFEM_VERIFY(ok, "failed to cache JIT module");
|
||||
return *inserted->second;
|
||||
}
|
||||
|
||||
static std::vector<std::string> DefaultExtraArgs()
|
||||
{
|
||||
return {"-fplugin=/Users/andrej1/local/enzyme/lib/ClangEnzyme-20.dylib"};
|
||||
}
|
||||
|
||||
std::string qfunc_name = "qfunc_wrapper";
|
||||
std::string fn;
|
||||
size_t nparams = 0;
|
||||
std::vector<bool> activity_map;
|
||||
std::vector<std::string> param_types;
|
||||
std::string return_type;
|
||||
bool return_is_void = false;
|
||||
std::string impl_type_name;
|
||||
std::vector<std::string> jit_var_names;
|
||||
std::unordered_map<std::string, std::unique_ptr<proteus::CppJitModule>> modules;
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
const size_t N = 4;
|
||||
const size_t M = 5;
|
||||
const double A = 123.4;
|
||||
|
||||
std::vector<double> X(N);
|
||||
std::vector<double> Y(N);
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
X[i] = static_cast<double>(i + 1);
|
||||
Y[i] = static_cast<double>(N - i);
|
||||
}
|
||||
|
||||
// // >>> user interface calls
|
||||
// const std::string kernel_path = std::string(util::thisFileDir) +
|
||||
// "/jitplayground.hpp";
|
||||
// JitQFunction qf(daxpy_op{}, kernel_path, std::array{false, true, false});
|
||||
// // <<< user interface calls
|
||||
|
||||
// // this will happen internally in dFEM
|
||||
|
||||
daxpy_op op;
|
||||
printf("\n\nfunction call\n");
|
||||
op(&A, X.data(), Y.data(), &N);
|
||||
|
||||
// reset X for the derivative test
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
X[i] = static_cast<double>(i + 1);
|
||||
Y[i] = static_cast<double>(N - i);
|
||||
}
|
||||
|
||||
std::vector<double> dX(N, 1.0);
|
||||
printf("\n\nforward diff call\n");
|
||||
daxpy_op_fwddiff(&A, X.data(), dX.data(), Y.data(), &N);
|
||||
|
||||
std::vector<double> dX_manual(N, A);
|
||||
|
||||
printf("\n\nderivative checks\n");
|
||||
std::cout << "dX: ";
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
std::cout << dX[i] << (i + 1 == N ? '\n' : ' ');
|
||||
}
|
||||
|
||||
std::cout << "dX_manual: ";
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
std::cout << dX_manual[i] << (i + 1 == N ? '\n' : ' ');
|
||||
}
|
||||
|
||||
double max_abs_err = 0.0;
|
||||
for (size_t i = 0; i < N; ++i)
|
||||
{
|
||||
max_abs_err = std::max(max_abs_err, std::abs(dX[i] - dX_manual[i]));
|
||||
}
|
||||
std::cout << "max |dX - dX_manual| = " << max_abs_err << "\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -1,58 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstddef>
|
||||
#include <vector>
|
||||
#include <type_traits>
|
||||
|
||||
#include "proteus/JitInterface.h"
|
||||
|
||||
struct daxpy_op
|
||||
{
|
||||
void operator()(
|
||||
const double *a,
|
||||
double *x,
|
||||
const double *y,
|
||||
const size_t *N) const
|
||||
{
|
||||
const size_t n = *N;
|
||||
auto lam = [=, n = proteus::jit_variable(n)]
|
||||
() __attribute__((annotate("jit")))
|
||||
{
|
||||
printf("N = %zu\n", n);
|
||||
for (size_t i = 0; i < n; ++i)
|
||||
{
|
||||
printf("x[%zu] = %f, y[%zu] = %f\n", i, x[i], i, y[i]);
|
||||
x[i] = *a * x[i] + y[i];
|
||||
printf("updated x[%zu] = %f\n", i, x[i]);
|
||||
}
|
||||
};
|
||||
|
||||
proteus::register_lambda(lam);
|
||||
|
||||
lam();
|
||||
}
|
||||
};
|
||||
|
||||
template <typename return_type, typename... Args>
|
||||
return_type __enzyme_fwddiff(Args...);
|
||||
|
||||
extern int enzyme_const;
|
||||
extern int enzyme_dup;
|
||||
|
||||
void daxpy_op_wrapper(const double * Arg0, double * Arg1,
|
||||
const double * Arg2, const size_t *Arg3)
|
||||
{
|
||||
daxpy_op qf;
|
||||
qf(Arg0, Arg1, Arg2, Arg3);
|
||||
}
|
||||
|
||||
void daxpy_op_fwddiff(const double * Arg0, double * Arg1,
|
||||
double * dArg1, const double * Arg2, const size_t *Arg3)
|
||||
{
|
||||
__enzyme_fwddiff<void>(
|
||||
(void*)daxpy_op_wrapper,
|
||||
enzyme_const, Arg0,
|
||||
enzyme_dup, Arg1, dArg1,
|
||||
enzyme_const, Arg2,
|
||||
enzyme_const, Arg3);
|
||||
}
|
||||
+5
-2
@@ -71,6 +71,7 @@ endif
|
||||
|
||||
SUBDIRS_ALL = $(addsuffix /all,$(SUBDIRS))
|
||||
SUBDIRS_TEST = $(addsuffix /test,$(SUBDIRS))
|
||||
SUBDIRS_TEST_NOCLEAN = $(addsuffix /test-noclean,$(SUBDIRS))
|
||||
SUBDIRS_CLEAN = $(addsuffix /clean,$(SUBDIRS))
|
||||
SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
|
||||
|
||||
@@ -87,8 +88,9 @@ SUBDIRS_TPRINT = $(addsuffix /test-print,$(SUBDIRS))
|
||||
|
||||
all: $(EXAMPLES) $(SUBDIRS_ALL)
|
||||
|
||||
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
|
||||
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_CLEAN):
|
||||
.PHONY: $(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) \
|
||||
$(SUBDIRS_CLEAN) $(SUBDIRS_TPRINT)
|
||||
$(SUBDIRS_ALL) $(SUBDIRS_TEST) $(SUBDIRS_TEST_NOCLEAN) $(SUBDIRS_CLEAN):
|
||||
$(MAKE) -C $(@D) $(@F)
|
||||
$(SUBDIRS_TPRINT):
|
||||
@$(MAKE) -C $(@D) $(@F)
|
||||
@@ -107,6 +109,7 @@ endif
|
||||
MFEM_TESTS = EXAMPLES
|
||||
include $(MFEM_TEST_MK)
|
||||
test: $(SUBDIRS_TEST)
|
||||
test-noclean: $(SUBDIRS_TEST_NOCLEAN)
|
||||
test-print: $(SUBDIRS_TPRINT)
|
||||
|
||||
# Testing: Parallel vs. serial runs
|
||||
|
||||
+19
-20
@@ -121,11 +121,6 @@ set(SRCS
|
||||
qinterp/eval_hdiv.cpp
|
||||
qinterp/grad_by_nodes.cpp
|
||||
qinterp/grad_by_vdim.cpp
|
||||
qinterp/grad_transpose.cpp
|
||||
qinterp/grad_transpose_by_nodes.cpp
|
||||
qinterp/grad_transpose_by_vdim.cpp
|
||||
qinterp/eval_transpose.cpp
|
||||
qinterp/eval_transpose_by_vdim.cpp
|
||||
qspace.cpp
|
||||
quadinterpolator.cpp
|
||||
quadinterpolator_face.cpp
|
||||
@@ -138,7 +133,7 @@ set(SRCS
|
||||
tmop/assemble/diag2.cpp
|
||||
tmop/assemble/grad2_limit.cpp
|
||||
tmop/assemble/grad2.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3.cpp
|
||||
tmop/assemble/grad3_limit.cpp
|
||||
tmop/assemble/grad3.cpp
|
||||
@@ -176,8 +171,12 @@ set(SRCS
|
||||
tmop_tools.cpp
|
||||
tmop_amr.cpp
|
||||
gslib.cpp
|
||||
gslib/findptsedge_local_2.cpp
|
||||
gslib/findptsedge_local_3.cpp
|
||||
gslib/findptssurf_local_3.cpp
|
||||
gslib/findpts_local_2.cpp
|
||||
gslib/findpts_local_3.cpp
|
||||
gslib/interpolate_local_1.cpp
|
||||
gslib/interpolate_local_2.cpp
|
||||
gslib/interpolate_local_3.cpp
|
||||
transfer.cpp
|
||||
@@ -196,12 +195,14 @@ set(HDRS
|
||||
integ/bilininteg_dgtrace_kernels.hpp
|
||||
integ/bilininteg_vecdiffusion_kernels.hpp
|
||||
integ/bilininteg_convection_kernels.hpp
|
||||
integ/bilininteg_diffusion_pa_simplices.hpp
|
||||
integ/bilininteg_diffusion_kernels.hpp
|
||||
integ/bilininteg_elasticity_kernels.hpp
|
||||
integ/bilininteg_hcurl_kernels.hpp
|
||||
integ/bilininteg_hdiv_kernels.hpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.hpp
|
||||
integ/bilininteg_mass_kernels.hpp
|
||||
integ/bilininteg_mass_pa_simplices.hpp
|
||||
integ/bilininteg_vecdiffusion_pa.hpp
|
||||
integ/bilininteg_vecmass_pa.hpp
|
||||
coefficient.hpp
|
||||
@@ -283,10 +284,8 @@ set(HDRS
|
||||
qfunction.hpp
|
||||
qinterp/det.hpp
|
||||
qinterp/eval.hpp
|
||||
qinterp/eval_transpose.hpp
|
||||
qinterp/eval_hdiv.hpp
|
||||
qinterp/grad.hpp
|
||||
qinterp/grad_transpose.hpp
|
||||
qspace.hpp
|
||||
quadinterpolator.hpp
|
||||
quadinterpolator_face.hpp
|
||||
@@ -320,36 +319,36 @@ set(HDRS
|
||||
)
|
||||
|
||||
if (MFEM_USE_SIDRE)
|
||||
list(APPEND SRCS sidredatacollection.cpp)
|
||||
list(APPEND HDRS sidredatacollection.hpp)
|
||||
list(APPEND SRCS sidredatacollection.cpp)
|
||||
list(APPEND HDRS sidredatacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CONDUIT)
|
||||
list(APPEND SRCS conduitdatacollection.cpp)
|
||||
list(APPEND HDRS conduitdatacollection.hpp)
|
||||
list(APPEND SRCS conduitdatacollection.cpp)
|
||||
list(APPEND HDRS conduitdatacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_ADIOS2)
|
||||
list(APPEND SRCS adios2datacollection.cpp)
|
||||
list(APPEND HDRS adios2datacollection.hpp)
|
||||
list(APPEND SRCS adios2datacollection.cpp)
|
||||
list(APPEND HDRS adios2datacollection.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_FMS)
|
||||
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
|
||||
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
|
||||
list(APPEND SRCS fmsdatacollection.cpp fmsconvert.cpp)
|
||||
list(APPEND HDRS fmsdatacollection.hpp fmsconvert.hpp)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
list(APPEND SRCS
|
||||
list(APPEND SRCS
|
||||
pbilinearform.cpp
|
||||
pfespace.cpp
|
||||
pgridfunc.cpp
|
||||
plinearform.cpp
|
||||
pnonlinearform.cpp
|
||||
prestriction.cpp)
|
||||
# If this list (HDRS -> HEADERS) is used for install, we probably want the
|
||||
# headers added all the time.
|
||||
list(APPEND HDRS
|
||||
# If this list (HDRS -> HEADERS) is used for install, we probably want the
|
||||
# headers added all the time.
|
||||
list(APPEND HDRS
|
||||
pbilinearform.hpp
|
||||
pfespace.hpp
|
||||
pgridfunc.hpp
|
||||
|
||||
+22
-4
@@ -1345,7 +1345,8 @@ real_t DiffusionIntegrator::ComputeFluxEnergy
|
||||
}
|
||||
|
||||
const IntegrationRule &DiffusionIntegrator::GetRule(
|
||||
const FiniteElement &trial_fe, const FiniteElement &test_fe)
|
||||
const FiniteElement &trial_fe, const FiniteElement &test_fe,
|
||||
const bool stroud)
|
||||
{
|
||||
int order;
|
||||
if (trial_fe.Space() == FunctionSpace::Pk)
|
||||
@@ -1362,7 +1363,15 @@ const IntegrationRule &DiffusionIntegrator::GetRule(
|
||||
{
|
||||
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
|
||||
if (stroud)
|
||||
{
|
||||
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
else
|
||||
{
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
}
|
||||
|
||||
MassIntegrator::MassIntegrator(const IntegrationRule *ir)
|
||||
@@ -1449,7 +1458,8 @@ void MassIntegrator::AssembleElementMatrix2(
|
||||
|
||||
const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe,
|
||||
const ElementTransformation &Trans)
|
||||
const ElementTransformation &Trans,
|
||||
const bool stroud)
|
||||
{
|
||||
// int order = trial_fe.GetOrder() + test_fe.GetOrder();
|
||||
const int order = trial_fe.GetOrder() + test_fe.GetOrder() + Trans.OrderW();
|
||||
@@ -1458,7 +1468,15 @@ const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
|
||||
{
|
||||
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
|
||||
if (stroud)
|
||||
{
|
||||
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
else
|
||||
{
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
+52
-19
@@ -2178,22 +2178,29 @@ class DiffusionIntegrator: public BilinearFormIntegrator
|
||||
{
|
||||
public:
|
||||
|
||||
using DiffusionApplyKernelType = void(*)(const int, const bool,
|
||||
const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&,
|
||||
const Vector&, const Vector&,
|
||||
Vector&, const int, const int);
|
||||
using ApplyKernelType = void(*)(const int, const bool, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&,
|
||||
const Vector&, const Vector&,
|
||||
Vector&, const int, const int);
|
||||
|
||||
using DiffusionDiagonalKernelType = void(*)(const int, const bool,
|
||||
const Array<real_t>&,
|
||||
const Array<real_t>&, const Vector&, Vector&,
|
||||
const int, const int);
|
||||
using ApplySimplexKernelType = void(*)(const int, const bool, const Array<int>&,
|
||||
const Array<int>&,
|
||||
const Array<int>&, const Array<int>&, const Array<int>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Vector&, const Vector&,
|
||||
Vector&, const int, const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(DiffusionApplyPAKernel, DiffusionApplyKernelType,
|
||||
(int, int, int));
|
||||
MFEM_REGISTER_KERNELS(DiffusionDiagonalPAKernel, DiffusionDiagonalKernelType,
|
||||
(int, int, int));
|
||||
using DiagonalKernelType = void(*)(const int, const bool, const Array<real_t>&,
|
||||
const Array<real_t>&, const Vector&, Vector&,
|
||||
const int, const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
|
||||
int));
|
||||
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
protected:
|
||||
@@ -2213,7 +2220,6 @@ private:
|
||||
const FiniteElementSpace *fespace;
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
public:
|
||||
int dim, ne, dofs1D, quad1D;
|
||||
Vector pa_data;
|
||||
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
|
||||
@@ -2346,7 +2352,8 @@ public:
|
||||
void AddMultPatchPA(const int patch, const Vector &x, Vector &y) const;
|
||||
|
||||
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe);
|
||||
const FiniteElement &test_fe,
|
||||
const bool stroud = false);
|
||||
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
@@ -2355,8 +2362,15 @@ public:
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
static void AddSpecialization()
|
||||
{
|
||||
DiffusionApplyPAKernel::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiffusionDiagonalPAKernel::Specialization<DIM,D1D,Q1D>::Add();
|
||||
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
AddSimplexSpecialization<DIM,D1D,Q1D>();
|
||||
}
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
static void AddSimplexSpecialization()
|
||||
{
|
||||
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
}
|
||||
protected:
|
||||
const IntegrationRule* GetDefaultIntegrationRule(
|
||||
@@ -2393,11 +2407,22 @@ public:
|
||||
const Array<real_t>&, const Vector&,
|
||||
const Vector&, Vector&, const int, const int);
|
||||
|
||||
using ApplySimplexKernelType = void(*)(const int, const Array<int>&,
|
||||
const Array<int>&,
|
||||
const Array<int>&, const Array<int>&, const Array<int>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Array<real_t>&, const Array<real_t>&,
|
||||
const Vector&, const Vector&, Vector&,
|
||||
const int, const int);
|
||||
|
||||
using DiagonalKernelType = void(*)(const int, const Array<real_t>&,
|
||||
const Vector&, Vector&, const int,
|
||||
const int);
|
||||
|
||||
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int));
|
||||
MFEM_REGISTER_KERNELS(ApplySimplexPAKernels, ApplySimplexKernelType, (int, int,
|
||||
int));
|
||||
MFEM_REGISTER_KERNELS(DiagonalPAKernels, DiagonalKernelType, (int, int, int));
|
||||
struct Kernels { Kernels(); };
|
||||
|
||||
@@ -2446,7 +2471,8 @@ public:
|
||||
|
||||
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe,
|
||||
const ElementTransformation &Trans);
|
||||
const ElementTransformation &Trans,
|
||||
const bool stroud = false);
|
||||
|
||||
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
|
||||
|
||||
@@ -2457,6 +2483,13 @@ public:
|
||||
{
|
||||
ApplyPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
DiagonalPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
AddSimplexSpecialization<DIM,D1D,Q1D>();
|
||||
}
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
static void AddSimplexSpecialization()
|
||||
{
|
||||
ApplySimplexPAKernels::Specialization<DIM,D1D,Q1D>::Add();
|
||||
}
|
||||
|
||||
protected:
|
||||
|
||||
+13
-26
@@ -830,15 +830,9 @@ ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
|
||||
int vsize = pfes->GetVSize();
|
||||
Vector::Load(input, 2*vsize);
|
||||
|
||||
real_t *data_ = const_cast<real_t*>(HostRead());
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
real_t *h_data = HostReadWrite();
|
||||
pfes->ApplyDofSigns(h_data);
|
||||
pfes->ApplyDofSigns(h_data + vsize);
|
||||
|
||||
|
||||
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
|
||||
@@ -1051,15 +1045,14 @@ void ParComplexGridFunction::Save(std::ostream &os) const
|
||||
os << '\n';
|
||||
|
||||
int vsize = pfes->GetVSize();
|
||||
real_t *data_ = const_cast<real_t*>(HostRead());
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
// We use const_cast + HostRead (instead of HostReadWrite) because we only
|
||||
// need to change the host data temporarily and this way we do not invalidate
|
||||
// the data if it is on device. If we use HostReadWrite here, later calls to
|
||||
// Read or ReadWrite will need to copy the data from host to device. With the
|
||||
// approach used here, the host-to-device copy is avoided.
|
||||
real_t *h_data = const_cast<real_t*>(HostRead());
|
||||
pfes->ApplyDofSigns(h_data);
|
||||
pfes->ApplyDofSigns(h_data + vsize);
|
||||
|
||||
if (pfes->GetOrdering() == Ordering::byNODES)
|
||||
{
|
||||
@@ -1070,14 +1063,8 @@ void ParComplexGridFunction::Save(std::ostream &os) const
|
||||
Vector::Print(os, pfes->GetVDim());
|
||||
}
|
||||
|
||||
for (int i = 0; i < vsize; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0)
|
||||
{
|
||||
data_[i] = -data_[i];
|
||||
data_[i+vsize] = -data_[i+vsize];
|
||||
}
|
||||
}
|
||||
pfes->ApplyDofSigns(h_data);
|
||||
pfes->ApplyDofSigns(h_data + vsize);
|
||||
|
||||
os.flush();
|
||||
}
|
||||
|
||||
@@ -114,6 +114,10 @@ void ConduitDataCollection::Save()
|
||||
n_mesh["fields"][name]);
|
||||
}
|
||||
|
||||
// TODO: in parallel, we need to call ParFiniteElementSpace::ApplyDofSigns
|
||||
// for all ParGridFunction objects before and after saving, see
|
||||
// ParGridFunction::Save.
|
||||
|
||||
// save mesh data
|
||||
SaveMeshAndFields(myid,
|
||||
n_mesh,
|
||||
|
||||
@@ -1,587 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
|
||||
// #include "fem/kernels.hpp"
|
||||
#include "fem/kernels3d.hpp"
|
||||
namespace ker = mfem::kernels::internal;
|
||||
namespace low = mfem::kernels::internal::low;
|
||||
#include "fem/kernel_dispatch.hpp"
|
||||
|
||||
// #include "linalg/kernels.hpp"
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
#undef NVTX_COLOR
|
||||
#define NVTX_COLOR ::nvtx::kOrchid
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1>` */
|
||||
template<typename T, int n1>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2>` */
|
||||
template<typename T, int n1, int n2>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3>` */
|
||||
template<typename T, int n1, int n2, int n3>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2, n3>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2, int n3>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2, n3>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3>*>(ptr));
|
||||
}
|
||||
|
||||
/** @brief Zero-copy view of a contiguous block as a `tensor<T, n1, n2, n3, n4>` */
|
||||
template<typename T, int n1, int n2, int n3, int n4>
|
||||
MFEM_HOST_DEVICE
|
||||
const tensor<T, n1, n2, n3, n4>& as_tensor(const T* ptr)
|
||||
{
|
||||
// std::launder makes this defined behavior under strict aliasing rules
|
||||
return *std::launder(reinterpret_cast<const tensor<T, n1, n2, n3, n4>*>(ptr));
|
||||
}
|
||||
|
||||
// convenience overload if you prefer a mutable view
|
||||
template<typename T, int n1, int n2, int n3, int n4>
|
||||
MFEM_HOST_DEVICE
|
||||
tensor<T, n1, n2, n3, n4>& as_tensor(T* ptr)
|
||||
{
|
||||
return *std::launder(reinterpret_cast<tensor<T, n1, n2, n3, n4>*>(ptr));
|
||||
}
|
||||
|
||||
|
||||
template <std::size_t N>
|
||||
MFEM_HOST_DEVICE inline
|
||||
std::array<real_t*, N>
|
||||
load_field_e_ptr(const std::array<DeviceTensor<2>, N> &fields_e,
|
||||
const int e)
|
||||
{
|
||||
std::array<real_t*, N> f;
|
||||
for_constexpr<N>([&](auto i) { f[i] = &fields_e[i](0, e); });
|
||||
return f;
|
||||
}
|
||||
|
||||
namespace qf
|
||||
{
|
||||
|
||||
template <int T_Q1D,
|
||||
size_t num_args,
|
||||
typename reg_t,
|
||||
typename qfunc_t,
|
||||
typename args_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel(reg_t &res /*output*/,
|
||||
reg_t ®,
|
||||
const real_t *rd,
|
||||
const int qx, const int qy, const int qz,
|
||||
const qfunc_t &qfunc, args_ts &args)
|
||||
{
|
||||
if constexpr (num_args == 2) // PAApply
|
||||
{
|
||||
// ∇u
|
||||
tensor<real_t, 3> &arg_0 = get<0>(args);
|
||||
arg_0[0] = reg[qz][qy][qx][0];
|
||||
arg_0[1] = reg[qz][qy][qx][1];
|
||||
arg_0[2] = reg[qz][qy][qx][2];
|
||||
|
||||
// D (PA data)
|
||||
tensor<real_t, 3, 3> &arg_1 = get<1>(args);
|
||||
|
||||
if constexpr (T_Q1D > 0)
|
||||
{
|
||||
const auto *D = (const real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
for (int k = 0; k < 3; k++)
|
||||
{
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
arg_1[k][j] = D[qx][qy][qz][k][j];
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(false);
|
||||
// const auto D = Reshape(r2, 3, 3, Q1D, Q1D, Q1D);
|
||||
// for (int j = 0; j < 3; j++)
|
||||
// {
|
||||
// for (int k = 0; k < 3; k++)
|
||||
// {
|
||||
// arg_1[k][j] = D(j, k, qz, qy, qx);
|
||||
// }
|
||||
// }
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// MFApply comes here
|
||||
assert(false);
|
||||
// MFEM_ABORT("Only two arguments (∇u and D) are supported in apply_kernel for now");
|
||||
}
|
||||
|
||||
const auto r = get<0>(apply(qfunc, args));
|
||||
|
||||
if constexpr (decltype(r)::ndim == 1)
|
||||
{
|
||||
// process_qf_result_from_reg(r0, qx, qy, qz, r);
|
||||
as_tensor<real_t, 3>(&res[qz][qy][qx][0]) = r;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(false);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace qf
|
||||
|
||||
#define MFEM_D2Q_MAX_SIZE 4
|
||||
static MFEM_CONSTANT real_t Bi[MFEM_D2Q_MAX_SIZE][8*8], Bo[8*8];
|
||||
static MFEM_CONSTANT real_t Gi[MFEM_D2Q_MAX_SIZE][8*8], Go[8*8];
|
||||
|
||||
template<size_t num_fields,
|
||||
size_t num_inputs,
|
||||
size_t num_outputs,
|
||||
typename restriction_cb_t,
|
||||
typename qfunc_t,
|
||||
typename input_t,
|
||||
typename output_fop_t>
|
||||
class NewActionCallback
|
||||
{
|
||||
restriction_cb_t &restriction_cb;
|
||||
qfunc_t &qfunc;
|
||||
input_t &inputs;
|
||||
const std::array<size_t, num_inputs> &input_to_field;
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps;
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps;
|
||||
const int num_entities;
|
||||
const int test_vdim;
|
||||
const int num_test_dof;
|
||||
const int dimension;
|
||||
const ThreadBlocks &thread_blocks;
|
||||
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info;
|
||||
const Array<int> &attributes;
|
||||
const output_fop_t &output_fop;
|
||||
const Array<int> *elem_attributes;
|
||||
// refs
|
||||
std::vector<Vector> &fields_e;
|
||||
Vector &residual_e;
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose;
|
||||
// args
|
||||
std::vector<Vector> &solutions_l;
|
||||
const std::vector<Vector> ¶meters_l;
|
||||
Vector &residual_l;
|
||||
|
||||
public:
|
||||
NewActionCallback() = delete;
|
||||
|
||||
NewActionCallback(const bool use_kernels_specialization,
|
||||
restriction_cb_t &restriction_cb,
|
||||
qfunc_t &qfunc,
|
||||
input_t &inputs,
|
||||
const std::array<size_t, num_inputs> &input_to_field,
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
const int num_entities,
|
||||
const int test_vdim,
|
||||
const int num_test_dof,
|
||||
const int dimension,
|
||||
const ThreadBlocks &thread_blocks,
|
||||
SharedMemoryInfo<num_fields, num_inputs, num_outputs> &shmem_info,
|
||||
const Array<int> &attributes,
|
||||
const output_fop_t &output_fop,
|
||||
const Array<int> *elem_attributes,
|
||||
// refs
|
||||
std::vector<Vector> &fields_e,
|
||||
Vector &residual_e,
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
|
||||
// args
|
||||
std::vector<Vector> &solutions_l,
|
||||
const std::vector<Vector> ¶meters_l,
|
||||
Vector &residual_l):
|
||||
restriction_cb(restriction_cb),
|
||||
qfunc(qfunc),
|
||||
inputs(inputs),
|
||||
input_to_field(input_to_field),
|
||||
input_dtq_maps(input_dtq_maps),
|
||||
output_dtq_maps(output_dtq_maps),
|
||||
num_entities(num_entities),
|
||||
test_vdim(test_vdim),
|
||||
num_test_dof(num_test_dof),
|
||||
dimension(dimension),
|
||||
thread_blocks(thread_blocks),
|
||||
shmem_info(shmem_info),
|
||||
attributes(attributes),
|
||||
output_fop(output_fop),
|
||||
elem_attributes(elem_attributes),
|
||||
fields_e(fields_e),
|
||||
residual_e(residual_e),
|
||||
output_restriction_transpose(output_restriction_transpose),
|
||||
solutions_l(solutions_l),
|
||||
parameters_l(parameters_l),
|
||||
residual_l(residual_l)
|
||||
{
|
||||
if (!use_kernels_specialization) { return; }
|
||||
NewActionCallbackKernels::template Specialization<3>::Add(); // 1
|
||||
NewActionCallbackKernels::template Specialization<4>::Add(); // 2
|
||||
NewActionCallbackKernels::template Specialization<5>::Add(); // 3
|
||||
NewActionCallbackKernels::template Specialization<6>::Add(); // 4
|
||||
NewActionCallbackKernels::template Specialization<7>::Add(); // 5
|
||||
NewActionCallbackKernels::template Specialization<8>::Add(); // 6
|
||||
}
|
||||
|
||||
template<int T_Q1D = 0>
|
||||
static void action_callback_new(const int d1d,
|
||||
restriction_cb_t &restriction_cb,
|
||||
qfunc_t &qfunc,
|
||||
[[maybe_unused]] input_t &inputs,
|
||||
[[maybe_unused]] const std::array<size_t, num_inputs> &input_to_field,
|
||||
const std::array<DofToQuadMap, num_inputs> &input_dtq_maps,
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
[[maybe_unused]] const int dimension,
|
||||
const int num_entities,
|
||||
[[maybe_unused]] const int test_vdim,
|
||||
[[maybe_unused]] const int num_test_dof,
|
||||
const ThreadBlocks &thread_blocks,
|
||||
[[maybe_unused]] SharedMemoryInfo<num_fields, num_inputs, num_outputs>
|
||||
&shmem_info,
|
||||
[[maybe_unused]] const Array<int> &attributes,
|
||||
[[maybe_unused]] const output_fop_t &output_fop,
|
||||
[[maybe_unused]] const Array<int> *elem_attributes,
|
||||
// refs
|
||||
std::vector<Vector> &fields_e,
|
||||
Vector &residual_e,
|
||||
std::function<void(Vector &, Vector &)> &output_restriction_transpose,
|
||||
// args
|
||||
std::vector<Vector> &solutions_l,
|
||||
const std::vector<Vector> ¶meters_l,
|
||||
Vector &residual_l,
|
||||
// fallback arguments
|
||||
const int q1d)
|
||||
{
|
||||
NVTX_MARK_FUNCTION;
|
||||
assert(dimension == 3);
|
||||
static_assert(MFEM_D2Q_MAX_SIZE >= num_inputs, "MFEM_D2Q_MAX_SIZE error");
|
||||
|
||||
constexpr int DIM = 3;
|
||||
|
||||
[[maybe_unused]] static bool ini = (for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
const auto dtq = input_dtq_maps[i];
|
||||
{
|
||||
const auto [q, _, p] = dtq.B.GetShape();
|
||||
const auto B = (const real_t*)input_dtq_maps[i].B;
|
||||
dbg("Loading Bi[{}]: q={} p={}", i.value, q, p);
|
||||
if (B) { Gpu(MemcpyToSymbol)(Bi[i], B, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
{
|
||||
const auto [q, _, p] = dtq.G.GetShape();
|
||||
const auto G = (const real_t*)input_dtq_maps[i].G;
|
||||
if (G) { Gpu(MemcpyToSymbol)(Gi[i], G, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
if constexpr (i == 0) // output B
|
||||
{
|
||||
const auto dtq_o = output_dtq_maps[0];
|
||||
const auto [q, _, p] = dtq_o.B.GetShape();
|
||||
const auto B = (const real_t*)dtq_o.B;
|
||||
if (B) { Gpu(MemcpyToSymbol)(Bo, B, (p*q)*sizeof(real_t)); }
|
||||
}
|
||||
if constexpr (i == 0) // output G
|
||||
{
|
||||
const auto dtq_o = output_dtq_maps[0];
|
||||
const auto [q, _, p] = dtq_o.G.GetShape();
|
||||
const auto G = (const real_t*)dtq_o.G;
|
||||
if (G) { Gpu(MemcpyToSymbol)(Go, G, (p*q)*sizeof(real_t)); }
|
||||
dbg("Loaded B and G to constant memory");
|
||||
}
|
||||
}), true);
|
||||
|
||||
// types
|
||||
using qf_signature =
|
||||
typename create_function_signature<decltype(&qfunc_t::operator())>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
restriction_cb(solutions_l, parameters_l, fields_e);
|
||||
|
||||
NVTX_INI("res=0");
|
||||
residual_e = 0.0;
|
||||
NVTX_END("res=0");
|
||||
|
||||
// auto wrapped_fields_e =
|
||||
// wrap_fields(fields_e, shmem_info.field_sizes, num_entities);
|
||||
|
||||
const bool has_attr = attributes.Size() > 0;
|
||||
const auto d_attr = attributes.Read();
|
||||
const auto d_elem_attr = elem_attributes->Read();
|
||||
|
||||
// const int vdim = input.vdim;
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
const int NE = num_entities;
|
||||
constexpr int VDIM = 1;
|
||||
|
||||
const auto XE = Reshape(fields_e[0].Read(), d1d, d1d, d1d, VDIM, NE);
|
||||
const real_t *dx_ptr = fields_e[1].Read();
|
||||
|
||||
auto YE = Reshape(residual_e.ReadWrite(), d1d, d1d, d1d, VDIM, NE);
|
||||
|
||||
const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
|
||||
const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
|
||||
|
||||
NVTX_INI("forall");
|
||||
dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
|
||||
{
|
||||
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
|
||||
MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
|
||||
MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
|
||||
// real_t (&sm0_ptr)[MQ1][MQ1][MQ1][3] = sm0;
|
||||
// real_t (&sm1_ptr)[MQ1][MQ1][MQ1][3] = sm1;
|
||||
|
||||
low::regs3d_t<DIM, MQ1> reg;
|
||||
const real_t *rd = dx_ptr;
|
||||
|
||||
// const auto fields_e_ptr = load_field_e_ptr(wrapped_fields_e, e);
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
// real_t (&sB_ptr)[MD1][MQ1] = sB;
|
||||
// real_t (&sG_ptr)[MD1][MQ1] = sG;
|
||||
|
||||
// Interpolate
|
||||
// for_constexpr<num_inputs>(
|
||||
// [ D1D, Q1D, MQ1, e,
|
||||
// &input_dtq_maps,
|
||||
// &sm0_ptr, &sm1_ptr,
|
||||
// &sB = sB_ptr, &sG = sG_ptr,
|
||||
// &inputs,
|
||||
// // &fields_e_ptr,
|
||||
// ®, &rd,
|
||||
// &input_to_field ] (auto i)
|
||||
{
|
||||
// const auto input = get<0/*i*/>(inputs);
|
||||
// using field_operator_t = std::decay_t<decltype(input)>;
|
||||
|
||||
// if constexpr (is_gradient_fop<field_operator_t>::value) // Grad
|
||||
{
|
||||
// const int vdim = input.vdim;
|
||||
// const real_t *field_e_r = fields_e_ptr[input_to_field[i]];
|
||||
// const auto XE = Reshape(field_e_r, D1D, D1D, D1D, vdim);
|
||||
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bi[i]);
|
||||
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Gi[i]);
|
||||
low::LoadMatrix(d1d, q1d, B, sB);
|
||||
low::LoadMatrix(d1d, q1d, G, sG);
|
||||
// for (int c = 0; c < vdim; c++)
|
||||
// constexpr int c = 0;
|
||||
{
|
||||
low::LoadDofs3d(e, d1d, XE, sm0);
|
||||
low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
|
||||
}
|
||||
}
|
||||
// else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
|
||||
{
|
||||
// db1("Identity");
|
||||
// rd = fields_e_ptr[input_to_field[i]];
|
||||
// rd = dx_ptr;
|
||||
}
|
||||
// else if constexpr (is_weight_fop<field_operator_t>::value) // Weight
|
||||
// {
|
||||
// dbg("Weight");
|
||||
// rw = fields_e_ptr[input_to_field[i]]; // 🔥
|
||||
// }
|
||||
// else
|
||||
{
|
||||
// MFApply comes here
|
||||
// assert(false);
|
||||
// MFEM_ABORT("Only Grad and Identity field operators are supported");
|
||||
}
|
||||
}//); // for_constexpr<num_inputs>
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
#if 0
|
||||
auto qf_args = decay_tuple<qf_param_ts> {};
|
||||
qf::apply_kernel<T_Q1D, num_inputs>
|
||||
(reg, reg, rd, qx, qy, qz, qfunc, qf_args);
|
||||
#elif 0
|
||||
real_t v[3], u[3] = { reg[qz][qy][qx][0],
|
||||
reg[qz][qy][qx][1],
|
||||
reg[qz][qy][qx][2]
|
||||
};
|
||||
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
kernels::Mult(3, 3, &D[qx][qy][qz][0][0], u, v);
|
||||
reg[qz][qy][qx][0] = v[0];
|
||||
reg[qz][qy][qx][1] = v[1];
|
||||
reg[qz][qy][qx][2] = v[2];
|
||||
#elif 0
|
||||
const auto *D = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
const auto args = decay_tuple<qf_param_ts>
|
||||
{
|
||||
{{ reg[qz][qy][qx][0], reg[qz][qy][qx][1], reg[qz][qy][qx][2] }},
|
||||
{{
|
||||
{{ D[qx][qy][qz][0][0], D[qx][qy][qz][0][1], D[qx][qy][qz][0][2] }},
|
||||
{{ D[qx][qy][qz][1][0], D[qx][qy][qz][1][1], D[qx][qy][qz][1][2] }},
|
||||
{{ D[qx][qy][qz][2][0], D[qx][qy][qz][2][1], D[qx][qy][qz][2][2] }}
|
||||
}
|
||||
}
|
||||
};
|
||||
const auto r = get<0>(apply(qfunc, args));
|
||||
reg[qz][qy][qx][0] = r[0];
|
||||
reg[qz][qy][qx][1] = r[1];
|
||||
reg[qz][qy][qx][2] = r[2];
|
||||
#elif 0
|
||||
auto u = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
const auto *d = (real_t (*)[T_Q1D][T_Q1D][3][3]) rd;
|
||||
auto D = as_tensor<real_t, 3, 3>(&d[qx][qy][qz][0][0]);
|
||||
auto r = D * u;
|
||||
reg[qz][qy][qx][0] = r[0];
|
||||
reg[qz][qy][qx][1] = r[1];
|
||||
reg[qz][qy][qx][2] = r[2];
|
||||
#else
|
||||
auto args = decay_tuple<qf_param_ts> {};
|
||||
get<0>(args) = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
if constexpr (T_Q1D > 0)
|
||||
{
|
||||
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
|
||||
}
|
||||
else
|
||||
{
|
||||
get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
|
||||
}
|
||||
auto r = get<0>(apply(qfunc, args));
|
||||
if constexpr (decltype(r)::ndim == 1)
|
||||
{
|
||||
as_tensor<real_t, 3>(®[qz][qy][qx][0]) = r;
|
||||
}
|
||||
else { static_assert(false); }
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// Integrate
|
||||
// if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
|
||||
{
|
||||
// const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
|
||||
// const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
|
||||
low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
|
||||
low::WriteDofs3d(d1d, 0, e, reg, YE);
|
||||
}
|
||||
},
|
||||
num_entities, thread_blocks, 0, nullptr);
|
||||
NVTX_END("forall");
|
||||
|
||||
NVTX_INI("out^T");
|
||||
output_restriction_transpose(residual_e, residual_l);
|
||||
NVTX_END("out^T");
|
||||
}
|
||||
|
||||
using NewActionKernelType = decltype(&NewActionCallback::action_callback_new<>);
|
||||
MFEM_REGISTER_KERNELS(NewActionCallbackKernels, NewActionKernelType, (int));
|
||||
|
||||
void Apply(const int d1d, const int q1d)
|
||||
{
|
||||
db1();
|
||||
NewActionCallbackKernels::Run(q1d,
|
||||
// args
|
||||
d1d,
|
||||
restriction_cb,
|
||||
qfunc,
|
||||
inputs,
|
||||
input_to_field,
|
||||
input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
dimension,
|
||||
num_entities,
|
||||
test_vdim,
|
||||
num_test_dof,
|
||||
thread_blocks,
|
||||
shmem_info,
|
||||
attributes,
|
||||
output_fop,
|
||||
elem_attributes,
|
||||
fields_e,
|
||||
residual_e,
|
||||
output_restriction_transpose,
|
||||
solutions_l,
|
||||
parameters_l,
|
||||
residual_l,
|
||||
// fallback arguments
|
||||
q1d);
|
||||
}
|
||||
};
|
||||
|
||||
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
|
||||
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
|
||||
template<int T_Q1D>
|
||||
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
|
||||
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Kernel()
|
||||
{
|
||||
return action_callback_new<T_Q1D>;
|
||||
}
|
||||
|
||||
template<size_t num_fields, size_t num_inputs, size_t num_outputs,
|
||||
typename restriction_cb_t, typename qfunc_t, typename input_t, typename output_fop_t>
|
||||
typename NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionKernelType
|
||||
NewActionCallback<num_fields, num_inputs, num_outputs, restriction_cb_t, qfunc_t, input_t, output_fop_t>::NewActionCallbackKernels::Fallback
|
||||
(int q1d)
|
||||
{
|
||||
dbg("\x1b[33mFallback q1d:{}", q1d);
|
||||
// MFEM_ABORT("No kernel for q1d=" << q1d);
|
||||
// return nullptr;
|
||||
return action_callback_new<>;
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
@@ -1,111 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "../util.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace GlobalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct Action
|
||||
{
|
||||
Action(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(std::move(qfunc)),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
|
||||
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
|
||||
|
||||
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
|
||||
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
gnqp = nqp * ctx.nentities;
|
||||
|
||||
xq_offsets.SetSize(ninputs + 1);
|
||||
xq_offsets[0] = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(inputs);
|
||||
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
|
||||
});
|
||||
xq_offsets.PartialSum();
|
||||
xq.Update(xq_offsets);
|
||||
|
||||
yq_offsets.SetSize(noutputs + 1);
|
||||
yq_offsets[0] = 0;
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
const auto output = get<i>(outputs);
|
||||
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
|
||||
});
|
||||
yq_offsets.PartialSum();
|
||||
yq.Update(yq_offsets);
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
|
||||
// E -> Q
|
||||
interpolate(input_to_infd, input_bases, xe, xq);
|
||||
|
||||
// Q -> Q
|
||||
static_assert(
|
||||
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
|
||||
"qfunc signature not supported by default backend Action");
|
||||
|
||||
detail::call_qfunc(
|
||||
qfunc, xq, yq, gnqp, input_qlayouts, output_qlayouts,
|
||||
std::make_index_sequence<ninputs> {},
|
||||
std::make_index_sequence<noutputs> {});
|
||||
|
||||
// Q -> E
|
||||
integrate(output_to_outfd, output_bases, yq, ye);
|
||||
}
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<FieldBasis, ninputs> input_bases;
|
||||
std::array<FieldBasis, noutputs> output_bases;
|
||||
|
||||
std::array<std::vector<int>, ninputs> input_qlayouts;
|
||||
std::array<std::vector<int>, noutputs> output_qlayouts;
|
||||
|
||||
int gnqp = 0;
|
||||
Array<int> xq_offsets, yq_offsets;
|
||||
mutable BlockVector xq, yq;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -1,131 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "../fem/quadinterpolator.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
#include "../util.hpp"
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace GlobalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct DerivativeActionEnzyme
|
||||
{
|
||||
DerivativeActionEnzyme(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t &qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(qfunc),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
create_fieldbases(inputs, input_to_infd, ctx.infds, ctx.ir, input_bases);
|
||||
create_fieldbases(outputs, output_to_outfd, ctx.outfds, ctx.ir, output_bases);
|
||||
|
||||
create_qlayouts(inputs, ctx.in_qlayouts, input_qlayouts);
|
||||
create_qlayouts(outputs, ctx.out_qlayouts, output_qlayouts);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
gnqp = nqp * ctx.nentities;
|
||||
|
||||
xq_offsets.SetSize(ninputs + 1);
|
||||
xq_offsets[0] = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(inputs);
|
||||
xq_offsets[i + 1] = nqp * input.size_on_qp * ctx.nentities;
|
||||
});
|
||||
xq_offsets.PartialSum();
|
||||
xq.Update(xq_offsets);
|
||||
|
||||
yq_offsets.SetSize(noutputs + 1);
|
||||
yq_offsets[0] = 0;
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
const auto output = get<i>(outputs);
|
||||
yq_offsets[i + 1] = nqp * output.size_on_qp * ctx.nentities;
|
||||
});
|
||||
yq_offsets.PartialSum();
|
||||
yq.Update(yq_offsets);
|
||||
|
||||
// For each dependent input in the dependency map we create a shadow
|
||||
// memory variable at the quadrature point level.
|
||||
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
|
||||
shadow_xq_offsets.SetSize(ninputs + 1);
|
||||
shadow_xq_offsets = 0;
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
if (activity_map[i])
|
||||
{
|
||||
shadow_xq_offsets[i + 1] =
|
||||
xq_offsets[i + 1] - xq_offsets[i];;
|
||||
}
|
||||
});
|
||||
shadow_xq_offsets.PartialSum();
|
||||
shadow_xq.Update(shadow_xq_offsets);
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
const Vector *de,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
// E -> Q
|
||||
interpolate(input_to_infd, input_bases, xe, xq);
|
||||
|
||||
const auto activity_map = detail::make_activity_map<derivative_id>(inputs);
|
||||
interpolate(input_to_infd, input_bases, xe, shadow_xq, activity_map);
|
||||
|
||||
// Q -> Q
|
||||
static_assert(
|
||||
detail::supports_tensor_array_qfunc<qfunc_t, inputs_t, outputs_t>::value,
|
||||
"qfunc signature not supported by default backend Action");
|
||||
|
||||
detail::enzyme_fwddiff<derivative_id, qfunc_t, inputs_t, outputs_t>(
|
||||
qfunc, xq, shadow_xq, yq, gnqp, input_qlayouts, output_qlayouts,
|
||||
std::make_index_sequence<ninputs> {},
|
||||
std::make_index_sequence<noutputs> {});
|
||||
|
||||
// Q -> E
|
||||
integrate(output_to_outfd, output_bases, yq, ye);
|
||||
}
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t &qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<FieldBasis, ninputs> input_bases;
|
||||
std::array<FieldBasis, noutputs> output_bases;
|
||||
|
||||
std::array<std::vector<int>, ninputs> input_qlayouts;
|
||||
std::array<std::vector<int>, noutputs> output_qlayouts;
|
||||
|
||||
int gnqp = 0;
|
||||
Array<int> xq_offsets, shadow_xq_offsets, yq_offsets;
|
||||
mutable BlockVector xq, shadow_xq, yq;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -1,42 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "action.hpp"
|
||||
#include "derivative_action_enzyme.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct GlobalQFBackend
|
||||
{
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return GlobalQFImpl::Action(ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeDerivativeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return GlobalQFImpl::DerivativeActionEnzyme<
|
||||
derivative_id, qfunc_t, inputs_t, outputs_t>(
|
||||
ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
}
|
||||
@@ -1,166 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "../util.hpp"
|
||||
#include "../../integrator_ctx.hpp"
|
||||
|
||||
#include <utility>
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
namespace LocalQFImpl
|
||||
{
|
||||
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t,
|
||||
size_t ninputs = tuple_size<inputs_t>::value,
|
||||
size_t noutputs = tuple_size<outputs_t>::value>
|
||||
struct Action
|
||||
{
|
||||
Action(
|
||||
IntegratorContext ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs) :
|
||||
ctx(ctx),
|
||||
qfunc(std::move(qfunc)),
|
||||
inputs(inputs),
|
||||
outputs(outputs)
|
||||
{
|
||||
create_fop_to_fd(inputs, ctx.infds, input_to_infd);
|
||||
create_fop_to_fd(outputs, ctx.outfds, output_to_outfd);
|
||||
|
||||
check_consistency(inputs, input_to_infd, ctx.infds);
|
||||
check_consistency(outputs, output_to_outfd, ctx.outfds);
|
||||
|
||||
const int nqp = ctx.ir.GetNPoints();
|
||||
|
||||
// Initialize DofToQuad maps for inputs
|
||||
for_constexpr<ninputs>([&](auto i)
|
||||
{
|
||||
const auto &fd = ctx.infds[input_to_infd[i]];
|
||||
std::visit([&](auto* space_ptr)
|
||||
{
|
||||
using T = std::decay_t<decltype(*space_ptr)>;
|
||||
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
|
||||
std::is_same_v<T, ParFiniteElementSpace>)
|
||||
{
|
||||
const auto *fe = space_ptr->GetTypicalFE();
|
||||
input_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
|
||||
}
|
||||
}, fd.data);
|
||||
});
|
||||
|
||||
// Initialize DofToQuad maps for outputs
|
||||
for_constexpr<noutputs>([&](auto i)
|
||||
{
|
||||
const auto &fd = ctx.outfds[output_to_outfd[i]];
|
||||
std::visit([&](auto* space_ptr)
|
||||
{
|
||||
using T = std::decay_t<decltype(*space_ptr)>;
|
||||
if constexpr (std::is_same_v<T, FiniteElementSpace> ||
|
||||
std::is_same_v<T, ParFiniteElementSpace>)
|
||||
{
|
||||
const auto *fe = space_ptr->GetTypicalFE();
|
||||
output_dtq_maps[i] = &fe->GetDofToQuad(ctx.ir, DofToQuad::TENSOR);
|
||||
}
|
||||
}, fd.data);
|
||||
});
|
||||
}
|
||||
|
||||
void operator()(
|
||||
const std::vector<Vector *> &xe,
|
||||
std::vector<Vector *> &ye) const
|
||||
{
|
||||
if (ctx.attr.Size() == 0) { return; }
|
||||
|
||||
// input_dtq_maps
|
||||
|
||||
// const auto B = (const real_t*)input_dtq_maps[0/*i*/].B;
|
||||
// const auto G = (const real_t*)input_dtq_maps[0/*i*/].G;
|
||||
|
||||
// dfem::forall<T_Q1D*T_Q1D*T_Q1D>([=] MFEM_HOST_DEVICE (int e, void *)
|
||||
// {
|
||||
// if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
|
||||
|
||||
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
|
||||
// MFEM_SHARED real_t sm0[MQ1][MQ1][MQ1][3];
|
||||
// MFEM_SHARED real_t sm1[MQ1][MQ1][MQ1][3];
|
||||
|
||||
// low::regs3d_t<DIM, MQ1> reg;
|
||||
// const real_t *rd = dx_ptr;
|
||||
|
||||
// MFEM_SHARED real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
// {
|
||||
// low::LoadMatrix(d1d, q1d, B, sB);
|
||||
// low::LoadMatrix(d1d, q1d, G, sG);
|
||||
// {
|
||||
// low::LoadDofs3d(e, d1d, XE, sm0);
|
||||
// low::Grad3d(d1d, q1d, sB, sG, sm0, sm1, reg);
|
||||
// }
|
||||
// }
|
||||
// // else if constexpr (is_identity_fop<field_operator_t>::value) // Identity
|
||||
// {
|
||||
// // db1("Identity");
|
||||
// // rd = fields_e_ptr[input_to_field[i]];
|
||||
// // rd = dx_ptr;
|
||||
// }
|
||||
// }
|
||||
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
// {
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
// {
|
||||
// MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
// {
|
||||
|
||||
// auto args = decay_tuple<qf_param_ts> {};
|
||||
// get<0>(args) = as_tensor<real_t, 3>(®[qz][qy][qx][0]);
|
||||
// if constexpr (T_Q1D > 0)
|
||||
// {
|
||||
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*T_Q1D*T_Q1D + qy*T_Q1D + qz));
|
||||
// }
|
||||
// else
|
||||
// {
|
||||
// get<1>(args) = as_tensor<real_t, 3, 3>(rd + 9*(qx*q1d*q1d + qy*q1d + qz));
|
||||
// }
|
||||
// auto r = get<0>(apply(qfunc, args));
|
||||
// if constexpr (decltype(r)::ndim == 1)
|
||||
// {
|
||||
// as_tensor<real_t, 3>(®[qz][qy][qx][0]) = r;
|
||||
// }
|
||||
// else { static_assert(false); }
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
// MFEM_SYNC_THREAD;
|
||||
// // Integrate
|
||||
// // if constexpr (is_gradient_fop<std::decay_t<output_fop_t>>::value) // Gradient
|
||||
// {
|
||||
// // const auto sB = reinterpret_cast<const real_t (*)[MQ1]>(Bo);
|
||||
// // const auto sG = reinterpret_cast<const real_t (*)[MQ1]>(Go);
|
||||
// low::GradTranspose3d(d1d, q1d, sB, sG, reg, sm1, sm0);
|
||||
// low::WriteDofs3d(d1d, 0, e, reg, YE);
|
||||
// }
|
||||
// },
|
||||
// num_entities, thread_blocks, 0, nullptr);
|
||||
}
|
||||
|
||||
|
||||
IntegratorContext ctx;
|
||||
qfunc_t qfunc;
|
||||
inputs_t inputs;
|
||||
outputs_t outputs;
|
||||
|
||||
std::array<size_t, ninputs> input_to_infd;
|
||||
std::array<size_t, noutputs> output_to_outfd;
|
||||
|
||||
std::array<const DofToQuad*, ninputs> input_dtq_maps;
|
||||
std::array<const DofToQuad*, noutputs> output_dtq_maps;
|
||||
};
|
||||
|
||||
}
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "../../integrator_ctx.hpp"
|
||||
#include "action.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct LocalQFBackend
|
||||
{
|
||||
template<
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
return LocalQFImpl::Action(ctx, qfunc, inputs, outputs);
|
||||
}
|
||||
|
||||
template<
|
||||
int derivative_id,
|
||||
typename qfunc_t,
|
||||
typename inputs_t,
|
||||
typename outputs_t>
|
||||
auto static MakeDerivativeAction(
|
||||
const IntegratorContext &ctx,
|
||||
qfunc_t qfunc,
|
||||
inputs_t inputs,
|
||||
outputs_t outputs)
|
||||
{
|
||||
MFEM_ABORT("LocalQFBackend does not support derivative actions.");
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -1,659 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "../fem/quadinterpolator.hpp"
|
||||
#include "../util.hpp"
|
||||
#include "general/enzyme.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
template <size_t N, size_t... Is>
|
||||
constexpr std::array<bool, N> all_true_impl(std::index_sequence<Is...>)
|
||||
{
|
||||
return {{((void)Is, true)...}};
|
||||
}
|
||||
|
||||
template <size_t N>
|
||||
constexpr std::array<bool, N> all_true()
|
||||
{
|
||||
return all_true_impl<N>(std::make_index_sequence<N> {});
|
||||
}
|
||||
|
||||
struct FieldBasis
|
||||
{
|
||||
// E-vector -> Q-vector
|
||||
std::function<void(const Vector &, Vector &)> forward;
|
||||
|
||||
// Q-vector -> E-vector
|
||||
std::function<void(const Vector &, Vector &)> transpose;
|
||||
};
|
||||
|
||||
inline FieldBasis FromQI(const QuadratureInterpolator *qi,
|
||||
QuadratureInterpolator::EvalFlags mode)
|
||||
{
|
||||
return
|
||||
{
|
||||
[qi, mode](const Vector &xe, Vector &xq)
|
||||
{
|
||||
qi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
if (mode == QuadratureInterpolator::VALUES)
|
||||
{
|
||||
qi->Values(xe, xq);
|
||||
}
|
||||
else
|
||||
{
|
||||
qi->Derivatives(xe, xq);
|
||||
}
|
||||
},
|
||||
[qi, mode](const Vector &yq, Vector &ye)
|
||||
{
|
||||
Vector empty;
|
||||
qi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
if (mode == QuadratureInterpolator::VALUES)
|
||||
{
|
||||
qi->AddMultTranspose(QuadratureInterpolator::VALUES, yq, empty, ye);
|
||||
}
|
||||
else
|
||||
{
|
||||
qi->AddMultTranspose(QuadratureInterpolator::DERIVATIVES, empty, yq, ye);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// QuadratureFunction identity copy
|
||||
inline FieldBasis FromQF()
|
||||
{
|
||||
return
|
||||
{
|
||||
[](const Vector &xe, Vector &xq) { xq = xe; },
|
||||
[](const Vector &yq, Vector &ye) { ye = yq; }
|
||||
};
|
||||
}
|
||||
|
||||
// User-defined parameter space B
|
||||
inline FieldBasis FromPS(const Operator *B, const Operator *Bt)
|
||||
{
|
||||
return
|
||||
{
|
||||
[B](const Vector &xe, Vector &xq) { B->Mult(xe, xq); },
|
||||
[Bt](const Vector &yq, Vector &ye) { Bt->Mult(yq, ye); }
|
||||
};
|
||||
}
|
||||
|
||||
inline FieldBasis FieldBasisFromWeight(const IntegrationRule &ir)
|
||||
{
|
||||
return
|
||||
{
|
||||
[&ir](const Vector &, Vector &xq)
|
||||
{
|
||||
const int nqp = ir.GetNPoints();
|
||||
MFEM_ASSERT(xq.Size() % nqp == 0, "weight block has unexpected size");
|
||||
|
||||
const int ne = xq.Size() / nqp;
|
||||
const real_t *wref = ir.GetWeights().Read();
|
||||
|
||||
for (int e = 0; e < ne; e++)
|
||||
{
|
||||
std::memcpy(xq.GetData() + e*nqp, wref, nqp*sizeof(real_t));
|
||||
}
|
||||
},
|
||||
[](const Vector &, Vector &) {}
|
||||
};
|
||||
}
|
||||
|
||||
inline const FieldBasis GetFieldBasis(const FieldDescriptor &f,
|
||||
const IntegrationRule &ir,
|
||||
QuadratureInterpolator::EvalFlags mode)
|
||||
{
|
||||
return std::visit([&ir, &mode](auto && arg) -> FieldBasis
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
|
||||
if constexpr (std::is_same_v<T, const FiniteElementSpace *>)
|
||||
{
|
||||
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParFiniteElementSpace *>)
|
||||
{
|
||||
return FromQI(arg->GetQuadratureInterpolator(ir), mode);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return FromQF();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return FromPS(arg->GetB(), arg->GetBt());
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const IntegrationRule *>)
|
||||
{
|
||||
return FieldBasis{};
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>, "internal error");
|
||||
}
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void create_fieldbases(
|
||||
fops_t &fops,
|
||||
const std::array<size_t, nfops> &fop_to_fd,
|
||||
const std::vector<FieldDescriptor> &fds,
|
||||
const IntegrationRule &ir,
|
||||
std::array<FieldBasis, nfops> &bases)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
const auto fop = get<i>(fops);
|
||||
using fop_t = std::decay_t<decltype(fop)>;
|
||||
|
||||
const auto fd = fds[fop_to_fd[i]];
|
||||
|
||||
constexpr QuadratureInterpolator::EvalFlags dummy_mode =
|
||||
QuadratureInterpolator::VALUES;
|
||||
if constexpr (is_identity_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, dummy_mode);
|
||||
}
|
||||
else if constexpr (is_weight_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = FieldBasisFromWeight(ir);
|
||||
}
|
||||
else if constexpr (is_value_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::VALUES);
|
||||
}
|
||||
else if constexpr (is_gradient_fop<fop_t>::value)
|
||||
{
|
||||
bases[i] = GetFieldBasis(fd, ir, QuadratureInterpolator::DERIVATIVES);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void check_consistency(
|
||||
fops_t &fops,
|
||||
const std::array<size_t, nfops> &fop_to_fd,
|
||||
const std::vector<FieldDescriptor> &fields)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
const auto input = get<i>(fops);
|
||||
using input_t = std::decay_t<decltype(input)>;
|
||||
|
||||
const auto fd = fields[fop_to_fd[i]];
|
||||
|
||||
if constexpr (is_identity_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const QuadratureFunction *>(fd.data),
|
||||
"Identity FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
else if constexpr (is_weight_fop<input_t>::value)
|
||||
{
|
||||
}
|
||||
else if constexpr (is_value_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParameterSpace *>(fd.data),
|
||||
"Value FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
else if constexpr (is_gradient_fop<input_t>::value)
|
||||
{
|
||||
MFEM_ASSERT(std::holds_alternative<const FiniteElementSpace *>(fd.data) ||
|
||||
std::holds_alternative<const ParFiniteElementSpace *>(fd.data),
|
||||
"Value FieldOperator requested on non "
|
||||
"QuadratureFunction");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t ninputs>
|
||||
void interpolate(
|
||||
const std::array<size_t, ninputs> &input_to_infd,
|
||||
const std::array<FieldBasis, ninputs> &input_bases,
|
||||
const std::vector<Vector *> &xe,
|
||||
BlockVector &xq,
|
||||
const std::array<bool, ninputs> &conditional = all_true<ninputs>())
|
||||
{
|
||||
constexpr_for<0, ninputs>([&](auto i)
|
||||
{
|
||||
if (!conditional.empty() && !conditional[i]) { return; }
|
||||
|
||||
input_bases[i].forward(*xe[input_to_infd[i]], xq.GetBlock(i));
|
||||
});
|
||||
}
|
||||
|
||||
template <size_t noutputs>
|
||||
void integrate(
|
||||
const std::array<size_t, noutputs> &output_to_outfd,
|
||||
const std::array<FieldBasis, noutputs> &output_bases,
|
||||
const BlockVector &yq,
|
||||
std::vector<Vector *> &ye)
|
||||
{
|
||||
for (auto v : ye) { *v = 0.0; }
|
||||
|
||||
constexpr_for<0, noutputs>([&](auto i)
|
||||
{
|
||||
output_bases[i].transpose(yq.GetBlock(i), *ye[output_to_outfd[i]]);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
namespace detail
|
||||
{
|
||||
|
||||
template <typename T>
|
||||
struct is_tensor_array : std::false_type {};
|
||||
|
||||
template <typename scalar_t, int... Dims>
|
||||
struct is_tensor_array<tensor_array<scalar_t, Dims...>> : std::true_type {};
|
||||
|
||||
template <typename T>
|
||||
struct is_tensor_array_mut : std::false_type {};
|
||||
|
||||
template <typename scalar_t, int... Dims>
|
||||
struct is_tensor_array_mut<tensor_array<scalar_t, Dims...>> :
|
||||
std::bool_constant<!std::is_const_v<scalar_t>> {};
|
||||
|
||||
|
||||
template <typename ndarray_t>
|
||||
inline void set_layout_default(ndarray_t &a)
|
||||
{
|
||||
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
|
||||
|
||||
constexpr std::size_t nd = ndarray_t::rank();
|
||||
constexpr std::size_t td = ndarray_t::tensor_rank();
|
||||
std::array<std::size_t, nd + td> perm{};
|
||||
|
||||
for (std::size_t i = 0; i < td; i++) { perm[i] = nd + i; }
|
||||
for (std::size_t i = 0; i < nd; i++) { perm[td + i] = i; }
|
||||
|
||||
a.set_layout(perm);
|
||||
}
|
||||
|
||||
template <typename ndarray_t>
|
||||
inline void set_layout(ndarray_t& a, const std::vector<int>& layout)
|
||||
{
|
||||
if constexpr (ndarray_t::tensor_rank() == 0) { return; }
|
||||
|
||||
constexpr std::size_t nd = ndarray_t::rank();
|
||||
constexpr std::size_t td = ndarray_t::tensor_rank();
|
||||
constexpr std::size_t N = nd + td;
|
||||
|
||||
// missing means default
|
||||
if (layout.empty()) { set_layout_default(a); return; }
|
||||
|
||||
MFEM_VERIFY(layout.size() == N,
|
||||
"layout size mismatch: expected " << N << " got " << layout.size());
|
||||
|
||||
// TODO: make a version of set_layout that takes `std::vector<int>`
|
||||
std::array<std::size_t, N> perm{};
|
||||
for (std::size_t i = 0; i < N; i++)
|
||||
{
|
||||
MFEM_VERIFY(layout[i] >= 0, "layout index must be >=0");
|
||||
perm[i] = static_cast<std::size_t>(layout[i]);
|
||||
}
|
||||
|
||||
a.set_layout(perm);
|
||||
}
|
||||
|
||||
/// Primary template: intentionally undefined — gives a clear error for unsupported types.
|
||||
template <typename T>
|
||||
struct tensor_array_traits;
|
||||
|
||||
/// Matches tensor<scalar_t, sizes...>
|
||||
template <typename scalar_t, int... sizes>
|
||||
struct tensor_array_traits<tensor<scalar_t, sizes...>>
|
||||
{
|
||||
using scalar_type = scalar_t;
|
||||
template <std::size_t ndims>
|
||||
using array_type = tensor_ndarray<scalar_t, ndims, sizes...>;
|
||||
};
|
||||
|
||||
/// Matches tensor_ndarray<scalar_t, ndims, tensor_sizes...>
|
||||
template <typename scalar_t, int ndims, int... tensor_sizes>
|
||||
struct tensor_array_traits<tensor_ndarray<scalar_t, ndims, tensor_sizes...>>
|
||||
{
|
||||
using scalar_type = scalar_t;
|
||||
template <std::size_t N>
|
||||
using array_type = tensor_ndarray<scalar_t, N, tensor_sizes...>;
|
||||
};
|
||||
|
||||
/// Entry point: explicit tensor type T as template argument.
|
||||
template <typename T, typename ptr_scalar_t, typename... dyn_sizes_t>
|
||||
decltype(auto) make_tensor_array(ptr_scalar_t *ptr,
|
||||
const std::vector<int>* layout,
|
||||
dyn_sizes_t... dynamic_sizes)
|
||||
{
|
||||
using traits = tensor_array_traits<T>;
|
||||
using array_t = typename traits::template array_type<sizeof...(dynamic_sizes)>;
|
||||
auto a = array_t(ptr, {std::size_t(dynamic_sizes)...});
|
||||
if (layout) { set_layout(a, *layout); }
|
||||
else { set_layout_default(a); }
|
||||
return a;
|
||||
}
|
||||
|
||||
template <typename qfunc_t, typename inputs_t, typename outputs_t>
|
||||
struct supports_tensor_array_qfunc
|
||||
{
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
static constexpr int ninputs = tuple_size<inputs_t>::value;
|
||||
static constexpr int noutputs = tuple_size<outputs_t>::value;
|
||||
static constexpr int nparams = tuple_size<qf_param_ts>::value;
|
||||
|
||||
template <std::size_t... Is>
|
||||
static constexpr bool InputsOk(std::index_sequence<Is...>)
|
||||
{
|
||||
return (is_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>::value && ...);
|
||||
}
|
||||
|
||||
template <std::size_t... Is>
|
||||
static constexpr bool OutputsOk(std::index_sequence<Is...>)
|
||||
{
|
||||
return (is_tensor_array_mut<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Is, qf_param_ts>::type>>>::value && ...);
|
||||
}
|
||||
|
||||
static constexpr bool value =
|
||||
(nparams == ninputs + noutputs) &&
|
||||
InputsOk(std::make_index_sequence<ninputs> {}) &&
|
||||
OutputsOk(std::make_index_sequence<noutputs> {});
|
||||
};
|
||||
|
||||
template <typename qfunc_t, std::size_t... Is, std::size_t... Os>
|
||||
inline void call_qfunc(
|
||||
const qfunc_t &qfunc,
|
||||
const BlockVector &xq,
|
||||
BlockVector &yq,
|
||||
int gnqp,
|
||||
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
|
||||
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
|
||||
std::index_sequence<Is...>,
|
||||
std::index_sequence<Os...>)
|
||||
{
|
||||
constexpr std::size_t ninputs = sizeof...(Is);
|
||||
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
|
||||
auto inputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
auto outputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
std::apply([&](auto&&... args)
|
||||
{
|
||||
qfunc(args...);
|
||||
}, std::tuple_cat(inputs, outputs));
|
||||
}
|
||||
|
||||
template <typename func_t, typename... arg_ts>
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto qfunction_wrapper(const func_t &f, arg_ts...args)
|
||||
{
|
||||
return f(args...);
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, std::size_t I, typename Tuple, std::size_t... Is>
|
||||
constexpr std::array<bool, sizeof...(Is)>
|
||||
make_activity_array(std::index_sequence<Is...>)
|
||||
{
|
||||
return { (std::decay_t<typename tuple_element<Is, Tuple>::type>::GetFieldId() == derivative_id)... };
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, typename inputs_t, std::size_t... Is>
|
||||
constexpr auto make_activity_map_impl(std::index_sequence<Is...>)
|
||||
{
|
||||
constexpr std::size_t N = sizeof...(Is);
|
||||
|
||||
if constexpr (N == 0)
|
||||
return std::array<bool, 0> {};
|
||||
|
||||
return make_activity_array<derivative_id, 0, inputs_t>
|
||||
(std::make_index_sequence<N> {});
|
||||
}
|
||||
|
||||
template <std::size_t derivative_id, typename inputs_t>
|
||||
constexpr auto make_activity_map(inputs_t)
|
||||
{
|
||||
return make_activity_map_impl<derivative_id, inputs_t>(
|
||||
std::make_index_sequence<tuple_size<inputs_t>::value> {});
|
||||
}
|
||||
|
||||
namespace enzyme_detail
|
||||
{
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t, typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
do_enzyme_call(AccArgs... acc)
|
||||
{
|
||||
__enzyme_fwddiff<qf_return_t>(wrapper_fn, acc...);
|
||||
}
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t,
|
||||
size_t CurO, size_t NO,
|
||||
typename primals_t, typename derivs_t,
|
||||
typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
process_outputs(primals_t &primals, derivs_t &derivs, AccArgs... acc)
|
||||
{
|
||||
if constexpr (CurO == NO)
|
||||
{
|
||||
do_enzyme_call<wrapper_fn, qf_return_t>(acc...);
|
||||
}
|
||||
else
|
||||
{
|
||||
process_outputs<wrapper_fn, qf_return_t, CurO + 1, NO>(
|
||||
primals, derivs,
|
||||
acc...,
|
||||
enzyme_dupnoneed,
|
||||
&std::get<CurO>(primals),
|
||||
&std::get<CurO>(derivs));
|
||||
}
|
||||
}
|
||||
|
||||
template <auto wrapper_fn, typename qf_return_t,
|
||||
size_t CurI, size_t NI, bool... ActivityMap,
|
||||
typename inputs_t, typename shadows_t,
|
||||
typename primals_t, typename derivs_t,
|
||||
typename... AccArgs>
|
||||
__attribute__((always_inline)) inline void
|
||||
process_inputs(inputs_t &inputs, shadows_t &shadows,
|
||||
primals_t &primals, derivs_t &derivs,
|
||||
AccArgs... acc)
|
||||
{
|
||||
if constexpr (CurI == NI)
|
||||
{
|
||||
constexpr size_t NO = std::tuple_size_v<primals_t>;
|
||||
process_outputs<wrapper_fn, qf_return_t, 0, NO>(
|
||||
primals, derivs, acc...);
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr bool active =
|
||||
std::array<bool, sizeof...(ActivityMap)> {ActivityMap...} [CurI];
|
||||
|
||||
if constexpr (active)
|
||||
{
|
||||
std::cout << "Input[" << CurI << "]: ACTIVE (enzyme_dup)\n"
|
||||
<< " primal ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n"
|
||||
<< " shadow ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(shadows))>() << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
std::cout << "Input[" << CurI << "]: INACTIVE (enzyme_const)\n"
|
||||
<< " primal ptr type: "
|
||||
<< get_type_name<decltype(&std::get<CurI>(inputs))>() << "\n";
|
||||
}
|
||||
|
||||
if constexpr (active)
|
||||
{
|
||||
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
|
||||
inputs, shadows, primals, derivs,
|
||||
acc...,
|
||||
enzyme_dup,
|
||||
&std::get<CurI>(inputs),
|
||||
&std::get<CurI>(shadows));
|
||||
}
|
||||
else
|
||||
{
|
||||
process_inputs<wrapper_fn, qf_return_t, CurI + 1, NI, ActivityMap...>(
|
||||
inputs, shadows, primals, derivs,
|
||||
acc...,
|
||||
enzyme_const,
|
||||
&std::get<CurI>(inputs));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace enzyme_detail
|
||||
|
||||
template <size_t derivative_id, typename qfunc_t, typename inputs_t, typename outputs_t,
|
||||
std::size_t... Is, std::size_t... Os>
|
||||
inline void enzyme_fwddiff(
|
||||
qfunc_t &qfunc,
|
||||
const BlockVector &xq,
|
||||
const BlockVector &shadow_xq,
|
||||
BlockVector &yq,
|
||||
const int &gnqp,
|
||||
const std::array<std::vector<int>, sizeof...(Is)>& in_layouts,
|
||||
const std::array<std::vector<int>, sizeof...(Os)>& out_layouts,
|
||||
std::index_sequence<Is...>,
|
||||
std::index_sequence<Os...>)
|
||||
{
|
||||
#ifdef MFEM_USE_ENZYME
|
||||
constexpr std::size_t ninputs = sizeof...(Is);
|
||||
constexpr std::size_t noutputs = sizeof...(Os);
|
||||
|
||||
using qf_signature = typename get_function_signature<qfunc_t>::type;
|
||||
using qf_param_ts = typename qf_signature::parameter_ts;
|
||||
using qf_return_t = typename qf_signature::return_t;
|
||||
|
||||
constexpr auto activity_map = make_activity_map<derivative_id>(inputs_t{});
|
||||
static_assert(activity_map.size() == ninputs, "activity map size mismatch");
|
||||
|
||||
std::cout << "activity_map: ";
|
||||
for (const auto &v : activity_map)
|
||||
{
|
||||
std::cout << v << " ";
|
||||
}
|
||||
std::cout << "\n";
|
||||
|
||||
auto inputs = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
auto shadows = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<Is, qf_param_ts>::type>>>(
|
||||
shadow_xq.GetBlock(Is).Read(), &in_layouts[Is], gnqp)...);
|
||||
|
||||
std::array<Vector, noutputs> primal_storage;
|
||||
((primal_storage[Os].SetSize(yq.GetBlock(Os).Size())), ...);
|
||||
|
||||
auto primals_out = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
primal_storage[Os].ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
auto derivs_out = std::make_tuple(
|
||||
make_tensor_array<std::remove_cv_t<std::remove_reference_t<
|
||||
typename tuple_element<ninputs + Os, qf_param_ts>::type>>>(
|
||||
yq.GetBlock(Os).ReadWrite(), &out_layouts[Os], gnqp)...);
|
||||
|
||||
using wrapper_fn_t = qf_return_t (*)(
|
||||
const qfunc_t &,
|
||||
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
|
||||
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...);
|
||||
|
||||
constexpr wrapper_fn_t wrapper_fn =
|
||||
qfunction_wrapper<qfunc_t,
|
||||
std::remove_reference_t<decltype(std::get<Is>(inputs))>...,
|
||||
std::remove_reference_t<decltype(std::get<Os>(primals_out))>...>;
|
||||
|
||||
// wrapper_fn travels as a non-type template parameter throughout without
|
||||
// being stored.
|
||||
enzyme_detail::process_inputs<
|
||||
wrapper_fn,
|
||||
qf_return_t,
|
||||
0,
|
||||
ninputs,
|
||||
activity_map[Is]...
|
||||
>(inputs, shadows,
|
||||
primals_out, derivs_out,
|
||||
enzyme_const, &qfunc // seed: qfunc is always inactive
|
||||
);
|
||||
|
||||
#else
|
||||
MFEM_ABORT("enzyme_fwddiff requires MFEM_USE_ENZYME");
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
// Create quadrature function fop to fields map
|
||||
template <typename fops_t, size_t N = tuple_size<fops_t>::value, size_t M>
|
||||
void create_fop_to_fd(const fops_t &fops,
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
std::array<size_t, M> &fop_to_fd)
|
||||
{
|
||||
static_assert(N == M, "sizes must match");
|
||||
constexpr_for<0, N>([&](auto i)
|
||||
{
|
||||
const auto fop = get<i>(fops);
|
||||
fop_to_fd[i] = std::numeric_limits<size_t>::max();
|
||||
for (size_t j = 0; j < fields.size(); j++)
|
||||
{
|
||||
// TODO: output.GetFieldId() should probably store/return size_t
|
||||
if (static_cast<int>(fields[j].id) == fop.GetFieldId())
|
||||
{
|
||||
fop_to_fd[i] = j;
|
||||
}
|
||||
}
|
||||
// Handle Weight type. There is no FieldDescriptor for the weight.
|
||||
// TODO: Create weight descriptor for the weight for internal use?
|
||||
// TODO: this is a hack...
|
||||
if (is_weight_fop<std::remove_cv_t<decltype(fop)>>::value)
|
||||
{
|
||||
fop_to_fd[i] = 0;
|
||||
}
|
||||
else if (fop_to_fd[i] == std::numeric_limits<size_t>::max())
|
||||
{
|
||||
MFEM_ABORT("not found");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template <typename fops_t, size_t nfops>
|
||||
void create_qlayouts(const fops_t &fops,
|
||||
const std::unordered_map<std::type_index, std::vector<int>> &a,
|
||||
std::array<std::vector<int>, nfops> &b)
|
||||
{
|
||||
constexpr_for<0, nfops>([&](auto i)
|
||||
{
|
||||
using fop_t =
|
||||
std::remove_cv_t<std::remove_reference_t<decltype(get<i>(fops))>>;
|
||||
auto it = a.find(std::type_index(typeid(fop_t)));
|
||||
if (it != a.end()) { b[i] = it->second; }
|
||||
else { b[i].clear(); }
|
||||
});
|
||||
}
|
||||
|
||||
}
|
||||
+23
-98
@@ -11,119 +11,44 @@
|
||||
|
||||
#include "doperator.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
using namespace mfem;
|
||||
using namespace mfem::future;
|
||||
|
||||
void DifferentiableOperator::SetParameters(std::vector<Vector *> p) const
|
||||
{
|
||||
MFEM_ASSERT(parameters.size() == p.size(),
|
||||
"number of parameters doesn't match descriptors");
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
{
|
||||
p[i]->Read();
|
||||
parameters_l[i] = *p[i];
|
||||
}
|
||||
}
|
||||
|
||||
DifferentiableOperator::DifferentiableOperator(
|
||||
const std::vector<FieldDescriptor> &infds,
|
||||
const std::vector<FieldDescriptor> &outfds,
|
||||
const std::vector<FieldDescriptor> &solutions,
|
||||
const std::vector<FieldDescriptor> ¶meters,
|
||||
const ParMesh &mesh) :
|
||||
Operator(),
|
||||
mesh(mesh),
|
||||
infds(infds),
|
||||
outfds(outfds)
|
||||
solutions(solutions),
|
||||
parameters(parameters)
|
||||
{
|
||||
unionfds.clear();
|
||||
unionfds.insert(unionfds.end(), infds.begin(), infds.end());
|
||||
unionfds.insert(unionfds.end(), outfds.begin(), outfds.end());
|
||||
std::sort(unionfds.begin(), unionfds.end());
|
||||
auto last = std::unique(unionfds.begin(), unionfds.end());
|
||||
unionfds.erase(last, unionfds.end());
|
||||
fields.resize(solutions.size() + parameters.size());
|
||||
fields_e.resize(fields.size());
|
||||
solutions_l.resize(solutions.size());
|
||||
parameters_l.resize(parameters.size());
|
||||
|
||||
infields_l.resize(infds.size());
|
||||
for (size_t i = 0; i < infds.size(); i++)
|
||||
for (size_t i = 0; i < solutions.size(); i++)
|
||||
{
|
||||
infields_l[i] = new Vector(GetVSize(infds[i]));
|
||||
fields[i] = solutions[i];
|
||||
}
|
||||
|
||||
infields_e.resize(infds.size());
|
||||
}
|
||||
|
||||
void DifferentiableOperator::SetMultLevel(MultLevel level)
|
||||
{
|
||||
mult_level = level;
|
||||
}
|
||||
|
||||
void DifferentiableOperator::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_ASSERT(!action_callbacks.empty(),
|
||||
"no integrators have been set");
|
||||
|
||||
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&x),
|
||||
"x needs to be a BlockVector");
|
||||
|
||||
MFEM_ASSERT(dynamic_cast<const BlockVector*>(&y),
|
||||
"y needs to be a BlockVector");
|
||||
|
||||
const auto &bx = static_cast<const BlockVector &>(x);
|
||||
auto &by = static_cast<BlockVector &>(y);
|
||||
|
||||
Mult(bx, by);
|
||||
}
|
||||
|
||||
void DifferentiableOperator::DisableTensorProductStructure(bool disable)
|
||||
{
|
||||
use_tensor_product_structure = !disable;
|
||||
}
|
||||
|
||||
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
|
||||
size_t derivative_id, const Vector &x)
|
||||
{
|
||||
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
|
||||
derivative_action_callbacks.end(),
|
||||
"no derivative action has been found for ID " << derivative_id);
|
||||
|
||||
const size_t dfidx = FindIdx(derivative_id, infds);
|
||||
|
||||
// Get transpose callbacks if available, otherwise pass empty vector
|
||||
std::vector<derivative_action_t> transpose_callbacks;
|
||||
auto it = daction_transpose_callbacks.find(derivative_id);
|
||||
if (it != daction_transpose_callbacks.end())
|
||||
for (size_t i = 0; i < parameters.size(); i++)
|
||||
{
|
||||
transpose_callbacks = it->second;
|
||||
fields[i + solutions.size()] = parameters[i];
|
||||
}
|
||||
|
||||
return std::make_shared<DerivativeOperator>(
|
||||
height,
|
||||
GetTrueVSize(infds[dfidx]),
|
||||
derivative_action_callbacks[derivative_id],
|
||||
transpose_callbacks,
|
||||
infds[dfidx],
|
||||
x,
|
||||
infds,
|
||||
outfds);
|
||||
}
|
||||
|
||||
std::shared_ptr<DerivativeOperator> DifferentiableOperator::GetDerivative(
|
||||
size_t derivative_id, const MultiVector &x)
|
||||
{
|
||||
MFEM_ASSERT(derivative_action_callbacks.find(derivative_id) !=
|
||||
derivative_action_callbacks.end(),
|
||||
"no derivative action has been found for ID " << derivative_id);
|
||||
|
||||
const size_t dfidx = FindIdx(derivative_id, infds);
|
||||
|
||||
// Get transpose callbacks if available, otherwise pass empty vector
|
||||
std::vector<derivative_action_t> transpose_callbacks;
|
||||
auto it = daction_transpose_callbacks.find(derivative_id);
|
||||
if (it != daction_transpose_callbacks.end())
|
||||
{
|
||||
transpose_callbacks = it->second;
|
||||
}
|
||||
|
||||
return std::make_shared<DerivativeOperator>(
|
||||
height,
|
||||
GetTrueVSize(infds[dfidx]),
|
||||
derivative_action_callbacks[derivative_id],
|
||||
transpose_callbacks,
|
||||
infds[dfidx],
|
||||
x,
|
||||
infds,
|
||||
outfds);
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
+897
-237
File diff suppressed because it is too large
Load Diff
@@ -1,63 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../fespace.hpp"
|
||||
#include "parameterspace.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
/// @brief FieldDescriptor struct
|
||||
///
|
||||
/// This struct is used to store information about a field.
|
||||
struct FieldDescriptor
|
||||
{
|
||||
using data_variant_t =
|
||||
std::variant<const FiniteElementSpace *,
|
||||
const ParFiniteElementSpace *,
|
||||
const QuadratureFunction *,
|
||||
const ParameterSpace *>;
|
||||
|
||||
/// Field ID
|
||||
std::size_t id;
|
||||
|
||||
/// Field variant
|
||||
data_variant_t data;
|
||||
|
||||
/// Default constructor
|
||||
FieldDescriptor() :
|
||||
id(SIZE_MAX), data(data_variant_t{}) {}
|
||||
|
||||
/// Constructor
|
||||
template <typename T>
|
||||
FieldDescriptor(std::size_t field_id, const T* v) :
|
||||
id(field_id), data(v) {}
|
||||
|
||||
bool operator==(const FieldDescriptor& other) const
|
||||
{
|
||||
return id == other.id;
|
||||
}
|
||||
|
||||
bool operator<(const FieldDescriptor& other) const
|
||||
{
|
||||
return id < other.id;
|
||||
}
|
||||
|
||||
friend void swap(FieldDescriptor& a, FieldDescriptor& b)
|
||||
{
|
||||
using std::swap;
|
||||
swap(a.id, b.id);
|
||||
swap(a.data, b.data);
|
||||
}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include "util.hpp"
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
struct IntegratorContext
|
||||
{
|
||||
const ParMesh &mesh;
|
||||
const Array<int> *elem_attr;
|
||||
Array<int> attr;
|
||||
int nentities;
|
||||
const std::vector<FieldDescriptor> &infds;
|
||||
const std::vector<FieldDescriptor> &outfds;
|
||||
const std::vector<FieldDescriptor> &unionfds;
|
||||
const IntegrationRule &ir;
|
||||
std::unordered_map<std::type_index, std::vector<int>> &in_qlayouts;
|
||||
std::unordered_map<std::type_index, std::vector<int>> &out_qlayouts;
|
||||
};
|
||||
|
||||
}
|
||||
@@ -9,23 +9,8 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
// #define NVTX_COLOR nvtx::kPeru
|
||||
|
||||
#include "util.hpp"
|
||||
#include "fem/kernels.hpp"
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template <class T>
|
||||
inline std::enable_if_t<!std::numeric_limits<T>::is_integer, bool>
|
||||
AlmostEq(T x, T y, T tolerance = 15.0 * std::numeric_limits<T>::epsilon())
|
||||
{
|
||||
const T neg = std::abs(x - y);
|
||||
constexpr T min = std::numeric_limits<T>::min();
|
||||
constexpr T eps = std::numeric_limits<T>::epsilon();
|
||||
const T min_abs = std::min(std::abs(x), std::abs(y));
|
||||
if (std::abs(min_abs) == 0.0) { return neg < eps; }
|
||||
return (neg / (1.0 + std::max(min, min_abs))) < tolerance;
|
||||
}
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
@@ -45,7 +30,6 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
|
||||
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
dbg("Value");
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
@@ -110,11 +94,10 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
else if constexpr (
|
||||
is_gradient_fop<std::decay_t<field_operator_t>>::value)
|
||||
{
|
||||
// dbg("Gradient");
|
||||
const auto [q1d, B_dim, d1d] = B.GetShape();
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&std::as_const(field_e[0]), d1d, d1d, d1d, vdim);
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
@@ -123,30 +106,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
|
||||
|
||||
// constexpr int MQ1 = T_Q1D > 0 ? T_Q1D : 8;
|
||||
// static constexpr int DIM = 3;
|
||||
// MFEM_VERIFY(q1d <= MQ1, "q1d > MQ1");
|
||||
// MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
|
||||
// kernels::internal::d_regs3d_t<DIM, MQ1> r0, r1;
|
||||
// real_t sB[MQ1][MQ1], sG[MQ1][MQ1];
|
||||
|
||||
/*
|
||||
{
|
||||
assert(B_dim == 1 && "1D B required!");
|
||||
kernels::internal::LoadMatrix(d1d, q1d, B, sB);
|
||||
kernels::internal::LoadMatrix(d1d, q1d, G, sG);
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
assert(AlmostEq(B(qx, 0, dx), sB[dx][qx]));
|
||||
assert(AlmostEq(G(qx, 0, dx), sG[dx][qx]));
|
||||
}
|
||||
}
|
||||
}*/
|
||||
|
||||
for (int c = 0; c < vdim; c++)
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
@@ -157,7 +117,7 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, c);
|
||||
const real_t f = field(dx, dy, dz, vd);
|
||||
uv[0] += f * B(qx, 0, dx);
|
||||
uv[1] += f * G(qx, 0, dx);
|
||||
}
|
||||
@@ -203,59 +163,19 @@ void map_field_to_quadrature_data_tensor_product_3d(
|
||||
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
|
||||
}
|
||||
fqp(c, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(c, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(c, 2, qx, qy, qz) = uvw[2];
|
||||
fqp(vd, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(vd, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(vd, 2, qx, qy, qz) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/*
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
kernels::internal::LoadDofs3d(d1d, c, field, r0);
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, c);
|
||||
assert(AlmostEq(f, r0[d][dz][dy][dx]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
kernels::internal::Grad3d(d1d, q1d, smem, sB, sG, r0, r1, c);
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
if (!AlmostEq(fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]))
|
||||
{
|
||||
dbg("\x1b[31m[{}:d] {} {}", c, fqp(c, d, qx, qy, qz), r1[d][qz][qy][qx]);
|
||||
dbg("❌❌❌"), std::exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// dbg("✅✅✅✅✅✅✅✅✅✅✅✅✅✅✅");//, std::exit(EXIT_SUCCESS);
|
||||
}*/
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
|
||||
{
|
||||
// dbg("None");
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
@@ -598,9 +518,6 @@ void map_fields_to_quadrature_data(
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
// dbg();
|
||||
assert(use_sum_factorization && "❌ use_sum_factorization required");
|
||||
|
||||
// When the input_to_field map returns -1, this means the requested input
|
||||
// is the integration weight. Weights don't have a user defined field
|
||||
// attached to them and we create a dummy field which is not accessed
|
||||
@@ -661,7 +578,6 @@ void map_field_to_quadrature_data_conditional(
|
||||
const int &dimension,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
assert(false && "❌ condition not implemented");
|
||||
if (condition)
|
||||
{
|
||||
if (use_sum_factorization)
|
||||
@@ -703,7 +619,6 @@ void map_fields_to_quadrature_data_conditional(
|
||||
const std::array<bool, num_inputs> &conditions,
|
||||
const bool &use_sum_factorization = false)
|
||||
{
|
||||
assert(false && "❌ condition not implemented");
|
||||
for_constexpr<num_inputs>([&](auto i)
|
||||
{
|
||||
map_field_to_quadrature_data_conditional(
|
||||
@@ -712,7 +627,7 @@ void map_fields_to_quadrature_data_conditional(
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_Q1D, size_t num_inputs, typename field_operator_ts>
|
||||
template <size_t num_inputs, typename field_operator_ts>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_direction_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_inputs> &directions_qp,
|
||||
@@ -745,7 +660,7 @@ void map_direction_to_quadrature_data_conditional(
|
||||
}
|
||||
else if (dimension == 3)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product_3d<T_Q1D>(
|
||||
map_field_to_quadrature_data_tensor_product_3d(
|
||||
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
|
||||
integration_weights, scratch_mem);
|
||||
}
|
||||
|
||||
@@ -43,7 +43,7 @@ public:
|
||||
/// Get spatial dimension
|
||||
///
|
||||
/// returns always 1.
|
||||
constexpr int Dimension() const
|
||||
int Dimension() const
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
@@ -74,14 +74,11 @@ public:
|
||||
return elem_restr.get();
|
||||
}
|
||||
|
||||
virtual const Operator* GetB() const = 0;
|
||||
|
||||
virtual const Operator* GetBt() const = 0;
|
||||
|
||||
protected:
|
||||
int vdim;
|
||||
DofToQuad dtq;
|
||||
mutable std::unique_ptr<Operator> prolongation, elem_restr, B, Bt;
|
||||
mutable std::unique_ptr<Operator> prolongation;
|
||||
mutable std::unique_ptr<Operator> elem_restr;
|
||||
};
|
||||
|
||||
/// @brief Uniform parameter space
|
||||
@@ -125,18 +122,6 @@ public:
|
||||
return lsize;
|
||||
}
|
||||
|
||||
const Operator* GetB() const override
|
||||
{
|
||||
MFEM_ABORT("UniformParameterSpace does not support GetB");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const Operator* GetBt() const override
|
||||
{
|
||||
MFEM_ABORT("UniformParameterSpace does not support GetBt");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
private:
|
||||
/// T-vector size
|
||||
int tsize;
|
||||
|
||||
@@ -243,8 +243,6 @@ void process_qf_arg(
|
||||
}
|
||||
}
|
||||
|
||||
// const tensor<real_t, DIM> ∇u
|
||||
// const tensor<real_t, DIM, DIM> D (PA_DATA)
|
||||
template <typename arg_type>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_qf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
|
||||
|
||||
@@ -1,76 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "tuple.hpp"
|
||||
#include "../linalg/tensor.hpp"
|
||||
|
||||
using namespace mfem::future;
|
||||
using mfem::future::tensor;
|
||||
|
||||
// Helper to add dimension to tensor type
|
||||
template<typename T, int qp>
|
||||
struct AddQPDimension;
|
||||
|
||||
// Specialization for tensor<real_t, dim>
|
||||
template<typename real_t, int dim, int qp>
|
||||
struct AddQPDimension<tensor<real_t, dim>, qp>
|
||||
{
|
||||
using type = tensor<real_t, dim, qp>;
|
||||
};
|
||||
|
||||
// Specialization for tensor<real_t, dim, dim>
|
||||
template<typename real_t, int dim, int qp>
|
||||
struct AddQPDimension<tensor<real_t, dim, dim>, qp>
|
||||
{
|
||||
using type = tensor<real_t, dim, dim, qp>;
|
||||
};
|
||||
|
||||
// Specialization for real_t (transforms to tensor<real_t, qp>)
|
||||
template<typename real_t, int qp>
|
||||
struct AddQPDimension
|
||||
{
|
||||
using type = tensor<real_t, qp>;
|
||||
};
|
||||
|
||||
// Helper to transform tuple
|
||||
template<typename Tuple, int qp>
|
||||
struct TransformTupleQP {};
|
||||
|
||||
// Specialization for mfem::future::tuple
|
||||
template<int qp, typename... Types>
|
||||
struct TransformTupleQP<mfem::future::tuple<Types...>, qp>
|
||||
{
|
||||
using type = mfem::future::tuple<typename AddQPDimension<Types, qp>::type...>;
|
||||
};
|
||||
|
||||
template<int qp, typename... Types>
|
||||
struct TransformTupleQP<std::tuple<Types...>, qp>
|
||||
{
|
||||
using type = std::tuple<typename AddQPDimension<Types, qp>::type...>;
|
||||
};
|
||||
|
||||
// Function to transform tuple type with qp dimension
|
||||
template<int qp, typename qf_param_ts>
|
||||
struct add_qp_dimension
|
||||
{
|
||||
using type = typename TransformTupleQP<qf_param_ts, qp>::type;
|
||||
};
|
||||
|
||||
// Helper alias template for cleaner usage
|
||||
template<int qp, typename qf_param_ts>
|
||||
using add_qp_dimension_t = typename add_qp_dimension<qp, qf_param_ts>::type;
|
||||
|
||||
// ...AddDomainIntegrator...
|
||||
// {
|
||||
// constexpr int Q1D = 4;
|
||||
// using qf_param_augmentd_ts = add_qp_dimension_t<Q1D, decay_tuple<qf_param_ts>>;
|
||||
// }
|
||||
+49
-532
@@ -21,7 +21,6 @@
|
||||
#include <type_traits>
|
||||
#include <numeric>
|
||||
#include <iomanip>
|
||||
#include <typeindex>
|
||||
|
||||
#include "../../general/communication.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
@@ -29,19 +28,13 @@
|
||||
#include "../fe/fe_base.hpp"
|
||||
#include "../fespace.hpp"
|
||||
#include "../pfespace.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../../mesh/mesh.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../quadinterpolator.hpp"
|
||||
|
||||
#include "fielddescriptor.hpp"
|
||||
#include "fieldoperator.hpp"
|
||||
#include "parameterspace.hpp"
|
||||
#include "tuple.hpp"
|
||||
|
||||
#undef NVTX_COLOR
|
||||
#define NVTX_COLOR ::nvtx::kLightBlue
|
||||
|
||||
namespace mfem::future
|
||||
{
|
||||
|
||||
@@ -82,7 +75,7 @@ constexpr void for_constexpr(lambda&& f,
|
||||
}
|
||||
|
||||
template <typename lambda>
|
||||
constexpr void for_constexpr(lambda&&, std::integer_sequence<std::size_t>) {}
|
||||
constexpr void for_constexpr(lambda&& f, std::integer_sequence<std::size_t>) {}
|
||||
|
||||
template <int... n, typename lambda>
|
||||
constexpr void for_constexpr(lambda&& f)
|
||||
@@ -91,7 +84,7 @@ constexpr void for_constexpr(lambda&& f)
|
||||
}
|
||||
|
||||
template <typename lambda, typename arg_t>
|
||||
constexpr void for_constexpr_with_arg(lambda&&, arg_t&&,
|
||||
constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg,
|
||||
std::integer_sequence<std::size_t>)
|
||||
{
|
||||
// Base case - do nothing for empty sequence
|
||||
@@ -115,16 +108,6 @@ constexpr void for_constexpr_with_arg(lambda&& f, arg_t&& arg)
|
||||
indices{});
|
||||
}
|
||||
|
||||
template <auto start, auto end, auto inc = 1, typename F>
|
||||
constexpr void constexpr_for(F&& f)
|
||||
{
|
||||
if constexpr (start < end)
|
||||
{
|
||||
f(std::integral_constant<decltype(start), start>());
|
||||
constexpr_for<start + inc, end, inc>(f);
|
||||
}
|
||||
}
|
||||
|
||||
template <std::size_t I, typename Tuple, std::size_t... Is>
|
||||
std::array<bool, sizeof...(Is)>
|
||||
make_dependency_array(const Tuple& inputs, std::index_sequence<Is...>)
|
||||
@@ -461,21 +444,6 @@ struct create_function_signature<output_t (*)(input_ts...)>
|
||||
using type = FunctionSignature<output_t(input_ts...)>;
|
||||
};
|
||||
|
||||
template <typename...>
|
||||
using void_t = void;
|
||||
|
||||
template <typename T, typename = void>
|
||||
struct get_function_signature
|
||||
{
|
||||
using type = typename create_function_signature<T>::type;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct get_function_signature<T, void_t<decltype(&T::operator())>>
|
||||
{
|
||||
using type = typename create_function_signature<decltype(&T::operator())>::type;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
constexpr int GetFieldId()
|
||||
{
|
||||
@@ -570,12 +538,38 @@ auto get_marked_entries(
|
||||
/// @param t the tuple to filter fields from.
|
||||
/// @returns a tuple containing only the fields with field IDs not equal to -1.
|
||||
template <typename... Ts>
|
||||
constexpr auto filter_fields(const std::tuple<Ts...>&)
|
||||
constexpr auto filter_fields(const std::tuple<Ts...>& t)
|
||||
{
|
||||
return std::tuple_cat(
|
||||
std::conditional_t<Ts::GetFieldId() != -1, std::tuple<Ts>, std::tuple<>> {}...);
|
||||
}
|
||||
|
||||
/// @brief FieldDescriptor struct
|
||||
///
|
||||
/// This struct is used to store information about a field.
|
||||
struct FieldDescriptor
|
||||
{
|
||||
using data_variant_t =
|
||||
std::variant<const FiniteElementSpace *,
|
||||
const ParFiniteElementSpace *,
|
||||
const ParameterSpace *>;
|
||||
|
||||
/// Field ID
|
||||
std::size_t id;
|
||||
|
||||
/// Field variant
|
||||
data_variant_t data;
|
||||
|
||||
/// Default constructor
|
||||
FieldDescriptor() :
|
||||
id(SIZE_MAX), data(data_variant_t{}) {}
|
||||
|
||||
/// Constructor
|
||||
template <typename T>
|
||||
FieldDescriptor(std::size_t field_id, const T* v) :
|
||||
id(field_id), data(v) {}
|
||||
};
|
||||
|
||||
namespace dfem
|
||||
{
|
||||
template <class... T> constexpr bool always_false = false;
|
||||
@@ -605,7 +599,7 @@ struct ThreadBlocks
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_extern_shmem(func_t f, int n)
|
||||
__global__ void forall_kernel_shmem(func_t f, int n)
|
||||
{
|
||||
int i = blockIdx.x;
|
||||
extern __shared__ real_t shmem[];
|
||||
@@ -614,48 +608,23 @@ __global__ void forall_kernel_extern_shmem(func_t f, int n)
|
||||
f(i, shmem);
|
||||
}
|
||||
}
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_static_smem(func_t f, int n)
|
||||
{
|
||||
int i = blockIdx.x;
|
||||
if (i >= n) { return; }
|
||||
f(i, nullptr);
|
||||
}
|
||||
template <int MAX_THREADS_PER_BLOCK, typename func_t>
|
||||
__global__
|
||||
MFEM_LAUNCH_BOUNDS(MAX_THREADS_PER_BLOCK)
|
||||
static void forall_kernel_static_smem_launch_bounds(func_t f, int n)
|
||||
{
|
||||
for (int k = blockIdx.x; k < n; k += gridDim.x) { f(k, nullptr); }
|
||||
}
|
||||
#endif
|
||||
|
||||
template </*typename kernel_tag,*/ typename func_t>
|
||||
template <typename func_t>
|
||||
void forall(func_t f,
|
||||
const int &N,
|
||||
[[maybe_unused]] const ThreadBlocks &blocks,
|
||||
[[maybe_unused]] int num_shmem = 0,
|
||||
const ThreadBlocks &blocks,
|
||||
int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
db1();
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
// int gridsize = (N + Z - 1) / Z;
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
db1("num_bytes:{}", num_bytes);
|
||||
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
// ForallKernel<kernel_tag>::run<<<N, block_size, num_bytes>>>(f, N);
|
||||
if (num_bytes > 0)
|
||||
{
|
||||
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
forall_kernel_static_smem<<<N, block_size>>>(f, N);
|
||||
}
|
||||
forall_kernel_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
@@ -666,7 +635,6 @@ void forall(func_t f,
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
db1("CPU_MASK");
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
for (int i = 0; i < N; i++)
|
||||
@@ -680,69 +648,6 @@ void forall(func_t f,
|
||||
}
|
||||
}
|
||||
|
||||
namespace dfem
|
||||
{
|
||||
|
||||
template <int MAX_THREADS_PER_BLOCK = 0, typename func_t>
|
||||
void forall(func_t f,
|
||||
const int &N,
|
||||
[[maybe_unused]] const ThreadBlocks &blocks,
|
||||
[[maybe_unused]] int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
db1();
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
db1("num_bytes:{}", num_bytes);
|
||||
db1("block: {}x{}x{}", blocks.x, blocks.y, blocks.z);
|
||||
db1("MAX_THREADS_PER_BLOCK:{}", MAX_THREADS_PER_BLOCK);
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
if constexpr (MAX_THREADS_PER_BLOCK > 0)
|
||||
{
|
||||
assert(num_bytes == 0);
|
||||
forall_kernel_static_smem_launch_bounds
|
||||
<MAX_THREADS_PER_BLOCK><<<N, block_size>>> (f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(MAX_THREADS_PER_BLOCK == 0);
|
||||
if (num_bytes == 0)
|
||||
{
|
||||
forall_kernel_static_smem<<<N, block_size>>>(f, N);
|
||||
}
|
||||
else
|
||||
{
|
||||
forall_kernel_extern_shmem<<<N, block_size, num_bytes>>>(f, N);
|
||||
}
|
||||
}
|
||||
#if defined(MFEM_USE_CUDA)
|
||||
MFEM_GPU_CHECK(cudaGetLastError());
|
||||
#elif defined(MFEM_USE_HIP)
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
#endif
|
||||
// MFEM_DEVICE_SYNC; // ⚠️
|
||||
#endif
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
db1("CPU_MASK");
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
f(i, shmem);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("no compute backend available");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// @todo To be removed.
|
||||
class FDJacobian : public Operator
|
||||
{
|
||||
@@ -867,10 +772,6 @@ int GetVSize(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetVSize();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->Size();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetVSize();
|
||||
@@ -909,10 +810,6 @@ void GetElementVDofs(const FieldDescriptor &f, int el, Array<int> &vdofs)
|
||||
{
|
||||
arg->GetElementVDofs(el, vdofs);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
MFEM_ABORT("internal error");
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
MFEM_ABORT("internal error");
|
||||
@@ -947,10 +844,6 @@ int GetTrueVSize(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetTrueVSize();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->Size();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetTrueVSize();
|
||||
@@ -981,10 +874,6 @@ int GetVDim(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetVDim();
|
||||
@@ -1020,10 +909,6 @@ int GetDimension(const FieldDescriptor &f)
|
||||
return arg->GetMesh()->Dimension() - 1;
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return arg->GetSpace()->GetMesh()->Dimension();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->Dimension();
|
||||
@@ -1036,36 +921,6 @@ int GetDimension(const FieldDescriptor &f)
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
inline
|
||||
std::variant<const QuadratureInterpolator *, const Operator *>get_qinterp(
|
||||
const FieldDescriptor &f,
|
||||
const IntegrationRule &ir)
|
||||
{
|
||||
return std::visit([&ir](auto && arg) -> const QuadratureInterpolator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
|
||||
std::is_same_v<T, const ParFiniteElementSpace *>)
|
||||
{
|
||||
return arg->GetQuadratureInterpolator(ir);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
// QuadratureFunction doesn't need a QuadratureInterpolator
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(dfem::always_false<T>, "internal error");
|
||||
}
|
||||
|
||||
return nullptr; // Unreachable, but avoids compiler warning
|
||||
}, f.data);
|
||||
}
|
||||
|
||||
/// @brief Get the prolongation operator for a field descriptor.
|
||||
///
|
||||
@@ -1074,7 +929,6 @@ std::variant<const QuadratureInterpolator *, const Operator *>get_qinterp(
|
||||
inline
|
||||
const Operator *get_prolongation(const FieldDescriptor &f)
|
||||
{
|
||||
NVTX("get P");
|
||||
return std::visit([](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -1083,10 +937,6 @@ const Operator *get_prolongation(const FieldDescriptor &f)
|
||||
{
|
||||
return arg->GetProlongationMatrix();
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetProlongationMatrix();
|
||||
@@ -1109,7 +959,6 @@ inline
|
||||
const Operator *get_element_restriction(const FieldDescriptor &f,
|
||||
ElementDofOrdering o)
|
||||
{
|
||||
NVTX("get ER");
|
||||
return std::visit([&o](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -1118,10 +967,6 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
|
||||
{
|
||||
return arg->GetElementRestriction(o);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return arg->GetElementRestriction(o);
|
||||
@@ -1149,7 +994,6 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
|
||||
FaceType ft,
|
||||
L2FaceValues m)
|
||||
{
|
||||
NVTX("get FR");
|
||||
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
|
||||
{
|
||||
using T = std::decay_t<decltype(arg)>;
|
||||
@@ -1158,11 +1002,6 @@ const Operator *get_face_restriction(const FieldDescriptor &f,
|
||||
{
|
||||
return arg->GetFaceRestriction(o, ft, m);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
// QuadratureFunction does not support face restrictions
|
||||
MFEM_ABORT("internal error");
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
// ParameterSpace does not support face restrictions
|
||||
@@ -1188,7 +1027,6 @@ inline
|
||||
const Operator *get_restriction(const FieldDescriptor &f,
|
||||
const ElementDofOrdering &o)
|
||||
{
|
||||
NVTX("get R");
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
return get_element_restriction(f, o);
|
||||
@@ -1214,14 +1052,12 @@ inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
|
||||
get_restriction_transpose(
|
||||
const FieldDescriptor &f,
|
||||
const ElementDofOrdering &o,
|
||||
[[maybe_unused]] const fop_t &fop)
|
||||
const fop_t &fop)
|
||||
{
|
||||
NVTX("get R^T");
|
||||
if constexpr (is_sum_fop<fop_t>::value)
|
||||
{
|
||||
auto RT = [=](const Vector &v_e, Vector &v_l)
|
||||
{
|
||||
NVTX("R^T sum");
|
||||
v_l += v_e;
|
||||
};
|
||||
return std::make_tuple(RT, 1);
|
||||
@@ -1231,7 +1067,6 @@ get_restriction_transpose(
|
||||
const Operator *R = get_restriction<entity_t>(f, o);
|
||||
std::function<void(const Vector&, Vector&)> RT = [=](const Vector &x, Vector &y)
|
||||
{
|
||||
NVTX("R^T+");
|
||||
R->AddMultTranspose(x, y);
|
||||
};
|
||||
return std::make_tuple(RT, R->Height());
|
||||
@@ -1251,26 +1086,11 @@ get_restriction_transpose(
|
||||
inline
|
||||
void prolongation(const FieldDescriptor field, const Vector &x, Vector &field_l)
|
||||
{
|
||||
NVTX("P");
|
||||
const auto P = get_prolongation(field);
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
field_l.SetSize(P->Height());
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("P->Mult");
|
||||
P->Mult(x, field_l);
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const FieldDescriptor &field, const Vector &field_l, Vector &x)
|
||||
{
|
||||
const auto P = get_prolongation(field);
|
||||
x.SetSize(P->Width());
|
||||
P->MultTranspose(field_l, x);
|
||||
}
|
||||
|
||||
/// @brief Apply the prolongation operator to a vector of fields.
|
||||
///
|
||||
/// x is a long vector containing the data for all fields on tdofs and
|
||||
@@ -1287,7 +1107,6 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
const Vector &x,
|
||||
std::array<Vector, M> &fields_l)
|
||||
{
|
||||
NVTX("P");
|
||||
int data_offset = 0;
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
@@ -1295,14 +1114,9 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
const int width = P->Width();
|
||||
// const Vector x_i(x.GetData() + data_offset, width);
|
||||
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
fields_l[i].SetSize(P->Height());
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("P->Mult");
|
||||
P->Mult(x_i, fields_l[i]);
|
||||
NVTX_END("P->Mult");
|
||||
data_offset += width;
|
||||
}
|
||||
}
|
||||
@@ -1316,259 +1130,20 @@ void prolongation(const std::array<FieldDescriptor, N> fields,
|
||||
/// @param fields the array of field descriptors.
|
||||
/// @param x the input vector in tdofs.
|
||||
/// @param fields_l the array of output vectors in vdofs.
|
||||
// inline
|
||||
// void prolongation(const std::vector<FieldDescriptor> fields,
|
||||
// const Vector &x,
|
||||
// std::vector<Vector> &fields_l)
|
||||
// {
|
||||
// int data_offset = 0;
|
||||
// for (std::size_t i = 0; i < fields.size(); i++)
|
||||
// {
|
||||
// const auto P = get_prolongation(fields[i]);
|
||||
// const int width = P->Width();
|
||||
// const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
// fields_l[i].SetSize(P->Height());
|
||||
// P->Mult(x_i, fields_l[i]);
|
||||
// data_offset += width;
|
||||
// }
|
||||
// }
|
||||
|
||||
inline
|
||||
void prolongation(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const BlockVector &x,
|
||||
std::vector<Vector *> &x_l)
|
||||
void prolongation(const std::vector<FieldDescriptor> fields,
|
||||
const Vector &x,
|
||||
std::vector<Vector> &fields_l)
|
||||
{
|
||||
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
|
||||
"error " << x.NumBlocks() << " vs " << x_l.size());
|
||||
for (int i = 0; i < x.NumBlocks(); i++)
|
||||
int data_offset = 0;
|
||||
for (std::size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
*x_l[i] = x.GetBlock(i);
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Width() << " vs " << x.GetBlock(i).Size());
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
P->Mult(x.GetBlock(i), *x_l[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const MultiVector &x,
|
||||
std::vector<Vector *> &x_l)
|
||||
{
|
||||
MFEM_ASSERT(x.NumBlocks() == static_cast<int>(x_l.size()),
|
||||
"error " << x.NumBlocks() << " vs " << x_l.size());
|
||||
for (int i = 0; i < x.NumBlocks(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
*x_l[i] = x[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
MFEM_ASSERT(P->Width() == x[i].Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Width() << " vs " << x[i].Size());
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
P->Mult(x[i], *x_l[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
BlockVector &x)
|
||||
{
|
||||
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
|
||||
"error " << x_l.size() << " vs " << x.NumBlocks());
|
||||
for (size_t i = 0; i < x_l.size(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
x.GetBlock(i) = *x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
MFEM_ASSERT(P->Width() == x.GetBlock(i).Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Width() << " vs " << x.GetBlock(i).Size());
|
||||
P->MultTranspose(*x_l[i], x.GetBlock(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline
|
||||
void prolongation_transpose(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
MultiVector &x)
|
||||
{
|
||||
MFEM_ASSERT(static_cast<int>(x_l.size()) == x.NumBlocks(),
|
||||
"error " << x_l.size() << " vs " << x.NumBlocks());
|
||||
for (size_t i = 0; i < x_l.size(); i++)
|
||||
{
|
||||
const auto P = get_prolongation(fields[i]);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (P == nullptr)
|
||||
{
|
||||
x[i] = *x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(P->Height() == x_l[i]->Size(),
|
||||
"prolongation not applicable to given input data size " <<
|
||||
P->Height() << " vs " << x_l[i]->Size());
|
||||
MFEM_ASSERT(P->Width() == x[i].Size(),
|
||||
"prolongation not applicable to given output data size " <<
|
||||
P->Width() << " vs " << x[i].Size());
|
||||
P->MultTranspose(*x_l[i], x[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void restriction(
|
||||
const std::vector<FieldDescriptor> fields,
|
||||
const std::vector<Vector *> &x_l,
|
||||
std::vector<Vector *> &x_e)
|
||||
{
|
||||
MFEM_ASSERT(x_l.size() == x_e.size(),
|
||||
"internal error " << x_l.size() << " vs " << x_e.size());
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
|
||||
// If nullptr, assume Identity.
|
||||
if (R == nullptr)
|
||||
{
|
||||
s = x_l[i]->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
s = R->Height();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (x_e[i] == nullptr)
|
||||
{
|
||||
x_e[i] = new Vector(s);
|
||||
}
|
||||
x_e[i]->SetSize(s);
|
||||
|
||||
if (R == nullptr)
|
||||
{
|
||||
x_e[i] = x_l[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(R->Width() == x_l[i]->Size(),
|
||||
"restriction not applicable to given input data size " <<
|
||||
R->Width() << " vs " << x_l[i]->Size());
|
||||
R->Mult(*x_l[i], *x_e[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void prepare_residual(
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
std::vector<Vector *> &r_e)
|
||||
{
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
if (std::holds_alternative<const QuadratureFunction *>(fields[i].data))
|
||||
{
|
||||
const auto fd = std::get<const QuadratureFunction *>(fields[i].data);
|
||||
s = fd->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
s = R->Height();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (r_e[i] == nullptr)
|
||||
{
|
||||
r_e[i] = new Vector(s);
|
||||
}
|
||||
else
|
||||
{
|
||||
r_e[i]->SetSize(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename entity_t>
|
||||
void restriction_transpose(
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
const std::vector<Vector *> &x_e,
|
||||
std::vector<Vector *> &x_l)
|
||||
{
|
||||
for (size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
int s = 0;
|
||||
const auto R = get_restriction<entity_t>(
|
||||
fields[i], ElementDofOrdering::LEXICOGRAPHIC);
|
||||
// TODO: if nullptr, assume Identity
|
||||
if (R == nullptr)
|
||||
{
|
||||
s = x_e[i]->Size();
|
||||
}
|
||||
else
|
||||
{
|
||||
s = R->Width();
|
||||
}
|
||||
|
||||
// TODO
|
||||
if (x_l[i] == nullptr)
|
||||
{
|
||||
x_l[i] = new Vector(s);
|
||||
}
|
||||
x_l[i]->SetSize(s);
|
||||
|
||||
// TODO: if nullptr, assume Identity
|
||||
if (R == nullptr)
|
||||
{
|
||||
x_l[i] = x_e[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
R->MultTranspose(*x_e[i], *x_l[i]);
|
||||
}
|
||||
const int width = P->Width();
|
||||
const Vector x_i(const_cast<Vector&>(x), data_offset, width);
|
||||
fields_l[i].SetSize(P->Height());
|
||||
P->Mult(x_i, fields_l[i]);
|
||||
data_offset += width;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1577,7 +1152,6 @@ void get_lvectors(const std::vector<FieldDescriptor> fields,
|
||||
const Vector &x,
|
||||
std::vector<Vector> &fields_l)
|
||||
{
|
||||
NVTX("get_lvectors");
|
||||
int data_offset = 0;
|
||||
for (std::size_t i = 0; i < fields.size(); i++)
|
||||
{
|
||||
@@ -1604,15 +1178,13 @@ template <typename fop_t>
|
||||
inline
|
||||
std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
const FieldDescriptor &f,
|
||||
[[maybe_unused]] const fop_t &fop,
|
||||
const fop_t &fop,
|
||||
MPI_Comm mpi_comm)
|
||||
{
|
||||
NVTX("get P^T");
|
||||
if constexpr (is_sum_fop<fop_t>::value)
|
||||
{
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T sum");
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
real_t local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM, mpi_comm);
|
||||
@@ -1623,7 +1195,6 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
{
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T Identity");
|
||||
y = r_local;
|
||||
};
|
||||
return PT;
|
||||
@@ -1631,7 +1202,6 @@ std::function<void(const Vector&, Vector&)> get_prolongation_transpose(
|
||||
const Operator *P = get_prolongation(f);
|
||||
auto PT = [=](const Vector &r_local, Vector &y)
|
||||
{
|
||||
NVTX("P^T");
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
return PT;
|
||||
@@ -1650,19 +1220,12 @@ void restriction(const FieldDescriptor u,
|
||||
Vector &field_e,
|
||||
ElementDofOrdering ordering)
|
||||
{
|
||||
NVTX("R");
|
||||
const auto R = get_restriction<entity_t>(u, ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l.Size(),
|
||||
"restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
field_e.SetSize(height);
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("R->Mult");
|
||||
R->Mult(u_l, field_e);
|
||||
NVTX_END("R->Mult");
|
||||
}
|
||||
|
||||
/// @brief Apply the restriction operator to a vector of fields.
|
||||
@@ -1680,29 +1243,14 @@ void restriction(const std::vector<FieldDescriptor> u,
|
||||
ElementDofOrdering ordering,
|
||||
const int offset = 0)
|
||||
{
|
||||
NVTX("R");
|
||||
for (std::size_t i = 0; i < u.size(); i++)
|
||||
{
|
||||
const auto R = get_restriction<entity_t>(u[i], ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l[i].Size(),
|
||||
"restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
// NVTX_INI("SetSize");
|
||||
fields_e[i + offset].SetSize(height);
|
||||
// NVTX_END("SetSize");
|
||||
|
||||
// NVTX_INI("R->Mult");
|
||||
if (dynamic_cast<const IdentityOperator*>(R))
|
||||
{
|
||||
NVTX("Identity");
|
||||
fields_e[i + offset].NewMemoryAndSize(u_l[i].GetMemory(), u_l[i].Size(), false);
|
||||
}
|
||||
else
|
||||
{
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
}
|
||||
// NVTX_END("R->Mult");
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1714,21 +1262,14 @@ void element_restriction(const std::array<FieldDescriptor, N> u,
|
||||
ElementDofOrdering ordering,
|
||||
const int offset = 0)
|
||||
{
|
||||
NVTX("ER");
|
||||
for (int i = 0; i < N; i++)
|
||||
{
|
||||
const auto R = get_element_restriction(u[i], ordering);
|
||||
MFEM_ASSERT(R->Width() == u_l[i].Size(),
|
||||
"element restriction not applicable to given data size");
|
||||
const int height = R->Height();
|
||||
|
||||
NVTX_INI("SetSize");
|
||||
fields_e[i + offset].SetSize(height);
|
||||
NVTX_END("SetSize");
|
||||
|
||||
NVTX_INI("R->Mult");
|
||||
R->Mult(u_l[i], fields_e[i + offset]);
|
||||
NVTX_END("R->Mult");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1785,10 +1326,6 @@ const DofToQuad *GetDofToQuad(const FieldDescriptor &f,
|
||||
return &arg->GetTypicalTraceElement()->GetDofToQuad(ir, mode);
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const QuadratureFunction *>)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
|
||||
{
|
||||
return &arg->GetDofToQuad();
|
||||
@@ -1920,7 +1457,7 @@ create_descriptors_to_fields_map(
|
||||
|
||||
auto f = [&](auto &fop, auto &map)
|
||||
{
|
||||
if constexpr (is_weight_fop<std::decay_t<decltype(fop)>>::value)
|
||||
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
|
||||
{
|
||||
// TODO-bug: stealing dimension from the first field
|
||||
fop.dim = GetDimension<entity_t>(fields[0]);
|
||||
@@ -2050,7 +1587,7 @@ get_shmem_info(
|
||||
const std::array<DofToQuadMap, num_outputs> &output_dtq_maps,
|
||||
const std::vector<FieldDescriptor> &fields,
|
||||
const int &num_entities,
|
||||
[[maybe_unused]] const input_t &inputs,
|
||||
const input_t &inputs,
|
||||
const int &num_qp,
|
||||
const std::vector<int> &input_size_on_qp,
|
||||
const int &residual_size_on_qp,
|
||||
@@ -2805,25 +2342,5 @@ std::array<DofToQuadMap, num_fields> create_dtq_maps(
|
||||
std::make_index_sequence<num_fields> {});
|
||||
}
|
||||
|
||||
struct QLayoutEntry
|
||||
{
|
||||
std::type_index type;
|
||||
std::vector<int> layout;
|
||||
|
||||
template <class Fop>
|
||||
QLayoutEntry(Fop, std::initializer_list<int> idx) :
|
||||
type(typeid(Fop)), layout(idx) {}
|
||||
};
|
||||
|
||||
static void ExtractQLayouts(
|
||||
const std::initializer_list<QLayoutEntry> entries,
|
||||
std::unordered_map<std::type_index, std::vector<int>>& out)
|
||||
{
|
||||
for (const auto& e : entries)
|
||||
{
|
||||
out[e.type] = e.layout;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::future
|
||||
#endif
|
||||
|
||||
@@ -57,7 +57,7 @@ void DGMassApply(const int e,
|
||||
}
|
||||
else if (DIM == 3)
|
||||
{
|
||||
SmemPAMassApply3D_Element<TD1D,TQ1D,ACCUM>(e, NE, B, pa_data, x, y);
|
||||
SmemPAMassApply3D_Element<TD1D,TQ1D,NBZ,ACCUM>(e, NE, B, pa_data, x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
+42
-1
@@ -167,7 +167,15 @@ public:
|
||||
/** @brief Full multidimensional representation which does not use tensor
|
||||
product structure. The ordering of the degrees of freedom is the
|
||||
same as TENSOR, but the sizes of B and G are the same as FULL.*/
|
||||
LEXICOGRAPHIC_FULL
|
||||
LEXICOGRAPHIC_FULL,
|
||||
|
||||
/** @brief Ragged tensor product representation using 1D matrices/tensors
|
||||
with dimensions using 1D number of quadrature points and ragged tensor degrees of
|
||||
freedom. */
|
||||
/** Used only for partial assembly of the H1 positive basis. The
|
||||
size of B is d1d x qnpt x dim. Since different Gauss-Jacobi quadrature rules
|
||||
are employed in each dimension, we need to store dim arrays. */
|
||||
RAGGED_TENSOR
|
||||
};
|
||||
|
||||
/// Describes the contents of the #B, #Bt, #G, and #Gt arrays, see #Mode.
|
||||
@@ -228,6 +236,39 @@ public:
|
||||
const Array<DofToQuad*> &dof2quad_array,
|
||||
const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode);
|
||||
|
||||
virtual ~DofToQuad() = default;
|
||||
};
|
||||
|
||||
/** @brief Structure representing the matrices/tensors needed to evaluate (in
|
||||
reference space) the values, gradients, divergences, or curls of a positive
|
||||
FiniteElement on simplices at the quadrature points of Stroud conical quadrature. */
|
||||
class RaggedDofToQuad : public DofToQuad
|
||||
{
|
||||
public:
|
||||
/** @brief Special basis function structures for positive (Bernstein) basis with
|
||||
partial assembly. The storage layout of Ba1 is ndof x nqpt for scalar elements.
|
||||
The storage layout of Ba2 is ndof x ndof x nqpt. In particular, we have
|
||||
Ba2(iqpt, a1, a2) = B^{p-a1}_{a2}(x_{iqpt}). */
|
||||
Array<real_t> Ba1, Ba2, Ba3;
|
||||
Array<real_t> Ba1t, Ba2t, Ba3t;
|
||||
|
||||
/** @brief Special structures for gradients of positive basis with partial assembly.
|
||||
The gradient arrays exploit properties of the Bernstein basis which allow grad(B^p_alpha)
|
||||
to be expressed as the sum of products of B^{p-1}_alpha and the barycentric coordinates.
|
||||
Thus, Ga1 and Ga2 simply contain the ragged tensor product components of B^{p-1}_alpha */
|
||||
Array<real_t> Ga1, Ga2, Ga3;
|
||||
Array<real_t> Ga1t, Ga2t, Ga3t;
|
||||
|
||||
/** @brief Mapping from the Bernstein multi-index (a_1, ..., a_d) to the lexicographic
|
||||
dof index. */
|
||||
Array<int> lex_map;
|
||||
|
||||
Array<int> forward_map2d_diff, forward_map3d_diff;
|
||||
Array<int> inverse_map2d_diff, inverse_map3d_diff;
|
||||
|
||||
Array<int> forward_map2d_mass, forward_map3d_mass;
|
||||
Array<int> inverse_map2d_mass, inverse_map3d_mass;
|
||||
};
|
||||
|
||||
/// Describes the function space on each element
|
||||
|
||||
@@ -557,6 +557,101 @@ H1Pos_TriangleElement::H1Pos_TriangleElement(const int p)
|
||||
}
|
||||
}
|
||||
|
||||
const DofToQuad &H1Pos_TriangleElement::GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array)
|
||||
{
|
||||
DofToQuad *d2q = nullptr;
|
||||
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
}
|
||||
if (!d2q)
|
||||
{
|
||||
d2q = new RaggedDofToQuad;
|
||||
const int ndof = fe.GetOrder() + 1; // verify
|
||||
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
|
||||
d2q->FE = &fe;
|
||||
d2q->IntRule = &ir;
|
||||
d2q->mode = mode;
|
||||
d2q->ndof = ndof;
|
||||
d2q->nqpt = nqpt;
|
||||
|
||||
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
|
||||
rd2q->Ba1.SetSize(nqpt*ndof);
|
||||
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba2.SetSize((int)nqpt*ndof*ndof);
|
||||
rd2q->Ba1t.SetSize(nqpt*ndof);
|
||||
rd2q->Ba2t.SetSize((int)nqpt*ndof*ndof);
|
||||
// stores first component of ragged tensor basis with order p-1, for gradients only
|
||||
rd2q->Ga1.SetSize(nqpt*(ndof -1));
|
||||
// stores second component of ragged tensor basis with order p-1
|
||||
rd2q->Ga2.SetSize(nqpt*(ndof-1)*(ndof -1));
|
||||
rd2q->Ga1t.SetSize(nqpt*(ndof -1));
|
||||
rd2q->Ga2t.SetSize(nqpt*(ndof-1)*(ndof -1));
|
||||
rd2q->lex_map.SetSize(ndof * ndof);
|
||||
Vector shape_a1(ndof), shape_a2(ndof * ndof);
|
||||
Vector shape_Ga1(ndof-1), shape_Ga2((ndof-1) * (ndof-1));
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
|
||||
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
|
||||
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
|
||||
// Gauss-Jacobi rule). Additionally, the Bernstein PA algorithms expect evaluation of the
|
||||
// component 1D bases at the Stroud nodes pulled back to the unit square, so perform the pullback
|
||||
// on the fly.
|
||||
const real_t x = ir.IntPoint(i).x;
|
||||
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
|
||||
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
|
||||
if (j < ndof-1)
|
||||
{
|
||||
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
|
||||
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
|
||||
for (int k = 0; k < ndof-j; k++)
|
||||
{
|
||||
rd2q->Ba2t[i + nqpt*(j + ndof*k)] = rd2q->Ba2[k + ndof*(j + ndof*i)] = shape_a2(
|
||||
k);
|
||||
if (j < ndof-1 && k < ndof-j-1)
|
||||
{
|
||||
rd2q->Ga2t[i + nqpt*(j + (ndof-1)*k)] = rd2q->Ga2[k + (ndof-1)*(j +
|
||||
(ndof-1)*i)] = shape_Ga2(k);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stores the mapping from 2D Bernstein multi-index (i,j,p-i-j) to the
|
||||
// lexicographic DOF ordering
|
||||
for (int i = 0; i < ndof; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i; j++)
|
||||
{
|
||||
int idx = ((2 * (ndof-1) + 3) - j) * j / 2 + i;
|
||||
rd2q->lex_map[j + ndof*i] = idx;
|
||||
}
|
||||
}
|
||||
dof2quad_array.Append(d2q);
|
||||
}
|
||||
}
|
||||
return *d2q;
|
||||
}
|
||||
|
||||
// static method
|
||||
void H1Pos_TriangleElement::CalcShape(
|
||||
const int p, const real_t l1, const real_t l2, real_t *shape)
|
||||
@@ -749,6 +844,213 @@ H1Pos_TetrahedronElement::H1Pos_TetrahedronElement(const int p)
|
||||
}
|
||||
}
|
||||
|
||||
const DofToQuad &H1Pos_TetrahedronElement::GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array)
|
||||
{
|
||||
DofToQuad *d2q = nullptr;
|
||||
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
}
|
||||
if (!d2q)
|
||||
{
|
||||
d2q = new RaggedDofToQuad;
|
||||
const int ndof = fe.GetOrder() + 1; // verify
|
||||
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
|
||||
const int basis_dim2d = ndof*(ndof+1) / 2;
|
||||
const int basis_dim3d = ndof*(ndof+1)*(ndof+2) / 6;
|
||||
const int basis_dim2d_diff = (ndof-1)*(ndof) / 2;
|
||||
const int basis_dim3d_diff = (ndof-1)*(ndof)*(ndof+1) / 6;
|
||||
d2q->FE = &fe;
|
||||
d2q->IntRule = &ir;
|
||||
d2q->mode = mode;
|
||||
d2q->ndof = ndof;
|
||||
d2q->nqpt = nqpt;
|
||||
|
||||
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
|
||||
rd2q->Ba1.SetSize(nqpt * ndof);
|
||||
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba2.SetSize(nqpt * basis_dim2d);
|
||||
// third component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba3.SetSize(nqpt * basis_dim3d);
|
||||
rd2q->Ba1t.SetSize(nqpt * ndof);
|
||||
rd2q->Ba2t.SetSize(nqpt * basis_dim2d);
|
||||
rd2q->Ba3t.SetSize(nqpt * basis_dim3d);
|
||||
// stores first component of ragged tensor basis with order p-1, for gradients only
|
||||
rd2q->Ga1.SetSize(nqpt * (ndof-1));
|
||||
// stores second component of ragged tensor basis with order p-1
|
||||
rd2q->Ga2.SetSize(nqpt * basis_dim2d_diff);
|
||||
// stores third component of ragged tensor basis with order p-1
|
||||
rd2q->Ga3.SetSize(nqpt * basis_dim3d_diff);
|
||||
rd2q->Ga1t.SetSize(nqpt * (ndof-1));
|
||||
rd2q->Ga2t.SetSize(nqpt * basis_dim2d_diff);
|
||||
rd2q->Ga3t.SetSize(nqpt * basis_dim3d_diff);
|
||||
rd2q->lex_map.SetSize(ndof * ndof * ndof);
|
||||
|
||||
rd2q->forward_map2d_diff.SetSize((ndof-1) * (ndof-1));
|
||||
rd2q->forward_map3d_diff.SetSize((ndof-1) * (ndof-1) * (ndof-1));
|
||||
rd2q->inverse_map2d_diff.SetSize(2 * basis_dim2d_diff);
|
||||
rd2q->inverse_map3d_diff.SetSize(3 * basis_dim3d_diff);
|
||||
|
||||
rd2q->forward_map2d_mass.SetSize(ndof * ndof);
|
||||
rd2q->forward_map3d_mass.SetSize(ndof * ndof * ndof);
|
||||
rd2q->inverse_map2d_mass.SetSize(2 * basis_dim2d);
|
||||
rd2q->inverse_map3d_mass.SetSize(2 * basis_dim3d);
|
||||
|
||||
// forward and inverse maps for multi-index to collpased 1d index for diffusion, can combine
|
||||
// these four loops, but need four idx's and clause for shorter diff loops
|
||||
int idx = 0;
|
||||
for (int i = 0; i < ndof-1; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i-1; j++)
|
||||
{
|
||||
rd2q->forward_map2d_diff[j + (ndof-1)*i] = idx;
|
||||
rd2q->inverse_map2d_diff[2*idx] = i;
|
||||
rd2q->inverse_map2d_diff[1 + 2*idx] = j;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
|
||||
idx = 0;
|
||||
for (int k = 0; k < ndof-1; k++)
|
||||
{
|
||||
for (int j = 0; j < ndof-k-1; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-k-j-1; i++)
|
||||
{
|
||||
rd2q->forward_map3d_diff[k + (ndof-1)*(j + (ndof-1)*i)] = idx;
|
||||
rd2q->inverse_map3d_diff[3*idx] = i;
|
||||
rd2q->inverse_map3d_diff[1 + 3*idx] = j;
|
||||
rd2q->inverse_map3d_diff[2 + 3*idx] = k;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// forward and inverse maps for multi-index to collpased 1d index for mass
|
||||
idx = 0;
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-j; i++)
|
||||
{
|
||||
rd2q->forward_map2d_mass[j + ndof*i] = idx;
|
||||
rd2q->inverse_map2d_mass[2*idx] = i;
|
||||
rd2q->inverse_map2d_mass[1 + 2*idx] = j;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
|
||||
idx = 0;
|
||||
for (int k = 0; k < ndof; k++)
|
||||
{
|
||||
for (int j = 0; j < ndof-k; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-k-j; i++)
|
||||
{
|
||||
rd2q->forward_map3d_mass[k + ndof*(j + ndof*i)] = idx;
|
||||
rd2q->inverse_map3d_mass[2*idx] = i;
|
||||
rd2q->inverse_map3d_mass[1 + 2*idx] = j;
|
||||
// d2q->inverse_map3d_mass[2 + 3*idx] = k;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector shape_a1(ndof), shape_a2(ndof * ndof), shape_a3(ndof * ndof * ndof);
|
||||
Vector shape_Ga1(ndof-1), shape_Ga2(ndof-1), shape_Ga3(ndof-1);
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
|
||||
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
|
||||
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
|
||||
// Gauss-Jacobi rule). The first 'nqpt' points in the third dimension have the same z-coordinates
|
||||
// as those of the 1D rule for the third dimension (i.e. Gauss-Legendre rule). Additionally,
|
||||
// the Bernstein PA algorithms expect evaluation of the component 1D bases at the Stroud nodes
|
||||
// pulled back to the unit cube, so perform the pullback on the fly.
|
||||
const real_t x = ir.IntPoint(i).x;
|
||||
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
|
||||
const real_t z = ir.IntPoint(nqpt*nqpt*i).z / (1.0 - ir.IntPoint(
|
||||
nqpt*nqpt*i).x - ir.IntPoint(nqpt*nqpt*i).y);
|
||||
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
|
||||
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
|
||||
if (j < ndof-1)
|
||||
{
|
||||
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
|
||||
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
|
||||
for (int k = 0; k < ndof-j; k++)
|
||||
{
|
||||
const int a_2d_mass = rd2q->forward_map2d_mass[k + ndof*j];
|
||||
rd2q->Ba2t[i + nqpt*a_2d_mass] = rd2q->Ba2[a_2d_mass + basis_dim2d*i] =
|
||||
shape_a2(
|
||||
k);
|
||||
if (j < ndof-1 && k < ndof-j-1)
|
||||
{
|
||||
const int a_2d_diff = rd2q->forward_map2d_diff[k + (ndof-1)*j];
|
||||
rd2q->Ga2t[i + nqpt*a_2d_diff] = rd2q->Ga2[a_2d_diff + basis_dim2d_diff*i] =
|
||||
shape_Ga2(k);
|
||||
Poly_1D::CalcBernstein(ndof-2-j-k, z, shape_Ga3);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j-k, z, shape_a3);
|
||||
for (int m = 0; m < ndof-j-k; m++)
|
||||
{
|
||||
const int a_3d_mass = rd2q->forward_map3d_mass[m + ndof*(k + ndof*j)];
|
||||
rd2q->Ba3t[i + nqpt*a_3d_mass] = rd2q->Ba3[a_3d_mass + basis_dim3d*i] =
|
||||
shape_a3(
|
||||
m);
|
||||
if (j < ndof-1 && k < ndof-j-1 && m < ndof-j-k-1)
|
||||
{
|
||||
// // collapsed 1D access
|
||||
// d2q->Ga3[i + nqpt*(m + d2q->offset3d[k + (ndof-1)*j])] = shape_Ga3(m);
|
||||
// collapsed 1D access with forward mapping
|
||||
const int a_3d_diff = rd2q->forward_map3d_diff[m + (ndof-1)*(k + (ndof-1)*j)];
|
||||
rd2q->Ga3t[i + nqpt*a_3d_diff] = rd2q->Ga3[a_3d_diff + basis_dim3d_diff*i] =
|
||||
shape_Ga3(m);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stores the mapping from 3D Bernstein multi-index (i,j,k,p-i-j-k) to the
|
||||
// lexicographic DOF ordering
|
||||
int p = ndof - 1;
|
||||
for (int i = 0; i < ndof; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i; j++)
|
||||
{
|
||||
for (int k = 0; k < ndof-i-j; k++)
|
||||
{
|
||||
int dof = (p+1)*(p+2)*(p+3) / 6;
|
||||
int tet = (p-k)*(p-k+1)*(p-k+2) / 6;
|
||||
int tri = (p+1-k-j)*(p+2-k-j)/2;
|
||||
int multi_idx = dof - tet - tri + i;
|
||||
rd2q->lex_map[k + ndof*(j + ndof*i)] = multi_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dof2quad_array.Append(d2q);
|
||||
}
|
||||
}
|
||||
return *d2q;
|
||||
}
|
||||
|
||||
// static method
|
||||
void H1Pos_TetrahedronElement::CalcShape(
|
||||
const int p, const real_t l1, const real_t l2, const real_t l3,
|
||||
|
||||
@@ -191,6 +191,21 @@ public:
|
||||
/// Construct the H1Pos_TriangleElement of order @a p
|
||||
H1Pos_TriangleElement(const int p);
|
||||
|
||||
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const override
|
||||
{
|
||||
return (mode == DofToQuad::RAGGED_TENSOR) ?
|
||||
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
|
||||
FiniteElement::GetDofToQuad(ir, mode);
|
||||
}
|
||||
|
||||
static const DofToQuad &GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array);
|
||||
|
||||
const Array<int> &GetDofMap() const { return dof_map; }
|
||||
|
||||
// The size of shape is (p+1)(p+2)/2 (dof).
|
||||
static void CalcShape(const int p, const real_t x, const real_t y,
|
||||
real_t *shape);
|
||||
@@ -220,6 +235,21 @@ public:
|
||||
/// Construct the H1Pos_TetrahedronElement of order @a p
|
||||
H1Pos_TetrahedronElement(const int p);
|
||||
|
||||
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const override
|
||||
{
|
||||
return (mode == DofToQuad::RAGGED_TENSOR) ?
|
||||
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
|
||||
FiniteElement::GetDofToQuad(ir, mode);
|
||||
}
|
||||
|
||||
static const DofToQuad &GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array);
|
||||
|
||||
const Array<int> &GetDofMap() const { return dof_map; }
|
||||
|
||||
// The size of shape is (p+1)(p+2)(p+3)/6 (dof).
|
||||
static void CalcShape(const int p, const real_t x, const real_t y,
|
||||
const real_t z, real_t *shape);
|
||||
|
||||
@@ -250,6 +250,14 @@ public:
|
||||
its GetOrder() method. */
|
||||
virtual FiniteElementCollection *Clone(int p) const;
|
||||
|
||||
/** @brief Return the order parameter used to construct this collection.
|
||||
* This differs from GetOrder() depending on the collection type. */
|
||||
virtual int GetConstructorOrder() const
|
||||
{
|
||||
MFEM_ABORT("Collection " << Name() << " does not support GetConstructorOrder");
|
||||
return -1;
|
||||
}
|
||||
|
||||
protected:
|
||||
const int base_p; ///< Order as returned by GetOrder().
|
||||
|
||||
@@ -314,6 +322,9 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new H1_FECollection(p, dim, b_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p; }
|
||||
|
||||
virtual ~H1_FECollection();
|
||||
};
|
||||
|
||||
@@ -343,6 +354,10 @@ class H1_Trace_FECollection : public H1_FECollection
|
||||
public:
|
||||
H1_Trace_FECollection(const int p, const int dim,
|
||||
const int btype = BasisType::GaussLobatto);
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new H1_Trace_FECollection(p, dim+1, b_type); }
|
||||
|
||||
};
|
||||
|
||||
/// Arbitrary order "L2-conforming" discontinuous finite elements.
|
||||
@@ -396,6 +411,9 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new L2_FECollection(p, dim, b_type, m_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p; }
|
||||
|
||||
virtual ~L2_FECollection();
|
||||
};
|
||||
|
||||
@@ -456,6 +474,9 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new RT_FECollection(p, dim, cb_type, ob_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p-1; }
|
||||
|
||||
virtual ~RT_FECollection();
|
||||
};
|
||||
|
||||
@@ -536,6 +557,9 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new ND_FECollection(p, dim, cb_type, ob_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return dim>1 ? base_p : base_p+1; }
|
||||
|
||||
virtual ~ND_FECollection();
|
||||
};
|
||||
|
||||
@@ -548,6 +572,9 @@ public:
|
||||
ND_Trace_FECollection(const int p, const int dim,
|
||||
const int cb_type = BasisType::GaussLobatto,
|
||||
const int ob_type = BasisType::GaussLegendre);
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new ND_Trace_FECollection(p, dim+1, cb_type, ob_type); }
|
||||
};
|
||||
|
||||
/// Arbitrary order 3D H(curl)-conforming Nedelec finite elements in 1D.
|
||||
|
||||
+1
-1
@@ -52,7 +52,7 @@
|
||||
#include "bounds.hpp"
|
||||
#include "particleset.hpp"
|
||||
|
||||
// #include "dfem/doperator.hpp"
|
||||
#include "dfem/doperator.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include "pfespace.hpp"
|
||||
|
||||
+1
-2
@@ -4631,9 +4631,8 @@ FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
|
||||
|
||||
ElementDofOrdering GetEVectorOrdering(const FiniteElementSpace& fes)
|
||||
{
|
||||
return UsesTensorBasis(fes)?
|
||||
return (UsesTensorBasis(fes) || fes.UsesRaggedTensorBasis()) ?
|
||||
ElementDofOrdering::LEXICOGRAPHIC:
|
||||
ElementDofOrdering::NATIVE;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -1514,6 +1514,18 @@ public:
|
||||
return dynamic_cast<const L2_FECollection*>(fec) != NULL;
|
||||
}
|
||||
|
||||
/// @brief Return true if the mesh contains only one topology, the elements are
|
||||
/// all triangles or tetrahedrons, and the elements are ragged tensor elements
|
||||
/// i.e. Bernstein/positive basis.
|
||||
bool UsesRaggedTensorBasis() const
|
||||
{
|
||||
bool simplex = this->GetMesh()->IsSimplexMesh();
|
||||
bool positive =
|
||||
dynamic_cast<const mfem::H1Pos_TriangleElement *>(this->GetTypicalFE()) ||
|
||||
dynamic_cast<const mfem::H1Pos_TetrahedronElement *>(this->GetTypicalFE());
|
||||
return simplex && positive;
|
||||
}
|
||||
|
||||
/** In variable-order spaces on nonconforming (NC) meshes, this function
|
||||
controls whether strict conformity is enforced in cases where coarse
|
||||
edges/faces have higher polynomial order than their fine NC neighbors.
|
||||
|
||||
@@ -2256,6 +2256,104 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::AccumulateAndCountTraceValues(
|
||||
Coefficient *coeff[], VectorCoefficient *vcoeff,
|
||||
Array<int> &values_counter)
|
||||
{
|
||||
if (vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
|
||||
"vcoeff vdim != fes VDim");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetMapType() ==
|
||||
FiniteElement::VALUE &&
|
||||
fes->GetTypicalTraceElement()->GetRangeType() ==
|
||||
FiniteElement::SCALAR,
|
||||
"Can only call ProjectTraceCoefficient on scalar value-type "
|
||||
"trace elements. "
|
||||
"Use ProjectTraceCoefficientNormal for RT and "
|
||||
"ProjectTraceCoefficientTangent for ND finite elements.");
|
||||
}
|
||||
|
||||
Array<int> vdofs;
|
||||
Vector vc;
|
||||
|
||||
values_counter.SetSize(Size());
|
||||
values_counter = 0;
|
||||
|
||||
const int vdim = fes->GetVDim();
|
||||
HostReadWrite();
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
|
||||
const FiniteElement *fe = fes->GetFaceElement(i);
|
||||
const int fdof = fe->GetDof();
|
||||
ElementTransformation *transf = fes->GetMesh()->GetFaceTransformation(i);
|
||||
const IntegrationRule &ir = fe->GetNodes();
|
||||
fes->GetFaceVDofs(i, vdofs);
|
||||
|
||||
for (int j = 0; j < fdof; j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
transf->SetIntPoint(&ip);
|
||||
if (vcoeff) { vcoeff->Eval(vc, *transf, ip); }
|
||||
for (int d = 0; d < vdim; d++)
|
||||
{
|
||||
if (!vcoeff && !coeff[d]) { continue; }
|
||||
|
||||
real_t val = vcoeff ? vc(d) : coeff[d]->Eval(*transf, ip);
|
||||
int ind = vdofs[fdof*d+j];
|
||||
if ( ind < 0 )
|
||||
{
|
||||
val = -val, ind = -1-ind;
|
||||
}
|
||||
if (++values_counter[ind] == 1)
|
||||
{
|
||||
(*this)(ind) = val;
|
||||
}
|
||||
else
|
||||
{
|
||||
(*this)(ind) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::AccumulateAndCountTraceTangentValues(
|
||||
VectorCoefficient &vcoeff, Array<int> &values_counter)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()
|
||||
->GetRangeType() == FiniteElement::VECTOR &&
|
||||
fes->GetTypicalTraceElement()
|
||||
->GetMapType() == FiniteElement::H_CURL,
|
||||
"Not an ND FE space!");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetPhysRangeDim(
|
||||
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
|
||||
"vcoeff vdim != PhysRangeDim");
|
||||
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
Vector lvec;
|
||||
|
||||
values_counter.SetSize(Size());
|
||||
values_counter = 0;
|
||||
|
||||
HostReadWrite();
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
fe = fes->GetFaceElement(i);
|
||||
T = fes->GetMesh()->GetFaceTransformation(i);
|
||||
fes->GetFaceVDofs(i, dofs);
|
||||
lvec.SetSize(fe->GetDof());
|
||||
fe->Project(vcoeff, *T, lvec);
|
||||
accumulate_dofs(dofs, lvec, *this, values_counter);
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ComputeMeans(AvgType type, Array<int> &zones_per_vdof)
|
||||
{
|
||||
switch (type)
|
||||
@@ -2698,6 +2796,74 @@ void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(Coefficient *coeff[])
|
||||
{
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceValues(coeff, NULL, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(Coefficient &coeff)
|
||||
{
|
||||
MFEM_VERIFY(FESpace()->GetVDim() == 1, "ProjectTraceCoefficient(Coefficient&)"
|
||||
"is only valid for scalar GridFunction");
|
||||
Coefficient *coeff_p = &coeff;
|
||||
ProjectTraceCoefficient(&coeff_p);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(VectorCoefficient &vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(FESpace()->GetVDim() == vcoeff.GetVDim(),
|
||||
"Incompatible vcoeff vdim and fes vdim");
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceValues(NULL, &vcoeff, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetRangeType() ==
|
||||
FiniteElement::SCALAR &&
|
||||
fes->GetTypicalTraceElement()->GetMapType() ==
|
||||
FiniteElement::INTEGRAL, "Not an RT FE space!");
|
||||
MFEM_VERIFY(vcoeff.GetVDim() == fes->GetMesh()->SpaceDimension(),
|
||||
"vcoeff vdim (" << vcoeff.GetVDim()
|
||||
<< ") != SpaceDimension ("
|
||||
<< fes->GetMesh()->SpaceDimension() << ")");
|
||||
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
int dim = vcoeff.GetVDim();
|
||||
Vector vc(dim), nor(dim), lvec;
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
fe = fes->GetFaceElement(i);
|
||||
T = fes->GetMesh()->GetFaceTransformation(i);
|
||||
const IntegrationRule &ir = fe->GetNodes();
|
||||
lvec.SetSize(fe->GetDof());
|
||||
for (int j = 0; j < ir.GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
T->SetIntPoint(&ip);
|
||||
vcoeff.Eval(vc, *T, ip);
|
||||
CalcOrtho(T->Jacobian(), nor);
|
||||
lvec(j) = (vc * nor);
|
||||
}
|
||||
fes->GetFaceVDofs(i, dofs);
|
||||
SetSubVector(dofs, lvec);
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff)
|
||||
{
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceTangentValues(vcoeff, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectCoefficientGlobalL2(VectorCoefficient &vcoeff,
|
||||
real_t rtol, int iter)
|
||||
{
|
||||
@@ -5286,6 +5452,7 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
|
||||
{
|
||||
int max_order = fes->GetMaxElementOrder();
|
||||
PLBound plb(fes, ref_factor*(max_order+1));
|
||||
|
||||
Vector lel, uel;
|
||||
GetElementBounds(plb, lel, uel, vdim);
|
||||
|
||||
|
||||
+27
-3
@@ -578,6 +578,13 @@ protected:
|
||||
const Array<int> &bdr_attr,
|
||||
Array<int> &values_counter);
|
||||
|
||||
void AccumulateAndCountTraceValues(Coefficient *coeff[],
|
||||
VectorCoefficient *vcoeff,
|
||||
Array<int> &values_counter);
|
||||
|
||||
void AccumulateAndCountTraceTangentValues(VectorCoefficient &vcoeff,
|
||||
Array<int> &values_counter);
|
||||
|
||||
// Complete the computation of averages; called e.g. after
|
||||
// AccumulateAndCountZones().
|
||||
void ComputeMeans(AvgType type, Array<int> &zones_per_vdof);
|
||||
@@ -663,6 +670,23 @@ public:
|
||||
ProjectBdrCoefficient(&coeff_p, attr);
|
||||
}
|
||||
|
||||
/// Project a Coefficient on a GridFunction defined on H1 trace space
|
||||
void ProjectTraceCoefficient(Coefficient *coeff[]);
|
||||
void ProjectTraceCoefficient(Coefficient &coeff);
|
||||
|
||||
/** @brief Project a VectorCoefficient @a vcoeff on a GridFunction
|
||||
defined on a Vector H1 trace space. Note that this also works
|
||||
for a scalar H1 trace space, where only the first component of
|
||||
@a vcoeff is used. */
|
||||
void ProjectTraceCoefficient(VectorCoefficient &vcoeff);
|
||||
/** @brief Project a VectorCoefficient on a GridFunction
|
||||
defined on an RT trace space */
|
||||
void ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff);
|
||||
/** @brief Project a VectorCoefficient on a GridFunction
|
||||
defined on an ND trace space */
|
||||
void ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff);
|
||||
|
||||
|
||||
/** @brief Project a VectorCoefficient on the GridFunction, modifying only
|
||||
DOFs on the boundary associated with the boundary attributes marked in
|
||||
the @a attr array. */
|
||||
@@ -1767,8 +1791,8 @@ public:
|
||||
const int ref_factor=1, const int vdim=-1) const;
|
||||
|
||||
/// Computes the \ref PLBound for the gridfunction with number of control
|
||||
/// points based on \p ref_factor, and returns the bounds for each element
|
||||
/// ordered byNodes:
|
||||
/// points based on @a ref_factor, and returns the bounds for each element
|
||||
/// ordered byNODES:
|
||||
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
|
||||
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}. We also return the
|
||||
/// PLBound object used to compute the bounds.
|
||||
@@ -1802,7 +1826,7 @@ public:
|
||||
const int vdim = -1) const;
|
||||
|
||||
/// Compute bounds on the grid function for all the elements. The bounds
|
||||
/// are returned in @b lower and @b upper, ordered byNodes:
|
||||
/// are returned in @b lower and @b upper, ordered byNODES:
|
||||
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
|
||||
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
|
||||
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
|
||||
|
||||
+2390
-158
File diff suppressed because it is too large
Load Diff
+343
-67
@@ -21,6 +21,45 @@
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
/* gslib license and copyright statement for code adapted from gslib:
|
||||
|
||||
Copyright (c) 2008-2024, UCHICAGO ARGONNE, LLC.
|
||||
|
||||
The UChicago Argonne, LLC as Operator of Argonne National
|
||||
Laboratory holds copyright in the Software. The copyright holder
|
||||
reserves all rights except those expressly granted to licensees,
|
||||
and U.S. Government license rights.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions
|
||||
are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the disclaimer below.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the disclaimer (as noted below)
|
||||
in the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
|
||||
3. Neither the name of ANL nor the names of its contributors
|
||||
may be used to endorse or promote products derived from this software
|
||||
without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
|
||||
FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL
|
||||
UCHICAGO ARGONNE, LLC, THE U.S. DEPARTMENT OF
|
||||
ENERGY OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED
|
||||
TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
namespace gslib
|
||||
{
|
||||
struct comm;
|
||||
@@ -86,7 +125,7 @@ protected:
|
||||
void *fdataD;
|
||||
struct gslib::crystal *cr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm; // gslib's internal data
|
||||
int dim, points_cnt; // mesh dimension and number of points
|
||||
int dim, spacedim, points_cnt; // mesh dimension and number of points
|
||||
Array<unsigned int> gsl_code, gsl_proc, gsl_elem, gsl_mfem_elem;
|
||||
Vector gsl_mesh, gsl_ref, gsl_dist, gsl_mfem_ref;
|
||||
Array<unsigned int> recv_proc, recv_index; // data for custom interpolation
|
||||
@@ -104,18 +143,23 @@ protected:
|
||||
bool gpu_to_cpu_fallback = false;
|
||||
|
||||
// Device specific data used for FindPoints
|
||||
struct
|
||||
struct DEV_STRUCT
|
||||
{
|
||||
bool setup_device = false;
|
||||
bool find_device = false;
|
||||
int local_hash_size, dof1d, dof1d_sol, h_o_size, h_nx;
|
||||
int local_hash_size, dof1d, dof1d_sol, lh_nx, gh_nx;
|
||||
double newt_tol; // Tolerance specified during setup for Newton solve
|
||||
struct gslib::crystal *cr;
|
||||
struct gslib::hash_data_3 *hash3;
|
||||
struct gslib::hash_data_2 *hash2;
|
||||
mutable Vector bb, wtend, gll1d, lagcoeff, gll1d_sol, lagcoeff_sol;
|
||||
mutable Array<unsigned int> loc_hash_offset;
|
||||
mutable Vector loc_hash_min, loc_hash_fac;
|
||||
mutable Array<unsigned int> lh_offset, gh_offset;
|
||||
mutable Vector lh_min, lh_fac, gh_min, gh_fac;
|
||||
// Tolerance to mark points found on the surface as CODE_INTERNAL
|
||||
// or CODE_BORDER. This is needed because we cannot only use reference
|
||||
// space coordinates to determine if a point is located inside the
|
||||
// element or not.
|
||||
mutable double surf_dist_tol;
|
||||
} DEV;
|
||||
|
||||
/// Use GSLIB for communication and interpolation
|
||||
@@ -127,80 +171,143 @@ protected:
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/// Since GSLIB is designed to work with quads/hexes, we split every
|
||||
/// triangle/tet/prism/pyramid element into quads/hexes.
|
||||
/** @brief Since GSLIB is designed to work with quads/hexes, we split every
|
||||
* triangle/tet/prism/pyramid element into quads/hexes. */
|
||||
virtual void SetupSplitMeshes();
|
||||
|
||||
/// Setup integration points that will be used to interpolate the nodal
|
||||
/// location at points expected by GSLIB.
|
||||
/** @brief Setup integration points that will be used to interpolate the
|
||||
* nodal location at points expected by GSLIB. */
|
||||
virtual void SetupIntegrationRuleForSplitMesh(Mesh *mesh,
|
||||
IntegrationRule *irule,
|
||||
int order);
|
||||
|
||||
/// Helper function that calls \ref SetupSplitMeshes and
|
||||
/// \ref SetupIntegrationRuleForSplitMesh.
|
||||
/** @brief Helper function that calls \ref SetupSplitMeshes and
|
||||
* \ref SetupIntegrationRuleForSplitMesh. */
|
||||
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
|
||||
|
||||
/// Get GridFunction value at the points expected by GSLIB.
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
|
||||
|
||||
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
|
||||
/// find the original element number (that was split into micro quads/hexes)
|
||||
/// during the setup phase.
|
||||
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
|
||||
* simplices, find the original element number (that was split into
|
||||
* micro quads/hexes) during the setup phase. */
|
||||
virtual void MapRefPosAndElemIndices();
|
||||
|
||||
// Device functions
|
||||
// FindPoints locally on device for 3D.
|
||||
/// FindPoints locally on device for 3D.
|
||||
void FindPointsLocal3(const Vector &point_pos, int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l, int npt);
|
||||
|
||||
// FindPoints locally on device for 2D.
|
||||
/// FindPoints locally on device for 2D.
|
||||
void FindPointsLocal2(const Vector &point_pos, int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l, int npt);
|
||||
|
||||
// Interpolate on device for 3D.
|
||||
/// FindPoints locally on device for 3D surface elements.
|
||||
void FindPointsSurfLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// FindPoints locally on device for 3D edge elements.
|
||||
void FindPointsEdgeLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// FindPoints locally on device for 2D edge elements.
|
||||
void FindPointsEdgeLocal2(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// Interpolate on device for 3D.
|
||||
void InterpolateLocal3(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1dsol);
|
||||
// Interpolate on device for 2D.
|
||||
int dof1dsol);
|
||||
|
||||
/// Interpolate on device for 2D.
|
||||
void InterpolateLocal2(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1dsol);
|
||||
int dof1dsol);
|
||||
|
||||
// Prepare data for device functions.
|
||||
/// Interpolate on device for 1D.
|
||||
void InterpolateLocal1(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp, int dof1dsol);
|
||||
|
||||
/// Prepare data for device execution for volume meshes.
|
||||
void SetupDevice();
|
||||
|
||||
/** Searches positions given in physical space by @a point_pos.
|
||||
/** @brief Searches positions given in physical space by @a point_pos.
|
||||
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
|
||||
void FindPointsOnDevice(const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/** Interpolation of field values at prescribed reference space positions.
|
||||
@param[in] field_in_evec E-vector of grid function to be interpolated.
|
||||
Assumed ordering is NDOFSxVDIMxNEL
|
||||
@param[in] nel Number of elements in the mesh.
|
||||
@param[in] ncomp Number of components in the field.
|
||||
@param[in] dof1dsol Number of degrees of freedom in each reference
|
||||
space direction.
|
||||
@param[in] ordering Ordering of the out field values: byNodes/byVDIM
|
||||
|
||||
@param[out] field_out Interpolated values. For points that are not found
|
||||
the value is set to #default_interp_value. */
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions.
|
||||
* @param[in] field_in_evec E-vector of grid function to be interpolated.
|
||||
* Assumed ordering is NDOFSxVDIMxNEL
|
||||
* @param[in] nel Number of elements in the mesh.
|
||||
* @param[in] ncomp Number of components in the field.
|
||||
* @param[in] dof1dsol Number of degrees of freedom in each reference
|
||||
* space direction.
|
||||
* @param[in] ordering Ordering of the out field values: byNodes/byVDIM
|
||||
*
|
||||
* @param[out] field_out Interpolated values. For points that are not
|
||||
* found the value is set to
|
||||
* #default_interp_value. */
|
||||
void InterpolateOnDevice(const Vector &field_in_evec, Vector &field_out,
|
||||
const int nel, const int ncomp,
|
||||
const int dof1dsol, const int ordering);
|
||||
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions for surface meshes. */
|
||||
void InterpolateSurfBase(const Vector &field_in, Vector &field_out,
|
||||
const int nel, const int ncomp,
|
||||
const int dof1dsol, const int field_out_ordering);
|
||||
|
||||
/// Preprocess 2D surface mesh needed for FindPoints.
|
||||
void findptsedge_setup_2(DEV_STRUCT &devs,
|
||||
const double *const elx[2],
|
||||
const unsigned n,
|
||||
const uint nel,
|
||||
const unsigned m,
|
||||
const double bbox_tol,
|
||||
const uint local_hash_size,
|
||||
const uint global_hash_size);
|
||||
|
||||
/// Preprocess 3D surface mesh needed for FindPoints.
|
||||
void findptssurf_setup_3(DEV_STRUCT &devs,
|
||||
const double *const elx[3],
|
||||
const unsigned n,
|
||||
const uint nel,
|
||||
const unsigned m,
|
||||
const double bbox_tol,
|
||||
const uint local_hash_size,
|
||||
const uint global_hash_size,
|
||||
const int rD);
|
||||
|
||||
public:
|
||||
/// Serial constructor
|
||||
FindPointsGSLIB();
|
||||
@@ -224,8 +331,10 @@ public:
|
||||
FindPointsGSLIB(const FindPointsGSLIB&) = delete;
|
||||
FindPointsGSLIB& operator=(const FindPointsGSLIB&) = delete;
|
||||
|
||||
/** Initializes the internal mesh in gslib, by sending the positions of the
|
||||
Gauss-Lobatto nodes of the input Mesh object \p m.
|
||||
/** @brief Preprocess the internal mesh in gslib.
|
||||
|
||||
@details Initializes the internal mesh in gslib, by sending the
|
||||
positions of the Gauss-Lobatto nodes of the input Mesh object \p m.
|
||||
Note: not tested with periodic (L2).
|
||||
Note: the input mesh \p m must have Nodes set.
|
||||
|
||||
@@ -236,13 +345,22 @@ public:
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for simultaneous
|
||||
iteration. This alters performance and
|
||||
memory footprint.*/
|
||||
|
||||
memory footprint.
|
||||
*/
|
||||
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
/** Searches positions given in physical space by \p point_pos.
|
||||
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
|
||||
/// Preprocess the surface mesh to compute data for FindPoints.
|
||||
void SetupSurf(Mesh &m,
|
||||
const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** @brief Searches positions given in physical space by \p point_pos.
|
||||
|
||||
@details These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) specified by \p point_pos_ordering.
|
||||
|
||||
This function populates the following member variables:
|
||||
#gsl_code Return codes for each point: inside element (0),
|
||||
element boundary (1), not found (2).
|
||||
@@ -261,19 +379,34 @@ public:
|
||||
#gsl_dist Distance between the sought and the found point
|
||||
in physical space. */
|
||||
void FindPoints(const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Convenience function when point positions are in a ParticleVector
|
||||
void FindPoints(const ParticleVector &point_pos)
|
||||
{
|
||||
FindPoints(point_pos, point_pos.GetOrdering());
|
||||
}
|
||||
|
||||
/** @brief Searches positions given in physical space by \p point_pos on
|
||||
* surface mesh. */
|
||||
void FindPointsSurf(const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Convenience function when point positions are in a ParticleVector
|
||||
void FindPointsSurf(const ParticleVector &point_pos)
|
||||
{
|
||||
FindPointsSurf(point_pos, point_pos.GetOrdering());
|
||||
}
|
||||
|
||||
/// Setup FindPoints and search positions
|
||||
void FindPoints(Mesh &m, const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** Interpolation of field values at prescribed reference space positions.
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions.
|
||||
|
||||
@param[in] field_in Function values that will be interpolated on the
|
||||
reference positions. Note: it is assumed that
|
||||
\p field_in is in H1 and in the same space as the
|
||||
@@ -282,19 +415,36 @@ public:
|
||||
the value is set to #default_interp_value.
|
||||
The output ordering is determined from field_in.*/
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
|
||||
|
||||
/// Interpolation of field values, with output ordering specification.
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
|
||||
the output values in \p field_out corresponds to the ordering used
|
||||
in the input GridFunction \p field_in. */
|
||||
|
||||
/** @brief Same as Interpolate but for surface meshes */
|
||||
virtual void InterpolateSurf(const GridFunction &field_in,
|
||||
Vector &field_out);
|
||||
|
||||
/** @brief Same as Interpolate but for surface meshes with specified output
|
||||
ordering */
|
||||
virtual void InterpolateSurf(const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** @brief Search positions and interpolate.
|
||||
*
|
||||
* @details The ordering (byNODES or byVDIM) of the output values in
|
||||
* \p field_out corresponds to the ordering used in the input
|
||||
* GridFunction \p field_in.
|
||||
*/
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Search positions and interpolate with given point and output ordering.
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out, const int point_pos_ordering,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
|
||||
or byVDIM) of the output values in \p field_out corresponds to the
|
||||
ordering used in the input GridFunction \p field_in. */
|
||||
@@ -302,32 +452,36 @@ public:
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Average type to be used for L2 functions in-case a point is located at
|
||||
/// an element boundary where the function might be multi-valued.
|
||||
/** @brief Average type to be used for L2 functions in-case a point is
|
||||
* located at an element boundary where the function might be multi-valued.
|
||||
*/
|
||||
virtual void SetL2AvgType(AvgType avgtype_) { avgtype = avgtype_; }
|
||||
|
||||
/// Set the default interpolation value for points that are not found in the
|
||||
/// mesh.
|
||||
/** @brief Set the default interpolation value for points that are not found in the mesh. */
|
||||
virtual void SetDefaultInterpolationValue(double interp_value_)
|
||||
{
|
||||
default_interp_value = interp_value_;
|
||||
}
|
||||
|
||||
/// Set the tolerance for detecting points outside the 'curvilinear' boundary
|
||||
/// that gslib may return as found on the boundary. Points found on boundary
|
||||
/// with distance greater than @ bdr_tol are marked as not found.
|
||||
/** @brief Tolerance for detecting points outside the 'curvilinear' boundary.
|
||||
*
|
||||
* @details When using FindPoints, gslib may return points as found on the
|
||||
* boundary even when they are slightly outside the domain. This tolerance
|
||||
* is used to filter such points based on the distance^2 value and mark them
|
||||
* as not found.*/
|
||||
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
|
||||
{
|
||||
bdr_tol = bdr_tol_;
|
||||
}
|
||||
|
||||
/// Enable/Disable use of CPU functions for GPU data if the gslib version
|
||||
/// is older.
|
||||
/** @brief Enable/Disable use of CPU functions for GPU data if the gslib
|
||||
* version is older. */
|
||||
virtual void SetGPUtoCPUFallback(bool mode) { gpu_to_cpu_fallback = mode; }
|
||||
|
||||
/** Cleans up memory allocated internally by gslib.
|
||||
Note that in parallel, this must be called before MPI_Finalize(), as it
|
||||
calls MPI_Comm_free() for internal gslib communicators. FreeData is
|
||||
/** @brief Cleans up memory allocated internally by gslib.
|
||||
|
||||
@details Note that in parallel, this must be called before MPI_Finalize,
|
||||
as it calls MPI_Comm_free() for internal gslib communicators. FreeData is
|
||||
also called by the class destructor and there are no memory leaks if the
|
||||
destructor is called before MPI_Finalize(). If the destructor is called
|
||||
after MPI_Finalize(), there will be an error because gslib will try to
|
||||
@@ -335,8 +489,8 @@ public:
|
||||
*/
|
||||
virtual void FreeData();
|
||||
|
||||
/// Return code for each point searched by FindPoints: inside element (0), on
|
||||
/// element boundary (1), or not found (2).
|
||||
/** @brief Return code for each point searched by FindPoints:
|
||||
* inside element (0), element boundary (1), or not found (2). */
|
||||
virtual const Array<unsigned int> &GetCode() const { return gsl_code; }
|
||||
/// Return element number for each point found by FindPoints.
|
||||
virtual const Array<unsigned int> &GetElem() const { return gsl_mfem_elem; }
|
||||
@@ -344,15 +498,15 @@ public:
|
||||
virtual const Array<unsigned int> &GetProc() const { return gsl_proc; }
|
||||
/// Return reference coordinates for each point found by FindPoints.
|
||||
virtual const Vector &GetReferencePosition() const { return gsl_mfem_ref; }
|
||||
/// Return distance between the sought and the found point in physical space,
|
||||
/// for each point found by FindPoints.
|
||||
/// Return distance between the sought and the found point in physical space.
|
||||
virtual const Vector &GetDist() const { return gsl_dist; }
|
||||
|
||||
/// Return element number for each point found by FindPoints corresponding to
|
||||
/// GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with simplices.
|
||||
/** @brief Return element number for each point found by FindPoints
|
||||
* corresponding to GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with
|
||||
* simplices. */
|
||||
virtual const Array<unsigned int> &GetGSLIBElem() const { return gsl_elem; }
|
||||
/// Return reference coordinates in [-1,1] (internal range in GSLIB) for each
|
||||
/// point found by FindPoints.
|
||||
/** @brief Return reference coordinates in [-1,1] (internal range in GSLIB)
|
||||
* for each point found by FindPoints. */
|
||||
virtual const Vector &GetGSLIBReferencePosition() const { return gsl_ref; }
|
||||
|
||||
/// Get array of indices of not-found points.
|
||||
@@ -395,7 +549,7 @@ public:
|
||||
|
||||
/// Return the axis-aligned bounding boxes (AABB) computed during \ref Setup.
|
||||
/// The size of the returned vector is (nel x nverts x dim), where nel is the
|
||||
/// number of elements (after splitting for simplcies), nverts is number of
|
||||
/// number of elements (after splitting for simplicies), nverts is number of
|
||||
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
|
||||
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
|
||||
|
||||
@@ -409,6 +563,18 @@ public:
|
||||
/// \p obbV, a vector of size (nel x nverts x dim) .
|
||||
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
|
||||
Vector &obbV) const;
|
||||
|
||||
/** @brief Return the bounding boxes as a mesh on rank 0.
|
||||
*
|
||||
* @param[in] type Bounding-box type: 0 - AABB, 1 - OBB.
|
||||
*
|
||||
* @return On rank 0, returns a newly allocated mesh containing the
|
||||
* bounding boxes. The caller owns the returned pointer and is responsible
|
||||
* for deleting it. On other ranks, returns nullptr.
|
||||
*/
|
||||
Mesh *GetBoundingBoxMesh(int type);
|
||||
|
||||
virtual const Vector &GetGLLMesh() const { return gsl_mesh; }
|
||||
};
|
||||
|
||||
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
|
||||
@@ -536,6 +702,116 @@ public:
|
||||
void GS(Vector &senddata, GSOp op);
|
||||
};
|
||||
|
||||
#if defined(MFEM_USE_MPI)
|
||||
/** \brief Class to map a point in physical space to candidate ranks.
|
||||
*
|
||||
* This class builds a Cartesian-aligned tensor grid that covers the entire
|
||||
* domain and precomputes which ranks have elements intersecting each
|
||||
* grid cell. Given a point in physical space, the grid cell containing
|
||||
* the point is determined, and the list of candidate ranks whose
|
||||
* elements intersect that cell is returned. This yields a fast, conservative
|
||||
* point-to-rank candidate query. This is used internally by FindPointsGSLIB
|
||||
* to speed up point searches in parallel.
|
||||
*
|
||||
* See Mittal et al., "General Field Evaluation in High-Order Meshes on GPUs".
|
||||
* (2025). Computers & Fluids. for technical details.
|
||||
*
|
||||
*/
|
||||
class GlobalBBoxTensorGridMap
|
||||
{
|
||||
private:
|
||||
struct gslib::crystal *cr = nullptr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm = nullptr; // gslib's internal data
|
||||
int sdim, n_local_cells, num_procs;
|
||||
Array<int> gmap_n;
|
||||
Vector gmap_bnd_min, gmap_bnd_max;
|
||||
Vector gmap_fac;
|
||||
Array<int> ggrid_map;
|
||||
|
||||
void SetupCrystal(const MPI_Comm &comm);
|
||||
public:
|
||||
/// Constructor for a given mesh and number of tensor grid divisions
|
||||
GlobalBBoxTensorGridMap(ParMesh &pmesh, int nx);
|
||||
|
||||
/** @brief Constructor for given element bounds and spatial dimension.
|
||||
*
|
||||
* @details This constructor must be called collectively on \a comm.
|
||||
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
|
||||
*
|
||||
* Assumes elmin, elmax Ordering::byNodes:
|
||||
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
|
||||
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
|
||||
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
|
||||
*
|
||||
* When by_max_size=false, n gives the number of tensor-grid divisions in
|
||||
* each direction. When by_max_size=true, n is a per-rank size hint used to
|
||||
* derive a uniform global resolution. The communicator-wide sum of n is
|
||||
* converted to nx = ceil(pow(sum(n), 1./sdim)) in each direction, so n is
|
||||
* not a hard cap on ggrid_map.Size().
|
||||
*/
|
||||
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
|
||||
Vector &elmax, int nel, int sdim, int n,
|
||||
bool by_max_size);
|
||||
|
||||
/** @brief Constructor for given element bounds, spatial dimension, and
|
||||
* tensor-grid divisions in each direction.
|
||||
*
|
||||
* @details This constructor must be called collectively on \a comm.
|
||||
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
|
||||
* Requires nx.Size() == sdim and positive entries in nx.
|
||||
*
|
||||
* Assumes elmin, elmax Ordering::byNodes:
|
||||
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
|
||||
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
|
||||
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
|
||||
*/
|
||||
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
|
||||
Vector &elmax, int nel, int sdim, Array<int> &nx);
|
||||
|
||||
~GlobalBBoxTensorGridMap();
|
||||
|
||||
/** @brief Get list of procs corresponding to the list of points.
|
||||
*
|
||||
* @details This method must be called collectively on the communicator
|
||||
* used to construct the map. The input points can be ordered byNodes:
|
||||
* (XXX...,YYY...,ZZZ) or byVDIM: (XYZ,XYZ,...), as specified by
|
||||
* \a ordering.
|
||||
*
|
||||
* The output map contains one entry for each input point, keyed by the
|
||||
* point's local index in \a xyz. Points with no candidate ranks, including
|
||||
* points outside the global bounding box, have an empty list of candidate
|
||||
* ranks.
|
||||
*/
|
||||
void MapPointsToProcs(Vector &xyz, int ordering,
|
||||
std::map<int, std::vector<int>> &pt_to_procs) const;
|
||||
|
||||
// Some getters
|
||||
const Array<int> &GetGridMap() const { return ggrid_map; }
|
||||
const Vector &GetGridFac() const { return gmap_fac; }
|
||||
const Vector &GetGridMin() const { return gmap_bnd_min; }
|
||||
const Vector &GetGridMax() const { return gmap_bnd_max; }
|
||||
const Array<int> &GetGridN() const { return gmap_n; }
|
||||
|
||||
private:
|
||||
/// Setup the map given element bounds and number of tensor grid divisions.
|
||||
void Setup(const MPI_Comm &comm, Vector &elmin, Vector &elmax,
|
||||
int nel, Array<int> &nx);
|
||||
|
||||
/// Get global hash cell index for a given point.
|
||||
int GetGlobalGridCellFromPoint(Vector &xyz) const;
|
||||
|
||||
/** @brief Get owning proc and local index on that proc for given global
|
||||
* grid cell index. */
|
||||
void GlobalGridCellToProcAndLocalIndex(int i, int &proc, int &idx) const;
|
||||
|
||||
/// Map a point to proc and local index of the corresponding grid cell
|
||||
void GetProcAndLocalIndexFromPoint(Vector &xyz, int &proc, int &idx) const;
|
||||
|
||||
/// Given local cell index, return list of procs saved in the map
|
||||
Array<int> MapCellToProcs(int l_idx) const;
|
||||
};
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_GSLIB
|
||||
|
||||
@@ -562,7 +562,7 @@ newton_area_fin:
|
||||
int f = flags >> (2 * dd) & 3u;
|
||||
res->r[dd] = f == 0 ? r0[dd] + dr[dd] : (f == 1 ? -1 : 1);
|
||||
}
|
||||
res->flags = flags | (p->flags << 5);
|
||||
res->flags = flags | ((p->flags & FLAG_MASK) << 5);
|
||||
}
|
||||
|
||||
// Full Newton solve on the face. One of r/s/t is constrained.
|
||||
@@ -635,7 +635,8 @@ newton_edge_fin:
|
||||
res->r[de] = nr;
|
||||
res->r[dn]=p->r[dn];
|
||||
res->dist2p = -v;
|
||||
res->flags = flags | new_flags | (p->flags << 5);
|
||||
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 5);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
// Find closest mesh node to the sought point.
|
||||
@@ -714,7 +715,6 @@ static void FindPointsLocal2D_Kernel(const int npt,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
#define MAX_CONST(a, b) (((a) > (b)) ? (a) : (b))
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NE = D1D*D1D;
|
||||
@@ -729,7 +729,7 @@ static void FindPointsLocal2D_Kernel(const int npt,
|
||||
// 3D1D for seed, 10D1D+6 for area, 3D1D+9 for edge
|
||||
constexpr int size1 = 10*MD1 + 6;
|
||||
constexpr int size2 = MD1*4; // edge constraints
|
||||
constexpr int size3 = MD1*MD1*MD1*DIM; // local element coordinates
|
||||
constexpr int size3 = MD1*MD1*DIM; // local element coordinates
|
||||
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
@@ -1162,9 +1162,9 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
|
||||
auto pgslm = gsl_mesh.Read();
|
||||
auto pwt = DEV.wtend.Read();
|
||||
auto pbb = DEV.bb.Read();
|
||||
auto plhm = DEV.loc_hash_min.Read();
|
||||
auto plhf = DEV.loc_hash_fac.Read();
|
||||
auto plho = DEV.loc_hash_offset.ReadWrite();
|
||||
auto plhm = DEV.lh_min.Read();
|
||||
auto plhf = DEV.lh_fac.Read();
|
||||
auto plho = DEV.lh_offset.ReadWrite();
|
||||
auto pcode = code.Write();
|
||||
auto pelem = elem.Write();
|
||||
auto pref = ref.Write();
|
||||
@@ -1177,30 +1177,32 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
|
||||
case 2:
|
||||
return FindPointsLocal2D_Kernel<2>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 3:
|
||||
return FindPointsLocal2D_Kernel<3>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 4:
|
||||
return FindPointsLocal2D_Kernel<4>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 5:
|
||||
return FindPointsLocal2D_Kernel<5>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pbb, DEV.lh_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
default:
|
||||
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx,
|
||||
plhm, plhf, plho, pcode, pelem,
|
||||
pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
}
|
||||
}
|
||||
#undef DIM2
|
||||
#undef DIM
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
|
||||
@@ -706,7 +706,7 @@ newton_vol_fin:
|
||||
int f = flags >> (2*dd) & 3u;
|
||||
res->r[dd] = f == 0 ? r0[dd]+dr[dd] : (f == 1 ? -1 : 1);
|
||||
}
|
||||
res->flags = flags | (p->flags << 7);
|
||||
res->flags = flags | ((p->flags & FLAG_MASK) << 7);
|
||||
}
|
||||
|
||||
// Full Newton solve on the face. One of r/s/t is constrained.
|
||||
@@ -889,7 +889,7 @@ newton_face_fin:
|
||||
res->r[dn] = p->r[dn];
|
||||
res->r[d1] = r[0];
|
||||
res->r[d2] = r[1];
|
||||
res->flags = new_flags | (p->flags << 7);
|
||||
res->flags = new_flags | ((p->flags & FLAG_MASK) << 7);
|
||||
}
|
||||
|
||||
// Full Newton solve on the edge. Two of r/s/t are constrained.
|
||||
@@ -973,7 +973,8 @@ newton_edge_fin:
|
||||
res->r[dn1] = p->r[dn1];
|
||||
res->r[dn2] = p->r[dn2];
|
||||
res->dist2p = -v;
|
||||
res->flags = flags | new_flags | (p->flags << 7);
|
||||
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 7);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
// Find closest mesh node to the sought point.
|
||||
@@ -1252,7 +1253,6 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
case 0: // findpt_vol
|
||||
{
|
||||
double *wtr = r_workspace_ptr;
|
||||
|
||||
double *resid = wtr+6*D1D;
|
||||
double *jac = resid+3;
|
||||
double *resid_temp = jac+9;
|
||||
@@ -1503,7 +1503,7 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
// Hes_T is transposed version (i.e. in col major)
|
||||
// n1*[2, 1, 1, 0, 0]
|
||||
// j==1 => wt_j = wt+n1
|
||||
double *wt_j = wt+D1D*(2-(row+1) / 2);
|
||||
double *wt_j = wt+D1D*(2 - (row+1)/2);
|
||||
const double *x = e_x[row+1][d];
|
||||
hes_T[j] = 0.0;
|
||||
for (int k = 0; k < D1D; ++k)
|
||||
@@ -1522,7 +1522,6 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
hes[j] += resid[d]*hes_T[j*3+d];
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(l,x,1)
|
||||
@@ -1780,6 +1779,7 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
} //findpts_local
|
||||
} //elp
|
||||
});
|
||||
#undef MAXC
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
@@ -1796,9 +1796,9 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
auto pgslm = gsl_mesh.Read();
|
||||
auto pwt = DEV.wtend.Read();
|
||||
auto pbb = DEV.bb.Read();
|
||||
auto plhm = DEV.loc_hash_min.Read();
|
||||
auto plhf = DEV.loc_hash_fac.Read();
|
||||
auto plho = DEV.loc_hash_offset.ReadWrite();
|
||||
auto plhm = DEV.lh_min.Read();
|
||||
auto plhf = DEV.lh_fac.Read();
|
||||
auto plho = DEV.lh_offset.ReadWrite();
|
||||
auto pcode = code.Write();
|
||||
auto pelem = elem.Write();
|
||||
auto pref = ref.Write();
|
||||
@@ -1809,31 +1809,31 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
{
|
||||
case 2:
|
||||
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 3:
|
||||
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 4:
|
||||
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 5:
|
||||
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.lh_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
default:
|
||||
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.h_nx, plhm, plhf,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc,
|
||||
DEV.dof1d);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,725 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
#define sDIM 2
|
||||
#define sDIM2 4
|
||||
#define rDIM 1
|
||||
|
||||
struct findptsElementPoint_t
|
||||
{
|
||||
double x[sDIM], r, oldr, dist2, dist2p, tr;
|
||||
int flags;
|
||||
};
|
||||
|
||||
struct findptsElementGEdge_t
|
||||
{
|
||||
double *x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsElementGPT_t
|
||||
{
|
||||
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[sDIM], A[sDIM*sDIM];
|
||||
dbl_range_t x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[sDIM];
|
||||
double fac[sDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x-z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p1[i] = 2.0 * lCoeff[i] * u1;
|
||||
p2[i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
double b_d;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
if (b_d < 0) // if outside in any dimension
|
||||
{
|
||||
return b_d;
|
||||
}
|
||||
}
|
||||
return b_d; // only positive if inside
|
||||
}
|
||||
|
||||
/* positive when given point is possibly inside given obbox b */
|
||||
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const double bxyz = obbox_axis_test(b,x);
|
||||
if (bxyz<0) // test if point is in AABB
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else // test OBB only if inside AABB
|
||||
{
|
||||
double dxyz[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e=0; e<sDIM; ++e)
|
||||
{
|
||||
rst += b->A[d*2 + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test<0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
/* Hash index in the hash table to the elements that possibly contain the point x */
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[2])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d=sDIM-1; d>=0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
|
||||
{
|
||||
return x[0] * x[0] + x[1] * x[1];
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
1 = 01b if r is constrained at -1, i.e., rmin
|
||||
2 = 10b if r is constrained at +1, i.e., rmax
|
||||
*/
|
||||
|
||||
#define CONVERGED_FLAG (1u<<2)
|
||||
#define FLAG_MASK 0x07u // = 111b
|
||||
|
||||
/* returns 1 if r direction (the only free direction in 2D) is constrained.
|
||||
returns 1 if either 1st or 2nd bit of flags is set.
|
||||
*/
|
||||
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
|
||||
{
|
||||
return ((flags | flags>>1) & 1u);
|
||||
}
|
||||
|
||||
/* pi=0, r=-1; pi=1, r=+1 */
|
||||
static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
{
|
||||
return ((x>>1) & 1u);
|
||||
}
|
||||
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out->dist2, out->index, out->x, out->oldr in any event,
|
||||
leaving out->r, out->dr, out->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
const double resid[2],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
out->x[0] = p->x[0];
|
||||
out->x[1] = p->x[1];
|
||||
out->oldr = p->r;
|
||||
out->dist2 = dist2;
|
||||
if (decr >= 0.01*pred)
|
||||
{
|
||||
if (decr >= 0.9*pred) // very good iteration
|
||||
{
|
||||
out->tr = p->tr*2;
|
||||
}
|
||||
else // somewhat good iteration
|
||||
{
|
||||
out->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
else
|
||||
{
|
||||
/* reject step; note: the point will pass through this routine
|
||||
again, and we set things up here so it gets classed as a
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out->tr = v0/4.0;
|
||||
out->dist2 = p->dist2;
|
||||
out->r = p->oldr;
|
||||
out->flags = p->flags>>3;
|
||||
out->dist2p = -HUGE_VAL;
|
||||
if (pred < dist2*tol)
|
||||
{
|
||||
out->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
|
||||
out,
|
||||
const double jac[2],
|
||||
const double rhess,
|
||||
const double resid[2],
|
||||
int flags,
|
||||
const findptsElementPoint_t *const p,
|
||||
const double tol )
|
||||
{
|
||||
const double tr = p->tr;
|
||||
const double A = jac[0] * jac[0] + jac[1] * jac[1] -
|
||||
rhess; // A = J^T J - resid_d H_d
|
||||
const double y = jac[0]*resid[0] + jac[1]*resid[1]; // y = J^T resid
|
||||
|
||||
const double oldr = p->r;
|
||||
double dr, newr, tdr, tnewr, v, tv;
|
||||
int new_flags=0, tnew_flags=0;
|
||||
|
||||
#define EVAL(dr) ( (dr*A - 2*y) * dr )
|
||||
if (A>0)
|
||||
{
|
||||
dr = y/A;
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
dr=0.0;
|
||||
newr = oldr;
|
||||
}
|
||||
else
|
||||
{
|
||||
newr = oldr+dr;
|
||||
}
|
||||
|
||||
if (fabs(dr)<tr && fabs(newr)<1)
|
||||
{
|
||||
v = EVAL(dr);
|
||||
goto newton_edge_fin;
|
||||
}
|
||||
}
|
||||
|
||||
if ((newr=oldr-tr) > -1)
|
||||
{
|
||||
dr = -tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
newr = -1, dr = -1-oldr, new_flags = flags|1u;
|
||||
}
|
||||
v = EVAL(dr);
|
||||
|
||||
if ((tnewr=oldr+tr) < 1)
|
||||
{
|
||||
tdr = tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
tnewr = 1, tdr = 1-oldr, tnew_flags = flags|2u;
|
||||
}
|
||||
tv = EVAL(tdr);
|
||||
|
||||
if (tv<v)
|
||||
{
|
||||
newr = tnewr, dr = tdr, v = tv, new_flags = tnew_flags;
|
||||
}
|
||||
#undef EVAL
|
||||
|
||||
newton_edge_fin:
|
||||
// check convergence by testing if change in r is less than tol
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r = newr;
|
||||
out->dist2p = -v;
|
||||
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
|
||||
const double x[sDIM],
|
||||
const double *z,
|
||||
double *dist2,
|
||||
double *r,
|
||||
const int ir,
|
||||
const int pN )
|
||||
{
|
||||
double dx[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dx[d] = x[d] - elx[d][ir];
|
||||
}
|
||||
dist2[ir] = HUGE_VAL;
|
||||
const double dist2_rs = l2norm2(dx);
|
||||
if (dist2[ir]>dist2_rs)
|
||||
{
|
||||
dist2[ir] = dist2_rs;
|
||||
r[ir] = z[ir];
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal2D_Kernel( const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0 )
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NEL = nel*D1D;
|
||||
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
|
||||
const int nThreads = D1D*sDIM;
|
||||
|
||||
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
// 2D1D for seed, 3D1D + 7 for edge
|
||||
constexpr int size1 = 3*MD1 + 7;
|
||||
// edge coordinates = D1D*2
|
||||
constexpr int size2 = 2*MD1;
|
||||
// local element coordinates in shared memory
|
||||
constexpr int size3 = MD1*sDIM;
|
||||
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
|
||||
MFEM_SHARED double constraint_workspace[size2];
|
||||
|
||||
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
|
||||
|
||||
double *r_workspace_ptr = r_workspace;
|
||||
findptsElementPoint_t *fpt, *tmp;
|
||||
fpt = el_pts + 0;
|
||||
tmp = el_pts + 1;
|
||||
|
||||
// x and y coord index within point_pos for point i
|
||||
int id_x = point_pos_ordering == 0 ? i : i*sDIM;
|
||||
int id_y = point_pos_ordering == 0 ? i+npt : i*sDIM+1;
|
||||
double x_i[2] = {x[id_x], x[id_y]};
|
||||
|
||||
unsigned int *code_i = code_base + i;
|
||||
double *dist2_i = dist2_base + i;
|
||||
|
||||
//---------------- map_points_to_els --------------------
|
||||
findptsLocalHashData_t hash;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
hash.bnd[d].min = hashMin[d];
|
||||
hash.fac[d] = hashFac[d];
|
||||
}
|
||||
hash.hash_n = hash_n;
|
||||
hash.offset = hashOffset;
|
||||
|
||||
const int hi = hash_index(&hash, x_i);
|
||||
const unsigned int *elp = hash.offset + hash.offset[hi];
|
||||
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
|
||||
*code_i = CODE_NOT_FOUND;
|
||||
*dist2_i = HUGE_VAL;
|
||||
|
||||
for (; elp!=ele; ++elp)
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
obbox_t box;
|
||||
int n_box_ents = 3*sDIM + sDIM2;
|
||||
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
|
||||
if (obbox_test(&box,x_i)>=0)
|
||||
{
|
||||
//------------ findpts_local ------------------
|
||||
{
|
||||
if (MD1 <= 6)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
|
||||
{
|
||||
const int qp = j % D1D;
|
||||
const int d = j / D1D;
|
||||
elem_coords[qp + d*D1D] =
|
||||
xElemCoord[qp + el*D1D + d*p_NEL];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
const double *elx[sDIM];
|
||||
for (int d=0; d<sDIM; d++)
|
||||
{
|
||||
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
|
||||
xElemCoord + d*p_NEL + el*D1D;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
//// findpts_el ////
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
fpt->dist2p = 0;
|
||||
fpt->tr = 1;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
fpt->x[j] = x_i[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
{
|
||||
double *dist2_temp = r_workspace_ptr;
|
||||
double *r_temp = dist2_temp + D1D;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
for (int ir=0; ir<D1D; ++ir)
|
||||
{
|
||||
if (dist2_temp[ir]<fpt->dist2)
|
||||
{
|
||||
fpt->dist2 = dist2_temp[ir];
|
||||
fpt->r = r_temp[ir];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //seed done
|
||||
|
||||
// Initialize tmp struct with fpt values before starting Newton iterations
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
tmp->dist2 = HUGE_VAL;
|
||||
tmp->dist2p = 0;
|
||||
tmp->tr = 1;
|
||||
tmp->flags = 0;
|
||||
tmp->r = fpt->r;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
tmp->x[j] = fpt->x[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
|
||||
for (int step=0; step<50; step++)
|
||||
{
|
||||
int nc = num_constrained(tmp->flags & FLAG_MASK);
|
||||
switch (nc)
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
double *wt = r_workspace_ptr;
|
||||
double *resid = wt + 3*D1D;
|
||||
double *jac = resid + sDIM;
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// compute basis function info upto 2nd derivative
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lag_eval_second_der(wt, tmp->r, j, gll1D,
|
||||
lagcoeff, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
resid[j] = tmp->x[j];
|
||||
jac[j] = 0.0;
|
||||
hess[j] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
resid[j] -= wt[ k]*edge.x[j][k];
|
||||
jac[j] += wt[D1D+k]*edge.x[j][k];
|
||||
hess[j] += wt[2*D1D+k]*edge.x[j][k];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
hess[2] = resid[0]*hess[0] + resid[1]*hess[1];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
if (!reject_prior_step_q(fpt, resid, tmp, tol))
|
||||
{
|
||||
newton_edge(fpt, jac, hess[2], resid,
|
||||
tmp->flags & FLAG_MASK, tmp, tol);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
}
|
||||
case 1: // r is constrained to either -1 or 1
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
const int pi = point_index(tmp->flags &
|
||||
FLAG_MASK);
|
||||
const double *wt = wtend + pi*3*D1D;
|
||||
findptsElementGPT_t gpt;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
gpt.x[d] = elx[d][pi*(D1D-1)];
|
||||
gpt.jac[d] = 0.0;
|
||||
gpt.hes[d] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
gpt.jac[d] += wt[D1D +k]*elx[d][k];
|
||||
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
|
||||
}
|
||||
}
|
||||
|
||||
const double *const pt_x = gpt.x;
|
||||
const double *const jac = gpt.jac;
|
||||
const double *const hes = gpt.hes;
|
||||
double resid[sDIM], steep, sr;
|
||||
resid[0] = fpt->x[0] - pt_x[0];
|
||||
resid[1] = fpt->x[1] - pt_x[1];
|
||||
steep = jac[0]*resid[0] + jac[1]*resid[1];
|
||||
sr = steep*tmp->r;
|
||||
if ( !reject_prior_step_q(fpt, resid, tmp, tol) )
|
||||
{
|
||||
if (sr<0)
|
||||
{
|
||||
const double rhess = resid[0]*hes[0] +
|
||||
resid[1]*hes[1];
|
||||
newton_edge(fpt, jac, rhess,
|
||||
resid, 0, tmp, tol);
|
||||
}
|
||||
else // sr==0
|
||||
{
|
||||
fpt->r = tmp->r;
|
||||
fpt->dist2p = 0;
|
||||
fpt->flags = tmp->flags | CONVERGED_FLAG;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
} // case 1
|
||||
} //switch
|
||||
if (fpt->flags & CONVERGED_FLAG)
|
||||
{
|
||||
break;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*tmp = *fpt;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //for int step<50
|
||||
} //findpts_el
|
||||
|
||||
bool converged_internal =
|
||||
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
|
||||
(fpt->dist2<dist2tol);
|
||||
|
||||
if (*code_i == CODE_NOT_FOUND || converged_internal ||
|
||||
fpt->dist2 < *dist2_i)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*(el_base+i) = el;
|
||||
*code_i = converged_internal ? CODE_INTERNAL : CODE_BORDER;
|
||||
*dist2_i = fpt->dist2;
|
||||
*(r_base+i) = fpt->r;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (converged_internal)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
} //findpts_local
|
||||
} //obbox_test
|
||||
} //elp
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt )
|
||||
{
|
||||
if (npt==0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
MFEM_VERIFY(dim==1 && spacedim==2,"Function for 2D edges only");
|
||||
bool use_dev = point_pos.UseDevice();
|
||||
auto pp = point_pos.Read(use_dev);
|
||||
auto pgslm = gsl_mesh.Read(use_dev);
|
||||
auto pwt = DEV.wtend.Read(use_dev);
|
||||
auto pbb = DEV.bb.Read(use_dev);
|
||||
auto plhm = DEV.lh_min.Read(use_dev);
|
||||
auto plhf = DEV.lh_fac.Read(use_dev);
|
||||
auto plho = DEV.lh_offset.ReadWrite(use_dev);
|
||||
auto pcode = code.Write(use_dev);
|
||||
auto pelem = elem.Write(use_dev);
|
||||
auto pref = ref.Write(use_dev);
|
||||
auto pdist = dist.Write(use_dev);
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsEdgeLocal2D_Kernel<2>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
case 3:
|
||||
return FindPointsEdgeLocal2D_Kernel<3>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
case 4:
|
||||
return FindPointsEdgeLocal2D_Kernel<4>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
default:
|
||||
return FindPointsEdgeLocal2D_Kernel(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
}
|
||||
}
|
||||
#undef sDIM
|
||||
#undef rDIM
|
||||
#undef sDIM2
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt ) {} ;
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
@@ -0,0 +1,733 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
#define sDIM 3
|
||||
#define rDIM 1
|
||||
#define sDIM2 (sDIM*sDIM)
|
||||
#define rDIM2 (rDIM*rDIM)
|
||||
|
||||
struct findptsElementPoint_t
|
||||
{
|
||||
double x[sDIM], r, oldr, dist2, dist2p, tr;
|
||||
int flags;
|
||||
};
|
||||
|
||||
struct findptsElementGEdge_t
|
||||
{
|
||||
double *x[sDIM], *dxdn[sDIM], *d2xdn[sDIM];
|
||||
};
|
||||
|
||||
struct findptsElementGPT_t
|
||||
{
|
||||
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
|
||||
};
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[sDIM], A[sDIM*sDIM];
|
||||
dbl_range_t x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[sDIM];
|
||||
double fac[sDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j=0; j<pN; ++j)
|
||||
{
|
||||
if (i!=j)
|
||||
{
|
||||
double d_j = 2 * (x-z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
double *p1 = p0 + pN, *p2 = p0 + 2 * pN;
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p1[i] = 2.0 * lCoeff[i] * u1;
|
||||
p2[i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_axis_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
double b_d;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
if (b_d < 0) // if outside in any dimension
|
||||
{
|
||||
return b_d;
|
||||
}
|
||||
}
|
||||
return b_d; // only positive if inside in all dimensions
|
||||
}
|
||||
|
||||
/* positive when possibly inside */
|
||||
static MFEM_HOST_DEVICE inline double obbox_test(const obbox_t *const b,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const double bxyz = obbox_axis_test(b, x);
|
||||
if (bxyz<0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[3];
|
||||
// dxyz: distance of the point from the center of the OBB
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
// transform dxyz to the local coordinate system of the OBB,
|
||||
// and check if the point is inside the OBB [-1,1]^sDIM
|
||||
double test = 1;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e=0; e<sDIM; ++e)
|
||||
{
|
||||
rst += b->A[d*sDIM + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test<0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
/* Hash index in the hash table to the elements that possibly contain the point x */
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[sDIM])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d=sDIM-1; d>=0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i<0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
|
||||
static MFEM_HOST_DEVICE inline double norm2(const double x[sDIM])
|
||||
{
|
||||
return ( x[0]*x[0] + x[1]*x[1] + x[2]*x[2] );
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
1 = 01b if r is constrained at -1, i.e., rmin
|
||||
2 = 10b if r is constrained at +1, i.e., rmax
|
||||
*/
|
||||
#define CONVERGED_FLAG (1u<<2)
|
||||
#define FLAG_MASK 0x07u
|
||||
|
||||
/* returns the number of constrained reference coordinates, max 2
|
||||
*/
|
||||
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
|
||||
{
|
||||
const int y = (flags | flags>>1);
|
||||
return (y & 1u) + (y>>2 & 1u);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
{
|
||||
return ((x>>1)&1u) | ((x>>2)&2u);
|
||||
}
|
||||
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out->dist2, out->index, out->x, out->oldr in any event,
|
||||
leaving out->r, out->dr, out->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out,
|
||||
const double resid[3],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = norm2(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
out->x[d] = p->x[d];
|
||||
}
|
||||
out->oldr = p->r;
|
||||
out->dist2 = dist2;
|
||||
if (decr>=0.01*pred)
|
||||
{
|
||||
if (decr>=0.9*pred) // very good iteration
|
||||
{
|
||||
out->tr = 2*p->tr;
|
||||
}
|
||||
else // good iteration
|
||||
{
|
||||
out->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
else // if the iteration in not good
|
||||
{
|
||||
/* reject step; note: the point will pass through this routine
|
||||
again, and we set things up here so it gets classed as a
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out->tr = v0/4.0;
|
||||
out->dist2 = p->dist2;
|
||||
out->r = p->oldr;
|
||||
out->flags = p->flags>>3;
|
||||
out->dist2p = -HUGE_VAL;
|
||||
if (pred<dist2*tol)
|
||||
{
|
||||
out->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
|
||||
out,
|
||||
const double jac[sDIM*rDIM],
|
||||
const double rhes,
|
||||
const double resid[sDIM],
|
||||
int flags,
|
||||
const findptsElementPoint_t *const p,
|
||||
const double tol)
|
||||
{
|
||||
const double tr = p->tr;
|
||||
/* A = J^T J - resid_d H_d */
|
||||
const double A = jac[0]*jac[0]+ jac[1] * jac[1] + jac[2] * jac[2]
|
||||
- rhes;
|
||||
/* y = J^T r */
|
||||
const double y = jac[0]*resid[0] + jac[1]*resid[1] + jac[0+2]*resid[2];
|
||||
|
||||
const double oldr = p->r;
|
||||
double dr, nr, tdr, tnr;
|
||||
double v, tv;
|
||||
int new_flags = 0, tnew_flags = 0;
|
||||
|
||||
#define EVAL(dr) (dr*A - 2*y)*dr
|
||||
|
||||
/* if A is not SPD, quadratic model has no minimum */
|
||||
if (A>0)
|
||||
{
|
||||
dr = y/A;
|
||||
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
dr=0.0;
|
||||
nr = oldr;
|
||||
}
|
||||
else
|
||||
{
|
||||
nr = oldr+dr;
|
||||
}
|
||||
if ( fabs(dr)<tr && fabs(nr)<1 )
|
||||
{
|
||||
v = EVAL(dr);
|
||||
goto newton_edge_fin;
|
||||
}
|
||||
}
|
||||
|
||||
if ( (nr=oldr-tr)>-1 )
|
||||
{
|
||||
dr = -tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
nr = -1, dr = -1-oldr, new_flags = flags | 1u;
|
||||
}
|
||||
v = EVAL(dr);
|
||||
|
||||
if ( (tnr = oldr+tr)<1 )
|
||||
{
|
||||
tdr = tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
tnr = 1, tdr = 1-oldr, tnew_flags = flags | 2u;
|
||||
}
|
||||
tv = EVAL(tdr);
|
||||
|
||||
if (tv<v)
|
||||
{
|
||||
nr = tnr, dr = tdr, v = tv, new_flags = tnew_flags;
|
||||
}
|
||||
|
||||
newton_edge_fin:
|
||||
/* check convergence */
|
||||
if ( fabs(dr)<tol )
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out->r = nr;
|
||||
out->dist2p = -v;
|
||||
out->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
|
||||
const double x[sDIM],
|
||||
const double *z,
|
||||
double *dist2,
|
||||
double *r,
|
||||
const int ir,
|
||||
const int pN)
|
||||
{
|
||||
if (ir>=pN)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
double dx[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dx[d] = x[d] - elx[d][ir];
|
||||
}
|
||||
dist2[ir] = norm2(dx);;
|
||||
r[ir] = z[ir];
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal3D_Kernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NEL = nel*D1D;
|
||||
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
|
||||
const int nThreads = D1D*sDIM;
|
||||
|
||||
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
constexpr int size1 = 3*MD1 + 13;
|
||||
constexpr int size2 = 3*MD1;
|
||||
constexpr int size3 = MD1*sDIM;
|
||||
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
|
||||
MFEM_SHARED double constraint_workspace[size2];
|
||||
|
||||
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
|
||||
|
||||
double *r_workspace_ptr = r_workspace;
|
||||
findptsElementPoint_t *fpt, *tmp;
|
||||
fpt = el_pts + 0;
|
||||
tmp = el_pts + 1;
|
||||
|
||||
int id_x = point_pos_ordering==0 ? i : i*sDIM;
|
||||
int id_y = point_pos_ordering==0 ? npt+i : 1+i*sDIM;
|
||||
int id_z = point_pos_ordering==0 ? 2*npt+i : 2+i*sDIM;
|
||||
double x_i[3] = {x[id_x], x[id_y], x[id_z]};
|
||||
|
||||
unsigned int *code_i = code_base + i;
|
||||
double *dist2_i = dist2_base + i;
|
||||
|
||||
//// map_points_to_els ////
|
||||
findptsLocalHashData_t hash;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
hash.bnd[d].min = hashMin[d];
|
||||
hash.fac[d] = hashFac[d];
|
||||
}
|
||||
hash.hash_n = hash_n;
|
||||
hash.offset = hashOffset;
|
||||
|
||||
const unsigned int hi = hash_index(&hash, x_i);
|
||||
const unsigned int *elp = hash.offset + hash.offset[hi];
|
||||
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
|
||||
*code_i = CODE_NOT_FOUND;
|
||||
*dist2_i = HUGE_VAL;
|
||||
|
||||
for (; elp!=ele; ++elp)
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
obbox_t box;
|
||||
int n_box_ents = 3*sDIM + sDIM2;
|
||||
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
|
||||
if (obbox_test(&box, x_i)>=0)
|
||||
{
|
||||
//// findpts_local ////
|
||||
{
|
||||
if (MD1 <= 6)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
|
||||
{
|
||||
const int qp = j % D1D;
|
||||
const int d = j / D1D;
|
||||
elem_coords[qp + d*D1D] =
|
||||
xElemCoord[qp + el*D1D + d*p_NEL];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
const double *elx[sDIM];
|
||||
for (int d=0; d<sDIM; d++)
|
||||
{
|
||||
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
|
||||
xElemCoord + d*p_NEL + el*D1D;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
//// findpts_el ////
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
fpt->dist2p = 0;
|
||||
fpt->tr = 1.0;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
fpt->x[j] = x_i[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
//// seed ////
|
||||
{
|
||||
double *dist2_temp = r_workspace_ptr;
|
||||
double *r_temp = dist2_temp + D1D;
|
||||
MFEM_FOREACH_THREAD(j,x,nThreads)
|
||||
{
|
||||
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
for (int ir=0; ir<D1D; ++ir)
|
||||
{
|
||||
if (dist2_temp[ir] < fpt->dist2)
|
||||
{
|
||||
fpt->dist2 = dist2_temp[ir];
|
||||
fpt->r = r_temp[ir];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //seed done
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
tmp->dist2 = HUGE_VAL;
|
||||
tmp->dist2p = 0;
|
||||
tmp->tr = 1;
|
||||
tmp->flags = 0;
|
||||
tmp->r = fpt->r;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
tmp->x[j] = fpt->x[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int step=0; step<50; step++)
|
||||
{
|
||||
switch (num_constrained(tmp->flags & FLAG_MASK))
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
double *wt = r_workspace_ptr;
|
||||
double *resid = wt + 3*D1D;
|
||||
double *jac = resid + sDIM;
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lag_eval_second_der(wt, tmp->r, j, gll1D,
|
||||
lagcoeff, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
resid[j] = tmp->x[j];
|
||||
jac[j] = 0.0;
|
||||
hess[j] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
resid[j] -= wt[ k]*edge.x[j][k];
|
||||
jac[j] += wt[D1D+k]*edge.x[j][k];
|
||||
hess[j] += wt[2*D1D+k]*edge.x[j][k];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
hess[3] = resid[0]*hess[0] + resid[1]*hess[1] +
|
||||
resid[2]*hess[2];
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(l,x,1)
|
||||
{
|
||||
if (!reject_prior_step_q(fpt,resid,tmp,tol))
|
||||
{
|
||||
newton_edge(fpt,jac,hess[3],resid,
|
||||
tmp->flags&FLAG_MASK,tmp,tol);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
}
|
||||
case 1:
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
const int pi = point_index(tmp->flags &
|
||||
FLAG_MASK);
|
||||
const double *wt = wtend + pi*3*D1D;
|
||||
findptsElementGPT_t gpt;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
gpt.x[d] = elx[d][pi*(D1D-1)];
|
||||
gpt.jac[d] = 0.0;
|
||||
gpt.hes[d] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
gpt.jac[d] += wt[D1D +k]*elx[d][k];
|
||||
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
|
||||
}
|
||||
}
|
||||
|
||||
const double *const pt_x = gpt.x;
|
||||
const double *const jac = gpt.jac;
|
||||
const double *const hes = gpt.hes;
|
||||
double resid[sDIM], steep, sr;
|
||||
resid[0] = fpt->x[0] - pt_x[0];
|
||||
resid[1] = fpt->x[1] - pt_x[1];
|
||||
resid[2] = fpt->x[2] - pt_x[2];
|
||||
steep = jac[0]*resid[0] + jac[1]*resid[1] +
|
||||
jac[2]*resid[2];
|
||||
sr = steep*tmp->r;
|
||||
if (!reject_prior_step_q(fpt, resid, tmp, tol))
|
||||
{
|
||||
if (sr<0)
|
||||
{
|
||||
const double rhess = resid[0]*hes[0] +
|
||||
resid[1]*hes[1] +
|
||||
resid[2]*hes[2];
|
||||
newton_edge(fpt, jac, rhess,
|
||||
resid, 0, tmp, tol);
|
||||
}
|
||||
else // sr==0
|
||||
{
|
||||
fpt->r = tmp->r;
|
||||
fpt->dist2p = 0;
|
||||
fpt->flags = tmp->flags | CONVERGED_FLAG;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
} // case 1
|
||||
} //switch
|
||||
if (fpt->flags & CONVERGED_FLAG)
|
||||
{
|
||||
break;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*tmp = *fpt;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} // for step<50
|
||||
} // findpts_el
|
||||
|
||||
bool converged_internal =
|
||||
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
|
||||
(fpt->dist2<dist2tol);
|
||||
if (*code_i==CODE_NOT_FOUND || converged_internal ||
|
||||
fpt->dist2<*dist2_i)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*(el_base+i) = el;
|
||||
*code_i = converged_internal?CODE_INTERNAL:CODE_BORDER;
|
||||
*dist2_i = fpt->dist2;
|
||||
*(r_base+i) = fpt->r;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (converged_internal)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
} // findpts_local
|
||||
} // obbox_test
|
||||
} // elp
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt)
|
||||
{
|
||||
if (npt == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
MFEM_VERIFY(spacedim==3 && dim == 1,"Function for 3D edges only");
|
||||
bool use_dev = point_pos.UseDevice();
|
||||
auto pp = point_pos.Read(use_dev);
|
||||
auto pgslm = gsl_mesh.Read(use_dev);
|
||||
auto pwt = DEV.wtend.Read(use_dev);
|
||||
auto pbb = DEV.bb.Read(use_dev);
|
||||
auto plhm = DEV.lh_min.Read(use_dev);
|
||||
auto plhf = DEV.lh_fac.Read(use_dev);
|
||||
auto plho = DEV.lh_offset.ReadWrite(use_dev);
|
||||
auto pcode = code.Write(use_dev);
|
||||
auto pelem = elem.Write(use_dev);
|
||||
auto pref = ref.Write(use_dev);
|
||||
auto pdist = dist.Write(use_dev);
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
return FindPointsEdgeLocal3D_Kernel<2>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
case 3:
|
||||
return FindPointsEdgeLocal3D_Kernel<3>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
case 4:
|
||||
return FindPointsEdgeLocal3D_Kernel<4>(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
default:
|
||||
return FindPointsEdgeLocal3D_Kernel(
|
||||
npt, DEV.newt_tol, dist2tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.lh_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
}
|
||||
}
|
||||
#undef rDIM2
|
||||
#undef sDIM2
|
||||
#undef rDIM
|
||||
#undef sDIM
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal3( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt ) {} ;
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,157 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j=0; j<p_Nq; ++j)
|
||||
{
|
||||
p_i *= j==i ? 1 : x-z[j];
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal1DKernel(const double *const gf_in,
|
||||
int *const el,
|
||||
double *const r,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int nfields,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_Nq = D1D;
|
||||
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
|
||||
// for each point of the npt points, create a thread block of size dof1Dsol
|
||||
mfem::forall_2D(npt, D1D, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
MFEM_SHARED double wtr[MD1];
|
||||
MFEM_SHARED double sums[MD1];
|
||||
|
||||
// Evaluate basis functions at the reference space coordinates
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lagrange_eval(wtr, r[i], j, p_Nq, gll1D, lagcoeff);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int fld=0; fld<nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Nq + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
const int elemOffset = el[i]*nfields*p_Nq + fld*p_Nq;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
sums[j] = wtr[j] * gf_in[elemOffset + j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
double sumv = 0.0;
|
||||
// sum the contributions of each lagrange polynomial
|
||||
for (int jj=0; jj<D1D; ++jj)
|
||||
{
|
||||
sumv += sums[jj];
|
||||
}
|
||||
int_out[fld*npt + i] = sumv;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt,
|
||||
int ncomp,
|
||||
int dof1Dsol )
|
||||
{
|
||||
MFEM_VERIFY(dim == 1, "Kernel for edges only.");
|
||||
if (npt == 0) { return; }
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
}
|
||||
}
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::InterpolateLocal1(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol) {};
|
||||
#endif
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
@@ -52,8 +52,6 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int ncomp,
|
||||
const int nel,
|
||||
const int gf_offset,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
@@ -64,6 +62,8 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
const int p_Np = D1D*D1D;
|
||||
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
|
||||
mfem::forall_2D(npt, D1D, D1D, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
@@ -82,9 +82,9 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
|
||||
for (int fld = 0; fld < Nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
|
||||
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
|
||||
//if using R->Mult for L -> E-Vec use below: NDOFSxVDIMxNEL
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Np + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
@@ -120,32 +120,32 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1Dsol)
|
||||
int dof1Dsol)
|
||||
{
|
||||
if (npt == 0) { return; }
|
||||
const int gf_offset = field_in.Size()/ncomp;
|
||||
auto pfin = field_in.Read();
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite();
|
||||
auto pgslr = gsl_ref_l.ReadWrite();
|
||||
auto pfout = field_out.Write();
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite();
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite();
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
}
|
||||
}
|
||||
@@ -160,7 +160,7 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1Dsol) {};
|
||||
int dof1Dsol) {};
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
@@ -52,8 +52,6 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int ncomp,
|
||||
const int nel,
|
||||
const int gf_offset,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
@@ -84,9 +82,9 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
|
||||
for (int fld = 0; fld < Nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
|
||||
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
|
||||
//if using R->Mult for L -> E-Vec use below.
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Np + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
@@ -125,37 +123,38 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1Dsol)
|
||||
int dof1Dsol)
|
||||
{
|
||||
if (npt == 0) { return; }
|
||||
const int gf_offset = field_in.Size()/ncomp;
|
||||
auto pfin = field_in.Read();
|
||||
auto pgsle = gsl_elem_dev_l.ReadWrite();
|
||||
auto pgslr = gsl_ref_l.ReadWrite();
|
||||
auto pfout = field_out.Write();
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite();
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite();
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsle = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
npt, ncomp,
|
||||
pgll, plcf, dof1Dsol);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#undef MAXC
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
@@ -165,7 +164,7 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int nel, int dof1Dsol) {};
|
||||
int dof1Dsol) {};
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_diffusion_pa_simplices.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -19,6 +20,13 @@ namespace mfem
|
||||
DiffusionIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// 2D
|
||||
// Q = P, only for simplex
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,2,1>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,3,2>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,4,3>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,5,4>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,6,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,7,6>();
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<2,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<2,2,2>();
|
||||
@@ -40,7 +48,18 @@ DiffusionIntegrator::Kernels::Kernels()
|
||||
DiffusionIntegrator::AddSpecialization<2,8,9>();
|
||||
DiffusionIntegrator::AddSpecialization<2,9,10>();
|
||||
// others
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,2,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,3,6>();
|
||||
|
||||
// 3D
|
||||
// Q = P, only for simplex
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,2,1>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,3,2>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,4,3>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,5,4>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,6,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,7,6>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,8,7>();
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<3,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<3,2,2>();
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
|
||||
#include "../kernel_dispatch.hpp"
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
@@ -637,8 +636,8 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &b_,
|
||||
const Array<real_t> &g_,
|
||||
const Array<real_t> &bt_,
|
||||
const Array<real_t> >_,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
@@ -1064,8 +1063,6 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad X
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
|
||||
@@ -1086,8 +1083,6 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad Y
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
|
||||
@@ -1109,8 +1104,6 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Grad Z + Q-function
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
|
||||
@@ -1223,48 +1216,48 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
|
||||
namespace
|
||||
{
|
||||
using DiffusionApplyKernelType =
|
||||
DiffusionIntegrator::DiffusionApplyKernelType;
|
||||
|
||||
using DiffusionDiagonalKernelType =
|
||||
DiffusionIntegrator::DiffusionDiagonalKernelType;
|
||||
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
|
||||
using ApplySimplexKernelType = DiffusionIntegrator::ApplySimplexKernelType;
|
||||
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
|
||||
}
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
DiffusionApplyKernelType DiffusionIntegrator::DiffusionApplyPAKernel::Kernel()
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<D1D, Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
inline DiffusionApplyKernelType
|
||||
DiffusionIntegrator::DiffusionApplyPAKernel::Fallback(int DIM, int, int)
|
||||
inline
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PADiffusionApply2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionApply3D; }
|
||||
if (dim == 2) { return internal::PADiffusionApply2D; }
|
||||
else if (dim == 3) { return internal::PADiffusionApply3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
DiffusionDiagonalKernelType
|
||||
DiffusionIntegrator::DiffusionDiagonalPAKernel::Kernel()
|
||||
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D, Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
inline DiffusionDiagonalKernelType
|
||||
DiffusionIntegrator::DiffusionDiagonalPAKernel::Fallback(int DIM, int, int)
|
||||
inline DiagonalKernelType
|
||||
DiffusionIntegrator::DiagonalPAKernels::Fallback(int dim, int, int)
|
||||
{
|
||||
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
|
||||
if (dim == 2) { return internal::PADiffusionDiagonal2D; }
|
||||
else if (dim == 3) { return internal::PADiffusionDiagonal3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include "../../mesh/nurbs.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_diffusion_pa_simplices.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -31,8 +32,8 @@ void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
const Array<real_t> &B = maps->B;
|
||||
const Array<real_t> &G = maps->G;
|
||||
const Vector &Dv = pa_data;
|
||||
DiffusionDiagonalPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
|
||||
diag, dofs1D, quad1D);
|
||||
DiagonalPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Dv,
|
||||
diag, dofs1D, quad1D);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,8 +69,26 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
|
||||
Gt, Dv, x, y, dofs1D, quad1D);
|
||||
if (fespace->UsesRaggedTensorBasis())
|
||||
{
|
||||
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
|
||||
return ApplySimplexPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
rmaps->lex_map,
|
||||
rmaps->forward_map2d_diff,
|
||||
rmaps->inverse_map2d_diff,
|
||||
rmaps->forward_map3d_diff,
|
||||
rmaps->inverse_map3d_diff,
|
||||
rmaps->Ga1,
|
||||
rmaps->Ga2,
|
||||
rmaps->Ga3,
|
||||
rmaps->Ga1t,
|
||||
rmaps->Ga2t,
|
||||
rmaps->Ga3t,
|
||||
Dv, x, y, dofs1D, quad1D);
|
||||
}
|
||||
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
|
||||
Gt, Dv, x, y, dofs1D, quad1D);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -94,7 +113,8 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
|
||||
const bool stroud = fes.UsesRaggedTensorBasis();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, stroud);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
@@ -119,13 +139,22 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
if (stroud)
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
|
||||
}
|
||||
else
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
}
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
|
||||
// QuadratureSpace expects ir defined in reference simplex for Bernstein
|
||||
// elements with partial assembly
|
||||
|
||||
if (MQ) { coeff.ProjectTranspose(*MQ); }
|
||||
else if (VQ) { coeff.Project(*VQ); }
|
||||
@@ -174,9 +203,9 @@ void DiffusionIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
abs_pa_data.Abs();
|
||||
auto abs_maps = maps->Abs();
|
||||
|
||||
DiffusionApplyPAKernel::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
|
||||
abs_pa_data, x, y, dofs1D, quad1D);
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
abs_maps.B, abs_maps.G, abs_maps.Bt, abs_maps.Gt,
|
||||
abs_pa_data, x, y, dofs1D, quad1D);
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::AddAbsMultTransposePA(const Vector &x,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "bilininteg_mass_kernels.hpp"
|
||||
#include "bilininteg_mass_pa_simplices.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -39,8 +40,10 @@ MassIntegrator::Kernels::Kernels()
|
||||
MassIntegrator::AddSpecialization<2,9,10>();
|
||||
// others
|
||||
MassIntegrator::AddSpecialization<2,2,4>();
|
||||
MassIntegrator::AddSpecialization<2,2,5>();
|
||||
MassIntegrator::AddSpecialization<2,3,6>();
|
||||
MassIntegrator::AddSpecialization<2,4,6>();
|
||||
|
||||
// 3D
|
||||
// Q=P+1
|
||||
MassIntegrator::AddSpecialization<3,1,1>();
|
||||
|
||||
@@ -181,6 +181,12 @@ constexpr int NBZ(int D1D)
|
||||
{
|
||||
return ipow(2, D(D1D) >= 0 ? D(D1D) : 0);
|
||||
}
|
||||
constexpr int NBZ3D(int MDQ)
|
||||
{
|
||||
return MDQ > 0 ? std::min<int>(
|
||||
(128 + MDQ * MDQ * MDQ - 1) / (MDQ * MDQ * MDQ), 64)
|
||||
: 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Diagonal 2D kernel
|
||||
@@ -804,19 +810,23 @@ void PAMassApply3D_Element(const int e,
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D, int T_Q1D, bool ACCUMULATE = true>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void SmemPAMassApply3D_Element(const int e,
|
||||
const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *d_,
|
||||
const real_t *x_,
|
||||
real_t *y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
template <int T_D1D, int T_Q1D, int TBATCH, bool ACCUMULATE = true>
|
||||
MFEM_HOST_DEVICE inline void
|
||||
SmemPAMassApply3D_Element(const int e, const int NE, const real_t *b_,
|
||||
const real_t *d_, const real_t *x_, real_t *y_,
|
||||
int d1d = 0, int q1d = 0)
|
||||
{
|
||||
constexpr int D1D = T_D1D ? T_D1D : d1d;
|
||||
constexpr int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
static_assert(TBATCH > 0, "TBATCH must be positive");
|
||||
#if defined(__CUDA_ARCH__) || defined(__HIP_DEVICE_COMPILE__)
|
||||
constexpr int tbatch = TBATCH;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
#else
|
||||
// host always batch size 1
|
||||
constexpr int tbatch = 1;
|
||||
constexpr int tidz = 0;
|
||||
#endif
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
@@ -829,33 +839,37 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
MFEM_SHARED real_t sDQ[MQ1*MD1];
|
||||
real_t (*B)[MD1] = (real_t (*)[MD1]) sDQ;
|
||||
real_t (*Bt)[MQ1] = (real_t (*)[MQ1]) sDQ;
|
||||
MFEM_SHARED real_t sm0[MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[MDQ*MDQ*MDQ];
|
||||
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm0;
|
||||
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) sm1;
|
||||
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm0;
|
||||
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) sm1;
|
||||
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) sm0;
|
||||
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) sm1;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_SHARED real_t sm0[tbatch][MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[tbatch][MDQ*MDQ*MDQ];
|
||||
real_t (*X)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+tidz);
|
||||
real_t (*DDQ)[MD1][MQ1] = (real_t (*)[MD1][MQ1]) (sm1+tidz);
|
||||
real_t (*DQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm0+tidz);
|
||||
real_t (*QQQ)[MQ1][MQ1] = (real_t (*)[MQ1][MQ1]) (sm1+tidz);
|
||||
real_t (*QQD)[MQ1][MD1] = (real_t (*)[MQ1][MD1]) (sm0+tidz);
|
||||
real_t (*QDD)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm1+tidz);
|
||||
MFEM_FOREACH_THREAD(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(dx, x, D1D)
|
||||
{
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
X[dz][dy][dx] = x(dx,dy,dz,e);
|
||||
X[dz][dy][dx] = x(dx, dy, dz, e);
|
||||
}
|
||||
}
|
||||
MFEM_FOREACH_THREAD(dx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
|
||||
}
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, D1D)
|
||||
{
|
||||
B[dx][dy] = b(dx,dy);
|
||||
MFEM_FOREACH_THREAD(dx, x, Q1D) { B[dx][dy] = b(dx, dy); }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(qx, x, Q1D)
|
||||
{
|
||||
real_t u[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
@@ -880,9 +894,9 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(qx, x, Q1D)
|
||||
{
|
||||
real_t u[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
@@ -907,9 +921,9 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(qx, x, Q1D)
|
||||
{
|
||||
real_t u[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
@@ -929,22 +943,22 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
QQQ[qz][qy][qx] = u[qz] * d(qx,qy,qz,e);
|
||||
QQQ[qz][qy][qx] = u[qz] * d(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(di,y,D1D)
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(di, y, D1D)
|
||||
{
|
||||
Bt[di][q] = b(q,di);
|
||||
MFEM_FOREACH_THREAD(q, x, Q1D) { Bt[di][q] = b(q, di); }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(dx, x, D1D)
|
||||
{
|
||||
real_t u[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
@@ -969,9 +983,9 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(dx, x, D1D)
|
||||
{
|
||||
real_t u[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
@@ -996,9 +1010,9 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dy, y, D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(dx, x, D1D)
|
||||
{
|
||||
real_t u[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
@@ -1020,11 +1034,11 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
{
|
||||
if (ACCUMULATE)
|
||||
{
|
||||
y(dx,dy,dz,e) += u[dz];
|
||||
y(dx, dy, dz, e) += u[dz];
|
||||
}
|
||||
else
|
||||
{
|
||||
y(dx,dy,dz,e) = u[dz];
|
||||
y(dx, dy, dz, e) = u[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1115,8 +1129,8 @@ inline void PAMassApply3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
// Shared memory PA Mass Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int TBATCH=1>
|
||||
inline void SmemPAMassApply3D(const int NE,
|
||||
const Array<real_t> &b_,
|
||||
const Array<real_t> &bt_,
|
||||
@@ -1126,6 +1140,9 @@ inline void SmemPAMassApply3D(const int NE,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static_assert(T_D1D > 0, "T_D1D must be positive");
|
||||
static_assert(T_Q1D > 0, "T_Q1D must be positive");
|
||||
static_assert(TBATCH > 0, "TBATCH must be positive");
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -1137,9 +1154,11 @@ inline void SmemPAMassApply3D(const int NE,
|
||||
const auto d = d_.Read();
|
||||
const auto x = x_.Read();
|
||||
auto y = y_.ReadWrite();
|
||||
mfem::forall_2D<T_Q1D*T_Q1D>(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
mfem::forall_2D_batch<T_Q1D * T_Q1D * TBATCH>(NE, Q1D, Q1D, TBATCH,
|
||||
[=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
|
||||
internal::SmemPAMassApply3D_Element<T_D1D, T_Q1D, TBATCH>(e, NE, b, d, x,
|
||||
y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1389,42 +1408,57 @@ using ApplyKernelType = MassIntegrator::ApplyKernelType;
|
||||
using DiagonalKernelType = MassIntegrator::DiagonalKernelType;
|
||||
}
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
ApplyKernelType MassIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 1) { return internal::PAMassApply1D; }
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPAMassApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassApply2D<D1D, Q1D>; }
|
||||
else if constexpr (DIM == 3)
|
||||
{
|
||||
constexpr int MDQ = D1D >= Q1D ? D1D : Q1D;
|
||||
// max 64 threads in z limit in cuda and hip
|
||||
if constexpr (MDQ > 0)
|
||||
{
|
||||
return internal::SmemPAMassApply3D<D1D, Q1D,
|
||||
internal::mass::NBZ3D(MDQ)>;
|
||||
}
|
||||
}
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
inline ApplyKernelType MassIntegrator::ApplyPAKernels::Fallback(
|
||||
int DIM, int, int)
|
||||
int dim, int, int)
|
||||
{
|
||||
if (DIM == 1) { return internal::PAMassApply1D; }
|
||||
else if (DIM == 2) { return internal::PAMassApply2D; }
|
||||
else if (DIM == 3) { return internal::PAMassApply3D; }
|
||||
if (dim == 1) { return internal::PAMassApply1D; }
|
||||
else if (dim == 2) { return internal::PAMassApply2D; }
|
||||
else if (dim == 3) { return internal::PAMassApply3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
DiagonalKernelType MassIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
else if constexpr (DIM == 2) { return internal::SmemPAMassAssembleDiagonal2D<D1D, Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPAMassAssembleDiagonal3D<D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
inline DiagonalKernelType MassIntegrator::DiagonalPAKernels::Fallback(
|
||||
int DIM, int, int)
|
||||
int dim, int, int)
|
||||
{
|
||||
if (DIM == 1) { return internal::PAMassAssembleDiagonal1D; }
|
||||
else if (DIM == 2) { return internal::PAMassAssembleDiagonal2D; }
|
||||
else if (DIM == 3) { return internal::PAMassAssembleDiagonal3D; }
|
||||
if (dim == 1) { return internal::PAMassAssembleDiagonal1D; }
|
||||
else if (dim == 2) { return internal::PAMassAssembleDiagonal2D; }
|
||||
else if (dim == 3) { return internal::PAMassAssembleDiagonal3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
#include "../qfunction.hpp"
|
||||
#include "../ceed/integrators/mass/mass.hpp"
|
||||
#include "bilininteg_mass_kernels.hpp"
|
||||
#include "bilininteg_mass_pa_simplices.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -29,9 +30,11 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
dim = mesh->Dimension();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
ElementTransformation *T0 = mesh->GetTypicalElementTransformation();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0);
|
||||
const bool stroud = fes.UsesRaggedTensorBasis();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0, stroud);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
@@ -48,17 +51,25 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
return;
|
||||
}
|
||||
int map_type = el.GetMapType();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetMesh()->GetNE();
|
||||
nq = ir->GetNPoints();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::DETERMINANTS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
if (stroud)
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
|
||||
}
|
||||
else
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
}
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(ne*nq, mt);
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
|
||||
// QuadratureSpace expects ir defined in reference simplex for Bernstein
|
||||
// elements with partial assembly
|
||||
{
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
@@ -147,9 +158,10 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int D1D = dofs1D;
|
||||
const int Q1D = quad1D;
|
||||
const Vector &D = pa_data;
|
||||
const Array<real_t> &B = maps->B;
|
||||
const Array<real_t> &Bt = maps->Bt;
|
||||
const Vector &D = pa_data;
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
if (DeviceCanUseOcca())
|
||||
{
|
||||
@@ -164,7 +176,31 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
|
||||
|
||||
if (fespace->UsesRaggedTensorBasis())
|
||||
{
|
||||
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
|
||||
|
||||
const Array<real_t> &Ba1 = rmaps->Ba1;
|
||||
const Array<real_t> &Ba2 = rmaps->Ba2;
|
||||
const Array<real_t> &Ba3 = rmaps->Ba3;
|
||||
const Array<real_t> &Ba1t = rmaps->Ba1t;
|
||||
const Array<real_t> &Ba2t = rmaps->Ba2t;
|
||||
const Array<real_t> &Ba3t = rmaps->Ba3t;
|
||||
const Array<int> &lex_map = rmaps->lex_map;
|
||||
const Array<int> &forward_map2d = rmaps->forward_map2d_mass;
|
||||
const Array<int> &inverse_map2d = rmaps->inverse_map2d_mass;
|
||||
const Array<int> &forward_map3d = rmaps->forward_map3d_mass;
|
||||
const Array<int> &inverse_map3d = rmaps->inverse_map3d_mass;
|
||||
ApplySimplexPAKernels::Run(dim, D1D, Q1D, ne, lex_map, forward_map2d,
|
||||
inverse_map2d,
|
||||
forward_map3d, inverse_map3d, Ba1, Ba2, Ba3, Ba1t, Ba2t, Ba3t,
|
||||
D, x, y, D1D, Q1D);
|
||||
}
|
||||
else
|
||||
{
|
||||
ApplyPAKernels::Run(dim, D1D, Q1D, ne, B, Bt, D, x, y, D1D, Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -177,6 +213,8 @@ void MassIntegrator::AddAbsMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_VERIFY(!fespace->UsesRaggedTensorBasis(),
|
||||
"AbsMultPA not implemented for ragged tensor basis");
|
||||
Vector abs_pa_data(pa_data);
|
||||
abs_pa_data.Abs();
|
||||
Array<real_t> absB(maps->B);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -236,6 +236,58 @@ IntegrationRule::ApplyToKnotIntervals(KnotVector const& kv) const
|
||||
return kvir;
|
||||
}
|
||||
|
||||
IntegrationRule IntegrationRule::Reorder(const Array<int> &ordering) const
|
||||
{
|
||||
const int np = GetNPoints();
|
||||
MFEM_VERIFY(np == ordering.Size(), "Invalid permutation size");
|
||||
IntegrationRule ir(np);
|
||||
ir.SetOrder(GetOrder());
|
||||
|
||||
for (int i = 0; i < np; i++)
|
||||
{
|
||||
IntegrationPoint &ip_new = ir.IntPoint(i);
|
||||
const IntegrationPoint &ip_old = IntPoint(ordering[i]);
|
||||
ip_new.Set(ip_old.x, ip_old.y, ip_old.z, ip_old.weight);
|
||||
}
|
||||
|
||||
return ir;
|
||||
}
|
||||
|
||||
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim)
|
||||
{
|
||||
IntegrationRule ir_mapped(ir.GetNPoints());
|
||||
ir_mapped.SetOrder(ir.GetOrder());
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
|
||||
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
|
||||
ip_mapped.x = ir.IntPoint(i).x;
|
||||
ip_mapped.weight = ir.IntPoint(i).weight;
|
||||
}
|
||||
return ir_mapped;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
IntegrationPoint &ip_mapped = ir_mapped.IntPoint(i);
|
||||
ip_mapped.z = ir.IntPoint(i).z * (1 - ir.IntPoint(i).x) * (1 - ir.IntPoint(
|
||||
i).y);
|
||||
ip_mapped.y = ir.IntPoint(i).y * (1 - ir.IntPoint(i).x);
|
||||
ip_mapped.x = ir.IntPoint(i).x;
|
||||
ip_mapped.weight = ir.IntPoint(i).weight;
|
||||
}
|
||||
return ir_mapped;
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Duffy transformation not implemented for this dimension!");
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPFR
|
||||
|
||||
// Class for computing hi-precision (HP) quadrature in 1D
|
||||
@@ -433,6 +485,142 @@ public:
|
||||
#endif // MFEM_USE_MPFR
|
||||
|
||||
|
||||
void QuadratureFunctions1D::GaussJacobi(const int np, const real_t alpha,
|
||||
const real_t beta, IntegrationRule* ir)
|
||||
{
|
||||
/* The np-point Gauss-Jacobi quadrature rule is exact for polynomials of
|
||||
degree 2np - 1 with weight function w(x) = (1-x)^alpha * x^beta. The
|
||||
nodes are the zeros of the Jacobi polynomial P_{np}^{alpha,beta} and
|
||||
the weights are
|
||||
|
||||
w_i = C / [(1 - x_i^2) * P'_{np}^{alpha,beta}(x_i)^2]
|
||||
C = 2^{alpha + beta + 1} * Gamma(np + alpha + 1) * Gamma(np + beta + 1)
|
||||
/ [Gamma(np + alpha + beta + 1) * Gamma(np + 1)].
|
||||
|
||||
The nodes are computed via nonlinear solve (Newton's method) with an
|
||||
initial guess corresponding to Gatteschi's asymptotic expansions of the
|
||||
Jacobi polynomial roots [1].
|
||||
|
||||
The current initial guess has been tested and performs well for
|
||||
np <= 200 and -1 <= alpha, beta <= 4. For larger np, it may be necessary
|
||||
utilize different initial guesses in the vicinity of x = -1,+1 [2].
|
||||
|
||||
[1] Gautschi, W., & Giordano, C. (2008). Luigi Gatteschi’s work on
|
||||
asymptotics of special functions and their zeros. Numerical Algorithms,
|
||||
49, 11-31.
|
||||
[2] Hale, N., & Townsend, A. (2013). Fast and accurate computation of
|
||||
Gauss--Legendre and Gauss--Jacobi quadrature nodes and weights.
|
||||
SIAM Journal on Scientific Computing, 35(2), A652-A674.
|
||||
*/
|
||||
ir->SetSize(np);
|
||||
ir->SetPointIndices();
|
||||
ir->SetOrder(2*np - 1);
|
||||
|
||||
if (alpha <= -1.0 || beta <= -1.0)
|
||||
{
|
||||
MFEM_ABORT("Gauss-Jacobi quadrature only defined for alpha > -1 and beta > -1");
|
||||
}
|
||||
// Jacobi weight function is undefined whenever alpha <= -1 or beta <= -1
|
||||
|
||||
if (alpha > 4.0 || beta > 4.0)
|
||||
{
|
||||
MFEM_ABORT("Current Gauss-Jacobi quadrature implementation only tested for alpha <= 4 and beta <= 4");
|
||||
}
|
||||
// current asymptotic expansions for initial guess may perform poorly for large alpha, beta
|
||||
|
||||
switch (np)
|
||||
{
|
||||
case 1:
|
||||
real_t x = (beta - alpha) / (alpha + beta + 2);
|
||||
real_t w = pow(2, alpha + beta + 1) * tgamma(alpha + 2) * tgamma(
|
||||
beta + 2) / (tgamma(alpha + beta + 2));
|
||||
w = 0.5 * w / pow(2, alpha + beta);
|
||||
// map weight to to [0,1], with additional 1/(2^(alpha + beta)) factor coming from mapping
|
||||
// the weight (1-x)^alpha * (1+x)^beta to [0,1] as well.
|
||||
ir->IntPoint(0).Set1w(0.5 * x + 0.5,
|
||||
4.0 * w / ((1.0 - x*x) * (alpha + beta + 2) * (alpha + beta + 2)));
|
||||
return;
|
||||
}
|
||||
|
||||
#ifndef MFEM_USE_MPFR
|
||||
|
||||
const int n = np;
|
||||
// common constants for Jacobi polynomials
|
||||
real_t ab = alpha + beta;
|
||||
real_t a2_minus_b2 = (alpha - beta) * (alpha + beta);
|
||||
|
||||
// roots of P^(alpha,beta)_n in the interval [-1,1]
|
||||
for (int i = 1; i <= n; i++)
|
||||
{
|
||||
// rather than using Chebyshev points for initial guess, use Gatteschi's asymptotic expansion for roots of Jacobi
|
||||
// polynomials
|
||||
real_t n_ab_plus_1 = 2 * n + alpha + beta + 1;
|
||||
real_t v = (2 * i + alpha - 0.5) * M_PI / n_ab_plus_1;
|
||||
real_t theta = v + 1.0 / (n_ab_plus_1*n_ab_plus_1) * ((0.25 - alpha*alpha) *
|
||||
1.0/tan(0.5*v) - (0.25 - beta*beta) * tan(0.5*v));
|
||||
real_t z = cos(theta);
|
||||
|
||||
real_t pp, p1, dz, xi = 0.;
|
||||
bool done = false;
|
||||
while (1)
|
||||
{
|
||||
real_t p2 = 1;
|
||||
p1 = ((alpha-beta) + (alpha + beta + 2) * z) / 2;
|
||||
for (int j = 1; j <= n-1; j++)
|
||||
{
|
||||
real_t p3 = p2;
|
||||
p2 = p1;
|
||||
|
||||
real_t jx2_ab = 2 * j + ab;
|
||||
real_t an = (jx2_ab) * (jx2_ab + 2);
|
||||
real_t bn = a2_minus_b2;
|
||||
real_t cn = 2 * (j + alpha) * (j + beta) * (jx2_ab + 2) / (jx2_ab + 1);
|
||||
|
||||
real_t D = (jx2_ab + 1) / (2 * (j + 1) * (j + ab + 1) * (jx2_ab));
|
||||
p1 = ((an * z + bn) * p2 - cn * p3) * D;
|
||||
}
|
||||
// p1 is Jacobi polynomial
|
||||
pp = n * (alpha - beta - (2 * n + ab) * z) * p1 + 2 * (n + alpha) *
|
||||
(n + beta) * p2;
|
||||
pp = pp / ((2 * n + ab) * (1 - z*z));
|
||||
// derivative of the Jacobi polynomial
|
||||
if (done) { break; }
|
||||
|
||||
dz = p1/pp;
|
||||
#ifdef MFEM_USE_SINGLE
|
||||
if (std::abs(dz) < 1e-7)
|
||||
#elif defined MFEM_USE_DOUBLE
|
||||
if (std::abs(dz) < std::numeric_limits<real_t>::epsilon())
|
||||
// this seems to cause trouble if we try std::abs(dz) < 1e-16
|
||||
#else
|
||||
MFEM_ABORT("Floating point type undefined");
|
||||
// if (std::abs(dz) < 1e-16)
|
||||
#endif
|
||||
{
|
||||
done = true;
|
||||
xi = z - dz;
|
||||
}
|
||||
z -= dz;
|
||||
}
|
||||
real_t c0 = exp(lgamma(n + alpha + 1) - lgamma(n + ab + 1)) * exp(lgamma(
|
||||
n + beta + 1) - lgamma(n + 1));
|
||||
// ratio of gamma functions prone to overflow for large n, so compute logarithms
|
||||
// of Gamma function instead, i.e. Gamma(a)/Gamma(b) = exp(lgamma(a) - lgamma(b))
|
||||
ir->IntPoint(n-i).x = 0.5 * xi + 0.5;
|
||||
ir->IntPoint(n-i).weight = 0.5 * c0 * pow(2.0,
|
||||
ab + 1) / ((1.0 - xi*xi)*pp*pp) / pow(2, ab);
|
||||
// map nodes and weights to the interval [0,1]
|
||||
}
|
||||
|
||||
#else // MFEM_USE_MPFR is defined
|
||||
|
||||
MFEM_ABORT("MPFR implementation of Gauss-Jacobi quadrature not defined yet");
|
||||
|
||||
#endif // MFEM_USE_MPFR
|
||||
|
||||
}
|
||||
|
||||
|
||||
void QuadratureFunctions1D::GaussLegendre(const int np, IntegrationRule* ir)
|
||||
{
|
||||
ir->SetSize(np);
|
||||
@@ -2362,6 +2550,194 @@ IntegrationRule *IntegrationRules::CubeIntegrationRule(int Order)
|
||||
return CubeIntRules[Order];
|
||||
}
|
||||
|
||||
StroudIntegrationRules StroudIntRules;
|
||||
|
||||
StroudIntegrationRules::StroudIntegrationRules()
|
||||
{
|
||||
const MemoryType h_mt = MemoryType::HOST;
|
||||
SquareStroudIntRules.SetSize(32, h_mt);
|
||||
SquareStroudIntRules = NULL;
|
||||
|
||||
TriangleStroudIntRules.SetSize(32, h_mt);
|
||||
TriangleStroudIntRules = NULL;
|
||||
|
||||
CubeStroudIntRules.SetSize(32, h_mt);
|
||||
CubeStroudIntRules = NULL;
|
||||
|
||||
TetrahedronStroudIntRules.SetSize(32, h_mt);
|
||||
TetrahedronStroudIntRules = NULL;
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
IntRuleLocks.SetSize(Geometry::NUM_GEOMETRIES, h_mt);
|
||||
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
|
||||
{
|
||||
omp_init_lock(&IntRuleLocks[i]);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
const IntegrationRule &StroudIntegrationRules::Get(int GeomType, int Order)
|
||||
{
|
||||
Array<IntegrationRule *> *ir_array = NULL;
|
||||
|
||||
switch (GeomType)
|
||||
{
|
||||
case Geometry::TRIANGLE: ir_array = &TriangleStroudIntRules; break;
|
||||
case Geometry::TETRAHEDRON: ir_array = &TetrahedronStroudIntRules; break;
|
||||
case Geometry::INVALID:
|
||||
case Geometry::NUM_GEOMETRIES:
|
||||
MFEM_ABORT("Unknown type of reference element!");
|
||||
default:
|
||||
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
|
||||
}
|
||||
|
||||
if (Order < 0)
|
||||
{
|
||||
Order = 0;
|
||||
}
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
omp_set_lock(&IntRuleLocks[GeomType]);
|
||||
#endif
|
||||
|
||||
if (!HaveIntRule(*ir_array, Order))
|
||||
{
|
||||
IntegrationRule *ir = GenerateIntegrationRule(GeomType, Order);
|
||||
#ifdef MFEM_DEBUG
|
||||
int RealOrder = Order;
|
||||
while (RealOrder+1 < ir_array->Size() && (*ir_array)[RealOrder+1] == ir)
|
||||
{
|
||||
RealOrder++;
|
||||
}
|
||||
MFEM_VERIFY(RealOrder == ir->GetOrder(), "internal error");
|
||||
#else
|
||||
MFEM_CONTRACT_VAR(ir);
|
||||
#endif
|
||||
}
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
omp_unset_lock(&IntRuleLocks[GeomType]);
|
||||
#endif
|
||||
|
||||
return *(*ir_array)[Order];
|
||||
}
|
||||
|
||||
void StroudIntegrationRules::DeleteIntRuleArray(
|
||||
Array<IntegrationRule *> &ir_array) const
|
||||
{
|
||||
// Many of the intrules have multiple contiguous copies in the ir_array
|
||||
// so we have to be careful to not delete them twice.
|
||||
IntegrationRule *ir = NULL;
|
||||
for (int i = 0; i < ir_array.Size(); i++)
|
||||
{
|
||||
if (ir_array[i] != NULL && ir_array[i] != ir)
|
||||
{
|
||||
ir = ir_array[i];
|
||||
delete ir;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
StroudIntegrationRules::~StroudIntegrationRules()
|
||||
{
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
for (int i = 0; i < Geometry::NUM_GEOMETRIES; i++)
|
||||
{
|
||||
omp_destroy_lock(&IntRuleLocks[i]);
|
||||
}
|
||||
#endif
|
||||
DeleteIntRuleArray(SquareStroudIntRules);
|
||||
DeleteIntRuleArray(TriangleStroudIntRules);
|
||||
DeleteIntRuleArray(CubeStroudIntRules);
|
||||
DeleteIntRuleArray(TetrahedronStroudIntRules);
|
||||
}
|
||||
|
||||
|
||||
IntegrationRule *StroudIntegrationRules::GenerateIntegrationRule(int GeomType,
|
||||
int Order)
|
||||
{
|
||||
switch (GeomType)
|
||||
{
|
||||
case Geometry::TRIANGLE:
|
||||
return TriangleStroudIntegrationRule(Order);
|
||||
case Geometry::TETRAHEDRON:
|
||||
return TetrahedronStroudIntegrationRule(Order);
|
||||
case Geometry::INVALID:
|
||||
case Geometry::NUM_GEOMETRIES:
|
||||
MFEM_ABORT("Unknown type of reference element!");
|
||||
default:
|
||||
MFEM_ABORT("Stroud rules only valid for triangular and tetrahedral elements!");
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Integration rule in reference triangle according to tensor product Gauss-Jacobi rule.
|
||||
The nodes and weights are used in the original form defined on the reference
|
||||
square to evaluate the component 1D basis functions. Mapping to the reference
|
||||
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
|
||||
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
|
||||
IntegrationRule *StroudIntegrationRules::TriangleStroudIntegrationRule(
|
||||
int Order)
|
||||
{
|
||||
int RealOrder = GetSegmentRealOrder(Order);
|
||||
// Order is one of {RealOrder-1,RealOrder}
|
||||
// if (!HaveIntRule(SegmentIntRules, RealOrder))
|
||||
// {
|
||||
// SegmentIntegrationRule(RealOrder);
|
||||
// }
|
||||
IntegrationRule ir_0_0;
|
||||
// Gauss-Jacobi is exact for 2*n-1
|
||||
int n = RealOrder/2 + 1;
|
||||
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
|
||||
|
||||
IntegrationRule ir_1_0;
|
||||
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
|
||||
|
||||
AllocIntRule(TriangleStroudIntRules, RealOrder); // RealOrder >= Order
|
||||
// create rule in unit square
|
||||
TriangleStroudIntRules[RealOrder-1] =
|
||||
TriangleStroudIntRules[RealOrder] =
|
||||
new IntegrationRule(ir_1_0, ir_0_0);
|
||||
// map rule to reference triangle
|
||||
// TriangleStroudIntRules[RealOrder-1]->DuffyTrans(2);
|
||||
*TriangleStroudIntRules[RealOrder-1] =
|
||||
DuffyTrans(*TriangleStroudIntRules[RealOrder-1], 2);
|
||||
return TriangleStroudIntRules[Order];
|
||||
}
|
||||
|
||||
/* Integration rule in reference tetrahedron according to tensor product Gauss-Jacobi rule.
|
||||
The nodes and weights are used in the original form defined on the reference
|
||||
square to evaluate the component 1D basis functions. Mapping to the reference
|
||||
triangle via IntegrationRule::DuffyTrans() occurs only in evaluation of coefficient
|
||||
vectors, see e.g. MassIntegrator::AssemblePASimplex. */
|
||||
IntegrationRule *StroudIntegrationRules::TetrahedronStroudIntegrationRule(
|
||||
int Order)
|
||||
{
|
||||
int RealOrder = GetSegmentRealOrder(Order);
|
||||
// Order is one of {RealOrder-1,RealOrder}
|
||||
|
||||
IntegrationRule ir_0_0;
|
||||
int n = RealOrder/2 + 1;
|
||||
QuadratureFunctions1D::GaussJacobi(n, 0.0, 0.0, &ir_0_0);
|
||||
|
||||
IntegrationRule ir_1_0;
|
||||
QuadratureFunctions1D::GaussJacobi(n, 1.0, 0.0, &ir_1_0);
|
||||
|
||||
IntegrationRule ir_2_0;
|
||||
QuadratureFunctions1D::GaussJacobi(n, 2.0, 0.0, &ir_2_0);
|
||||
|
||||
AllocIntRule(TetrahedronStroudIntRules, RealOrder); // RealOrder >= Order
|
||||
// create rule in unit cube
|
||||
TetrahedronStroudIntRules[RealOrder-1] =
|
||||
TetrahedronStroudIntRules[RealOrder] =
|
||||
new IntegrationRule(ir_2_0, ir_1_0, ir_0_0);
|
||||
// map rule to reference tetrahedron
|
||||
// TetrahedronStroudIntRules[RealOrder-1]->DuffyTrans(3);
|
||||
*TetrahedronStroudIntRules[RealOrder-1] =
|
||||
DuffyTrans(*TetrahedronStroudIntRules[RealOrder-1], 3);
|
||||
return TetrahedronStroudIntRules[Order];
|
||||
}
|
||||
|
||||
IntegrationRule& NURBSMeshRules::GetElementRule(const int elem,
|
||||
const int patch, const int *ijk,
|
||||
Array<const KnotVector*> const& kv) const
|
||||
|
||||
@@ -269,6 +269,13 @@ public:
|
||||
/// applying this rule on each knot interval.
|
||||
IntegrationRule* ApplyToKnotIntervals(KnotVector const& kv) const;
|
||||
|
||||
/** @brief Returns an integration rule such that the new IntegrationPoints
|
||||
* are re-ordered based on @a ordering.
|
||||
*
|
||||
* @details In the new integration rule, ip_new[i] = ip_old[ordering[i]]
|
||||
*/
|
||||
IntegrationRule Reorder(const Array<int> &ordering) const;
|
||||
|
||||
/// Destroys an IntegrationRule object
|
||||
~IntegrationRule() { }
|
||||
};
|
||||
@@ -378,6 +385,8 @@ public:
|
||||
These methods calculate the actual points and weights for the different
|
||||
types of quadrature rules. */
|
||||
///@{
|
||||
static void GaussJacobi(const int np, const real_t alpha, const real_t beta,
|
||||
IntegrationRule* ir);
|
||||
static void GaussLegendre(const int np, IntegrationRule* ir);
|
||||
static void GaussLobatto(const int np, IntegrationRule *ir);
|
||||
static void OpenUniform(const int np, IntegrationRule *ir);
|
||||
@@ -487,12 +496,71 @@ public:
|
||||
~IntegrationRules();
|
||||
};
|
||||
|
||||
/// Container class for integration rules
|
||||
class StroudIntegrationRules
|
||||
{
|
||||
private:
|
||||
Array<IntegrationRule *> SquareStroudIntRules;
|
||||
Array<IntegrationRule *> TriangleStroudIntRules;
|
||||
Array<IntegrationRule *> CubeStroudIntRules;
|
||||
Array<IntegrationRule *> TetrahedronStroudIntRules;
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
Array<omp_lock_t> IntRuleLocks;
|
||||
#endif
|
||||
|
||||
void AllocIntRule(Array<IntegrationRule *> &ir_array, int Order) const
|
||||
{
|
||||
if (ir_array.Size() <= Order)
|
||||
{
|
||||
ir_array.SetSize(Order + 1, NULL);
|
||||
}
|
||||
}
|
||||
bool HaveIntRule(Array<IntegrationRule *> &ir_array, int Order) const
|
||||
{
|
||||
return (ir_array.Size() > Order && ir_array[Order] != NULL);
|
||||
}
|
||||
int GetSegmentRealOrder(int Order) const
|
||||
{
|
||||
return Order | 1; // valid for all quad_type's
|
||||
}
|
||||
void DeleteIntRuleArray(Array<IntegrationRule *> &ir_array) const;
|
||||
|
||||
/// The following methods allocate new IntegrationRule objects without
|
||||
/// checking if they already exist. To avoid memory leaks use
|
||||
/// IntegrationRules::Get(int GeomType, int Order) instead.
|
||||
IntegrationRule *GenerateIntegrationRule(int GeomType, int Order);
|
||||
IntegrationRule *TriangleStroudIntegrationRule(int Order);
|
||||
IntegrationRule *TetrahedronStroudIntegrationRule(int Order);
|
||||
|
||||
public:
|
||||
/// Sets initial sizes for the integration rule arrays, but rules
|
||||
/// are defined the first time they are requested with the Get method.
|
||||
explicit StroudIntegrationRules();
|
||||
|
||||
/// Returns a Stroud integration rule for given GeomType and Order.
|
||||
const IntegrationRule &Get(int GeomType, int Order);
|
||||
|
||||
/// Destroys an StroudIntegrationRules object
|
||||
~StroudIntegrationRules();
|
||||
};
|
||||
|
||||
/// A global object with all integration rules (defined in intrules.cpp)
|
||||
extern MFEM_EXPORT IntegrationRules IntRules;
|
||||
|
||||
/// A global object with all refined integration rules
|
||||
extern MFEM_EXPORT IntegrationRules RefinedIntRules;
|
||||
|
||||
/// A global object with all Stroud integration rules (defined in intrules.cpp)
|
||||
extern MFEM_EXPORT StroudIntegrationRules StroudIntRules;
|
||||
|
||||
/// Duffy Transformation of 2D and 3D tensor product rules of the form
|
||||
/// $X(t) = \sum_{i=1}^{d+1} \lambda_i(t) * x_i$, where $x_i$ are the vertices
|
||||
/// of the simplex and $\lambda_i = t_i * (1-\lambda_1-...-\lambda_{i-1})$, with
|
||||
/// $t$ being the coordinates in the unit square/cube. This function is used only
|
||||
/// in the partial assembly of Bernstein elements on simplices and does NOT
|
||||
/// modify the quadrature weights.
|
||||
IntegrationRule DuffyTrans(const IntegrationRule &ir, int dim);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -207,28 +207,6 @@ inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input DIM vector at element offset into given register tensor
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int d1d, const int c,
|
||||
const DeviceTensor<4, const real_t> &X,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
|
||||
{
|
||||
for (int d = 0; d < DIM; d++)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
Y[c][d][dz][dy][dx] = X(dx, dy, dz, c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Load 3D input VDIM*DIM vector into given register tensor, specific component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d, const int c,
|
||||
@@ -354,28 +332,6 @@ inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// Write 3D DIM vector into given device tensor for specific component
|
||||
template <int VDIM, int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int d1d, const int c,
|
||||
vd_regs3d_t<VDIM, DIM, MQ1> &X,
|
||||
DeviceTensor<4, real_t> &Y)
|
||||
{
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
|
||||
{
|
||||
for (int d = 0; d < DIM; ++d)
|
||||
{
|
||||
Y(dx, dy, dz, c) += X(c, d, dz, dy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
/// 2D scalar contraction, X direction
|
||||
template <bool Transpose, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void ContractX2d(const int d1d, const int q1d,
|
||||
|
||||
@@ -1,332 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#pragma once
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include "kernels.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem::kernels::internal::low
|
||||
{
|
||||
|
||||
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
|
||||
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
|
||||
template <int DIM, int N>
|
||||
// struct regs3d_device_wrapper: mfem::future::tensor<real_t, DIM, 0, 0, 0> {};
|
||||
struct regs3d_device_wrapper: mfem::future::tensor<real_t, 0, 0, 0, DIM> {};
|
||||
template <int DIM, int N>
|
||||
using regs3d_t = regs3d_device_wrapper<DIM, N>;
|
||||
#else
|
||||
template <int DIM, int N>
|
||||
using regs3d_t = mfem::future::tensor<real_t, N, N, N, DIM>;
|
||||
// using regs3d_t = mfem::future::tensor<real_t, DIM, N, N, N>;
|
||||
#endif
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// Load 2D matrix into shared memory
|
||||
template <int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
|
||||
const real_t *M, real_t (*N)[MQ1])
|
||||
{
|
||||
if (MFEM_THREAD_ID(z) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
|
||||
{
|
||||
N[dy][qx] = M[dy * q1d + qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
|
||||
const DeviceTensor<5, const real_t> &XE,
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
sm0[dz][dy][dx][0] = XE(dx, dy, dz, 0, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 1/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradX(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dx = 0; dx < d1d; ++dx)
|
||||
{
|
||||
const auto x = sm0[dz][dy][dx][0];
|
||||
u = std::fma(B[dx][qx], x, u);
|
||||
v = std::fma(G[dx][qx], x, v);
|
||||
}
|
||||
sm1[dz][dy][qx][0] = u;
|
||||
sm1[dz][dy][qx][1] = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 2/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradY(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dy = 0; dy < d1d; ++dy)
|
||||
{
|
||||
u = std::fma(sm1[dz][dy][qx][1], B[dy][qy], u);
|
||||
v = std::fma(sm1[dz][dy][qx][0], G[dy][qy], v);
|
||||
w = std::fma(sm1[dz][dy][qx][0], B[dy][qy], w);
|
||||
}
|
||||
sm0[dz][qy][qx][0] = u;
|
||||
sm0[dz][qy][qx][1] = v;
|
||||
sm0[dz][qy][qx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradZ(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
const real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
real_t u[3] = {0.0, 0.0, 0.0};
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int dz = 0; dz < d1d; ++dz)
|
||||
{
|
||||
u[0] = std::fma(B[dz][qz], sm0[dz][qy][qx][0], u[0]);
|
||||
u[1] = std::fma(B[dz][qz], sm0[dz][qy][qx][1], u[1]);
|
||||
u[2] = std::fma(G[dz][qz], sm0[dz][qy][qx][2], u[2]);
|
||||
}
|
||||
reg[qz][qy][qx][0] = u[0];
|
||||
reg[qz][qy][qx][1] = u[1];
|
||||
reg[qz][qy][qx][2] = u[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D scalar gradient
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
GradX(d1d, q1d, B, G, sm0, sm1); // Grad X
|
||||
GradY(d1d, q1d, B, G, sm1, sm0); // Grad Y
|
||||
GradZ(d1d, q1d, B, G, sm0, reg); // Grad Z
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 1/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dX(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx,x,q1d)
|
||||
{
|
||||
sm1[qz][qy][qx][0] = reg[qz][qy][qx][0];
|
||||
sm1[qz][qy][qx][1] = reg[qz][qy][qx][1];
|
||||
sm1[qz][qy][qx][2] = reg[qz][qy][qx][2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy,y,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qx = 0; qx < q1d; ++qx)
|
||||
{
|
||||
u = std::fma(sm1[qz][qy][qx][0], G[dx][qx], u);
|
||||
v = std::fma(sm1[qz][qy][qx][1], B[dx][qx], v);
|
||||
w = std::fma(sm1[qz][qy][qx][2], B[dx][qx], w);
|
||||
}
|
||||
sm0[qz][qy][dx][0] = u;
|
||||
sm0[qz][qy][dx][1] = v;
|
||||
sm0[qz][qy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 2/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dY(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz,z,q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qy = 0; qy < q1d; ++qy)
|
||||
{
|
||||
u = std::fma(sm0[qz][qy][dx][0], B[dy][qy], u);
|
||||
v = std::fma(sm0[qz][qy][dx][1], G[dy][qy], v);
|
||||
w = std::fma(sm0[qz][qy][dx][2], B[dy][qy], w);
|
||||
}
|
||||
sm1[qz][dy][dx][0] = u;
|
||||
sm1[qz][dy][dx][1] = v;
|
||||
sm1[qz][dy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3dZ(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
regs3d_t<DIM,MQ1> ®)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
real_t u = 0.0, v = 0.0, w = 0.0;
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < q1d; ++qz)
|
||||
{
|
||||
u = std::fma(sm1[qz][dy][dx][0], B[dz][qz], u);
|
||||
v = std::fma(sm1[qz][dy][dx][1], B[dz][qz], v);
|
||||
w = std::fma(sm1[qz][dy][dx][2], G[dz][qz], w);
|
||||
}
|
||||
reg[dz][dy][dx][0] = u;
|
||||
reg[dz][dy][dx][1] = v;
|
||||
reg[dz][dy][dx][2] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D scalar gradient transposed
|
||||
template <int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
|
||||
const real_t (*B)[MQ1],
|
||||
const real_t (*G)[MQ1],
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
real_t (&sm1)[MQ1][MQ1][MQ1][DIM],
|
||||
real_t (&sm0)[MQ1][MQ1][MQ1][DIM])
|
||||
{
|
||||
GradTranspose3dX(d1d, q1d, B, G, reg, sm1, sm0); // Grad^T X
|
||||
GradTranspose3dY(d1d, q1d, B, G, sm0, sm1); // Grad^T Y
|
||||
GradTranspose3dZ(d1d, q1d, B, G, sm1, reg); // Grad^T Z
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
/// 3D Scalar Gradient Transposed, 3/3
|
||||
template<int DIM, int MQ1>
|
||||
inline MFEM_HOST_DEVICE void WriteDofs3d(const int d1d,
|
||||
const int c, const int e,
|
||||
regs3d_t<DIM,MQ1> ®,
|
||||
const DeviceTensor<5, real_t> &YE)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dz,z,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dy,y,d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(dx,x,d1d)
|
||||
{
|
||||
const real_t u = reg[dz][dy][dx][0];
|
||||
const real_t v = reg[dz][dy][dx][1];
|
||||
const real_t w = reg[dz][dy][dx][2];
|
||||
YE(dx, dy, dz, c, e) += (u + v + w);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::kernels::internal
|
||||
+4
-4
@@ -284,12 +284,12 @@ GeometricMultigrid::GeometricMultigrid(
|
||||
ownedProlongations.SetSize(nlevels - 1);
|
||||
ownedProlongations = have_ess_bdr;
|
||||
|
||||
if (have_ess_bdr)
|
||||
essentialTrueDofs.SetSize(nlevels);
|
||||
for (int level = 0; level < nlevels; ++level)
|
||||
{
|
||||
essentialTrueDofs.SetSize(nlevels);
|
||||
for (int level = 0; level < nlevels; ++level)
|
||||
essentialTrueDofs[level] = new Array<int>;
|
||||
if (have_ess_bdr)
|
||||
{
|
||||
essentialTrueDofs[level] = new Array<int>;
|
||||
fespaces.GetFESpaceAtLevel(level).GetEssentialTrueDofs(
|
||||
ess_bdr, *essentialTrueDofs[level]);
|
||||
}
|
||||
|
||||
+1
-2
@@ -187,8 +187,7 @@ public:
|
||||
/// mesh boundary element attributes that define the essential DOFs.
|
||||
///
|
||||
/// If @a ess_bdr is empty, or all its entries are 0, then no essential
|
||||
/// boundary conditions are imposed and the protected array essentialTrueDofs
|
||||
/// remains empty.
|
||||
/// boundary conditions are imposed.
|
||||
GeometricMultigrid(const FiniteElementSpaceHierarchy& fespaces_,
|
||||
const Array<int> &ess_bdr);
|
||||
|
||||
|
||||
@@ -224,9 +224,6 @@ public:
|
||||
/** @see GetGradient(const Vector &) */
|
||||
Operator &GetGradient(const Vector &x, bool finalize) const;
|
||||
|
||||
/// Suppress a warning about hiding overloaded virtual function.
|
||||
using Operator::GetGradient;
|
||||
|
||||
/// Update the NonlinearForm to propagate updates of the associated FE space.
|
||||
/** After calling this method, the essential boundary conditions need to be
|
||||
set again. */
|
||||
|
||||
+31
-3
@@ -349,6 +349,7 @@ void ParFiniteElementSpace::GetGroupComm(
|
||||
}
|
||||
}
|
||||
|
||||
bool have_sign_flips = false;
|
||||
if (g_ldof_sign)
|
||||
{
|
||||
g_ldof_sign->SetSize(GetNDofs());
|
||||
@@ -428,6 +429,7 @@ void ParFiniteElementSpace::GetGroupComm(
|
||||
if (g_ldof_sign)
|
||||
{
|
||||
(*g_ldof_sign)[dofs[l]] = -1;
|
||||
have_sign_flips = true;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -466,6 +468,7 @@ void ParFiniteElementSpace::GetGroupComm(
|
||||
if (g_ldof_sign)
|
||||
{
|
||||
(*g_ldof_sign)[dofs[l]] = -1;
|
||||
have_sign_flips = true;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -504,6 +507,7 @@ void ParFiniteElementSpace::GetGroupComm(
|
||||
if (g_ldof_sign)
|
||||
{
|
||||
(*g_ldof_sign)[dofs[l]] = -1;
|
||||
have_sign_flips = true;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -527,12 +531,18 @@ void ParFiniteElementSpace::GetGroupComm(
|
||||
group_ldof.GetI()[gr+1] = group_ldof_counter;
|
||||
}
|
||||
|
||||
if (g_ldof_sign && have_sign_flips == false)
|
||||
{
|
||||
g_ldof_sign->DeleteAll();
|
||||
}
|
||||
|
||||
gc.Finalize();
|
||||
}
|
||||
|
||||
void ParFiniteElementSpace::ApplyLDofSigns(Array<int> &dofs) const
|
||||
{
|
||||
MFEM_ASSERT(Conforming(), "wrong code path");
|
||||
if (!HaveDofSigns()) { return; }
|
||||
|
||||
for (int i = 0; i < dofs.Size(); i++)
|
||||
{
|
||||
@@ -559,6 +569,24 @@ void ParFiniteElementSpace::ApplyLDofSigns(Table &el_dof) const
|
||||
ApplyLDofSigns(all_dofs);
|
||||
}
|
||||
|
||||
void ParFiniteElementSpace::ApplyDofSigns(real_t *h_data) const
|
||||
{
|
||||
if (!HaveDofSigns()) { return; }
|
||||
|
||||
const bool byvdim = (ordering == Ordering::byVDIM);
|
||||
for (int i = 0; i < ndofs; i++)
|
||||
{
|
||||
if (ldof_sign[i] < 0)
|
||||
{
|
||||
for (int d = 0; d < vdim; d++)
|
||||
{
|
||||
const int idx = byvdim ? d+vdim*i : i+ndofs*d;
|
||||
h_data[idx] = -h_data[idx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ParFiniteElementSpace::GetElementDofs(int i, Array<int> &dofs,
|
||||
DofTransformation &doftrans) const
|
||||
{
|
||||
@@ -1193,15 +1221,15 @@ void ParFiniteElementSpace::GetEssentialTrueDofsVar(const Array<int>
|
||||
MFEM_VERIFY(IsVariableOrder() && R,
|
||||
"GetEssentialTrueDofsVar is only for variable-order spaces");
|
||||
|
||||
true_ess_dofs.SetSize(R->Height(), Device::GetDeviceMemoryType());
|
||||
true_ess_dofs.SetSize(R->Height());
|
||||
true_ess_dofs.HostWrite();
|
||||
true_ess_dofs = 0;
|
||||
|
||||
const int ntdofs = tdof2ldof.Size();
|
||||
MFEM_VERIFY(vdim * ntdofs == R->NumRows() &&
|
||||
vdim * ntdofs == true_ess_dofs.Size(), "");
|
||||
MFEM_VERIFY(ldof_ltdof.Size() == ndofs && ess_dofs.Size() == vdim * ndofs, "");
|
||||
|
||||
true_ess_dofs = 0;
|
||||
|
||||
const bool bynodes = (ordering == Ordering::byNODES);
|
||||
const int vdim_factor = bynodes ? 1 : vdim;
|
||||
const int num_true_dofs = R->NumRows() / vdim;
|
||||
|
||||
+14
-2
@@ -340,8 +340,20 @@ public:
|
||||
|
||||
inline ParMesh *GetParMesh() const { return pmesh; }
|
||||
|
||||
int GetDofSign(int i)
|
||||
{ return NURBSext || Nonconforming() ? 1 : ldof_sign[VDofToDof(i)]; }
|
||||
/** @brief Return true if the parallel FE space has DOFs with signs opposite
|
||||
of the DOFs in the respective serial FE space. */
|
||||
bool HaveDofSigns() const { return ldof_sign.Size() != 0; }
|
||||
|
||||
/** @brief Apply the DOF signs to the given host data @a h_data which must be
|
||||
of size GetVSize() if HaveDofSigns() is true. If HaveDofSigns() is false,
|
||||
this method is no-op and returns immediately. */
|
||||
void ApplyDofSigns(real_t *h_data) const;
|
||||
|
||||
/** @brief Return -1 if the given (vector) DOF @a i has a sign opposite of
|
||||
the DOF in the respecive serial FE space. Otherwise, return 1. */
|
||||
int GetDofSign(int i) const
|
||||
{ return !HaveDofSigns() ? 1 : ldof_sign[VDofToDof(i)]; }
|
||||
|
||||
HYPRE_BigInt *GetDofOffsets() const { return dof_offsets; }
|
||||
HYPRE_BigInt *GetTrueDofOffsets() const { return tdof_offsets; }
|
||||
HYPRE_BigInt GlobalVSize() const
|
||||
|
||||
+18
-9
@@ -80,6 +80,8 @@ ParGridFunction::ParGridFunction(ParMesh *pmesh, std::istream &input)
|
||||
fes->GetOrdering());
|
||||
delete fes;
|
||||
fes = pfes;
|
||||
|
||||
pfes->ApplyDofSigns(HostReadWrite());
|
||||
}
|
||||
|
||||
void ParGridFunction::Update()
|
||||
@@ -1082,18 +1084,17 @@ real_t ParGridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
|
||||
|
||||
void ParGridFunction::Save(std::ostream &os) const
|
||||
{
|
||||
real_t *data_ = const_cast<real_t*>(HostRead());
|
||||
for (int i = 0; i < size; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
|
||||
}
|
||||
// We use const_cast + HostRead (instead of HostReadWrite) because we only
|
||||
// need to change the host data temporarily and this way we do not invalidate
|
||||
// the data if it is on device. If we use HostReadWrite here, later calls to
|
||||
// Read or ReadWrite will need to copy the data from host to device. With the
|
||||
// approach used here, the host-to-device copy is avoided.
|
||||
real_t *h_data = const_cast<real_t*>(HostRead());
|
||||
pfes->ApplyDofSigns(h_data);
|
||||
|
||||
GridFunction::Save(os);
|
||||
|
||||
for (int i = 0; i < size; i++)
|
||||
{
|
||||
if (pfes->GetDofSign(i) < 0) { data_[i] = -data_[i]; }
|
||||
}
|
||||
pfes->ApplyDofSigns(h_data);
|
||||
}
|
||||
|
||||
void ParGridFunction::Save(const char *fname, int precision) const
|
||||
@@ -1264,7 +1265,13 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
|
||||
int *nfdofs = new int[NRanks];
|
||||
int *nrdofs = new int[NRanks];
|
||||
|
||||
// We use const_cast + HostRead (instead of HostReadWrite) because we only
|
||||
// need to change the host data temporarily and this way we do not invalidate
|
||||
// the data if it is on device. If we use HostReadWrite here, later calls to
|
||||
// Read or ReadWrite will need to copy the data from host to device. With the
|
||||
// approach used here, the host-to-device copy is avoided.
|
||||
real_t * h_data = const_cast<real_t *>(this->HostRead());
|
||||
pfes->ApplyDofSigns(h_data); // temporarily flip the dof signs
|
||||
|
||||
values[0] = h_data;
|
||||
nv[0] = pfes -> GetVSize();
|
||||
@@ -1371,6 +1378,8 @@ void ParGridFunction::SaveAsOne(std::ostream &os) const
|
||||
MPI_Send(h_data, nv[0], MPITypeMap<real_t>::mpi_type, 0, 460, MyComm);
|
||||
}
|
||||
|
||||
pfes->ApplyDofSigns(h_data); // restore the original h_data
|
||||
|
||||
delete [] values;
|
||||
delete [] nv;
|
||||
delete [] nvdofs;
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "eval_transpose.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// @cond Suppress_Doxygen_warnings
|
||||
|
||||
QuadratureInterpolator::TensorEvalTransposeKernelType
|
||||
QuadratureInterpolator::TensorEvalTransposeKernels::Fallback(
|
||||
int DIM, QVectorLayout Q_LAYOUT, int, int, int)
|
||||
{
|
||||
using namespace internal::quadrature_interpolator;
|
||||
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
if (DIM == 1) { return ValuesTranspose1D<QVectorLayout::byNODES>; }
|
||||
else if (DIM == 2) { return ValuesTranspose2D<QVectorLayout::byNODES>; }
|
||||
else if (DIM == 3) { return ValuesTranspose3D<QVectorLayout::byNODES>; }
|
||||
}
|
||||
else
|
||||
{
|
||||
if (DIM == 1) { return ValuesTranspose1D<QVectorLayout::byVDIM>; }
|
||||
else if (DIM == 2) { return ValuesTranspose2D<QVectorLayout::byVDIM>; }
|
||||
else if (DIM == 3) { return ValuesTranspose3D<QVectorLayout::byVDIM>; }
|
||||
}
|
||||
MFEM_ABORT("Invalid dimension");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// @endcond
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,300 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../quadinterpolator.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
#include "../kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
namespace internal
|
||||
{
|
||||
namespace quadrature_interpolator
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT>
|
||||
static void ValuesTranspose1D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
const auto b = Reshape(b_, q1d, d1d);
|
||||
const auto qd = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, q1d, vdim, NE) :
|
||||
Reshape(q_, vdim, q1d, NE);
|
||||
auto e = Reshape(e_, d1d, vdim, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
for (int c = 0; c < vdim; c++)
|
||||
{
|
||||
for (int d = 0; d < d1d; d++)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int q = 0; q < q1d; q++)
|
||||
{
|
||||
const real_t qval = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
qd(c, q, el) : qd(q, c, el);
|
||||
u += b(q, d) * qval;
|
||||
}
|
||||
e(d, c, el) += u;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1>
|
||||
static void ValuesTranspose2D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
static constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, VDIM, NE) :
|
||||
Reshape(q_, VDIM, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D_batch(NE, D1D, D1D, NBZ, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1*MD1];
|
||||
MFEM_SHARED real_t sm0[NBZ][MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[NBZ][MDQ*MDQ];
|
||||
|
||||
kernels::internal::LoadB<MD1,MQ1>(D1D,Q1D,b,sB);
|
||||
|
||||
ConstDeviceMatrix B(sB, D1D, Q1D);
|
||||
DeviceMatrix QQ(sm0[tidz], MQ1, MQ1);
|
||||
DeviceMatrix DQ(sm1[tidz], MD1, MQ1);
|
||||
DeviceMatrix DD(sm0[tidz], MD1, MD1);
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Load Q data
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
QQ(qx,qy) = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
q(c,qx,qy,el) : q(qx,qy,c,el);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in y: QQ -> DQ (apply B^T in y-direction)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * QQ(qx,qy);
|
||||
}
|
||||
DQ(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in x: DQ -> DD (apply B^T in x-direction)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DQ(dy,qx);
|
||||
}
|
||||
DD(dx,dy) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Store result
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,c,el) += DD(dx,dy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0>
|
||||
static void ValuesTranspose3D(const int NE,
|
||||
const real_t *b_,
|
||||
const real_t *q_,
|
||||
real_t *e_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto q = Q_LAYOUT == QVectorLayout::byNODES ?
|
||||
Reshape(q_, Q1D, Q1D, Q1D, VDIM, NE) :
|
||||
Reshape(q_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
auto e = Reshape(e_, D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_3D(NE, D1D, D1D, D1D, [=] MFEM_HOST_DEVICE (int el)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_INTERP_1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
|
||||
MFEM_SHARED real_t sB[MQ1*MD1];
|
||||
MFEM_SHARED real_t sm0[MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED real_t sm1[MDQ*MDQ*MDQ];
|
||||
|
||||
kernels::internal::LoadB<MD1,MQ1>(D1D,Q1D,b,sB);
|
||||
|
||||
ConstDeviceMatrix B(sB, D1D, Q1D);
|
||||
DeviceCube QQQ(sm0, MQ1, MQ1, MQ1);
|
||||
DeviceCube DQQ(sm1, MD1, MQ1, MQ1);
|
||||
DeviceCube DDQ(sm0, MD1, MD1, MQ1);
|
||||
DeviceCube DDD(sm1, MD1, MD1, MD1);
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
// Load Q data
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
QQQ(qx,qy,qz) = Q_LAYOUT == QVectorLayout::byVDIM ?
|
||||
q(c,qx,qy,qz,el) : q(qx,qy,qz,c,el);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in z
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += B(dz,qz) * QQQ(qx,qy,qz);
|
||||
}
|
||||
DQQ(dz,qx,qy) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in y
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += B(dy,qy) * DQQ(dz,qx,qy);
|
||||
}
|
||||
DDQ(dz,dy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Transpose in x
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
real_t u = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += B(dx,qx) * DDQ(dz,dy,qx);
|
||||
}
|
||||
DDD(dx,dy,dz) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
e(dx,dy,dz,c,el) += DDD(dx,dy,dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace quadrature_interpolator
|
||||
} // namespace internal
|
||||
|
||||
template<int DIM, QVectorLayout Q_LAYOUT,
|
||||
int VDIM, int D1D, int Q1D, int NBZ>
|
||||
QuadratureInterpolator::TensorEvalTransposeKernelType
|
||||
QuadratureInterpolator::TensorEvalTransposeKernels::Kernel()
|
||||
{
|
||||
if (DIM == 1) { return internal::quadrature_interpolator::ValuesTranspose1D<Q_LAYOUT>; }
|
||||
else if (DIM == 2) { return internal::quadrature_interpolator::ValuesTranspose2D<Q_LAYOUT, VDIM, D1D, Q1D, NBZ>; }
|
||||
else if (DIM == 3) { return internal::quadrature_interpolator::ValuesTranspose3D<Q_LAYOUT, VDIM, D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user