Compare commits
893
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fb62afa894 | ||
|
|
b3aeabd735 | ||
|
|
b7fbc8fce5 | ||
|
|
d771e5558a | ||
|
|
99212f8f0f | ||
|
|
70f97c79ed | ||
|
|
70e63b63bf | ||
|
|
92a0a7bc6c | ||
|
|
8640fa27af | ||
|
|
d3d25bb7e8 | ||
|
|
c46ef27c22 | ||
|
|
fdda4aaee8 | ||
|
|
8291390811 | ||
|
|
97800df5c5 | ||
|
|
2764af6227 | ||
|
|
6e315bb836 | ||
|
|
7759b6a0ef | ||
|
|
41f29f7e74 | ||
|
|
cee3e13d69 | ||
|
|
f4b1db61c8 | ||
|
|
89e0dbf127 | ||
|
|
b44c1846e9 | ||
|
|
1ef5c38cae | ||
|
|
2bb58958c9 | ||
|
|
22c5d7fc60 | ||
|
|
585a5645c0 | ||
|
|
f0b022e548 | ||
|
|
c6943f3072 | ||
|
|
4e44aa64a8 | ||
|
|
4ffea76cff | ||
|
|
645243afbe | ||
|
|
3631f93e20 | ||
|
|
37828f3de2 | ||
|
|
dcda408fd6 | ||
|
|
28a01f0395 | ||
|
|
b686bf1bdf | ||
|
|
70cbc94f20 | ||
|
|
6ed5221432 | ||
|
|
7d8ebcd33d | ||
|
|
d53bfa7b1d | ||
|
|
f29b1a3019 | ||
|
|
95017b2924 | ||
|
|
744aa63b80 | ||
|
|
ef1541fc5f | ||
|
|
84c977b3a7 | ||
|
|
cf98a85d67 | ||
|
|
74c87589f1 | ||
|
|
7604bbd03a | ||
|
|
59ab891f79 | ||
|
|
7a4d323b8c | ||
|
|
4b6f491e44 | ||
|
|
88a9b67749 | ||
|
|
4257d0584a | ||
|
|
34cdbd8388 | ||
|
|
7b1ac42629 | ||
|
|
c8fefb9022 | ||
|
|
0358eeb870 | ||
|
|
c281ba5e63 | ||
|
|
c773578cfe | ||
|
|
62ba4dd55a | ||
|
|
10c8c7a028 | ||
|
|
b8ff3df628 | ||
|
|
f14669ae67 | ||
|
|
2dca9fc4e0 | ||
|
|
fc5c88af62 | ||
|
|
7d29e739ec | ||
|
|
3eb87f6d42 | ||
|
|
552bd17416 | ||
|
|
ac78f39d70 | ||
|
|
b9a40daf5b | ||
|
|
78eb1edcaf | ||
|
|
66b57d1591 | ||
|
|
0832fda199 | ||
|
|
24b4e21407 | ||
|
|
05cccec7b7 | ||
|
|
c602e55b9f | ||
|
|
ba51fe2c53 | ||
|
|
f5e82a777f | ||
|
|
e4a649ca56 | ||
|
|
c4de483f85 | ||
|
|
49bfcf627a | ||
|
|
725534735c | ||
|
|
ae26979ee4 | ||
|
|
9276e884e4 | ||
|
|
6f0a8d356f | ||
|
|
898b68ff99 | ||
|
|
f0a057a1d1 | ||
|
|
d3fd4b8022 | ||
|
|
03afd2edf0 | ||
|
|
894779246f | ||
|
|
1f7800109c | ||
|
|
ad04dea26d | ||
|
|
f7f709a529 | ||
|
|
fd32cfa477 | ||
|
|
d4cc1c03ad | ||
|
|
0a37aef4d6 | ||
|
|
bf5fffd7c5 | ||
|
|
07c9965aa4 | ||
|
|
70e5ae0f70 | ||
|
|
09489a318f | ||
|
|
d13ad22587 | ||
|
|
d9d3f0e4ea | ||
|
|
17a2eafbaf | ||
|
|
d1c4057431 | ||
|
|
58c48b961a | ||
|
|
799cc16dc5 | ||
|
|
cecc93bf61 | ||
|
|
39280b7acd | ||
|
|
bffaf743ec | ||
|
|
2759a40066 | ||
|
|
727eafe373 | ||
|
|
aef0d14090 | ||
|
|
bfe605e0ff | ||
|
|
a2991c670b | ||
|
|
c602df9dbc | ||
|
|
5f7c204d3b | ||
|
|
8bdbf38318 | ||
|
|
f7ca315106 | ||
|
|
bba973db73 | ||
|
|
c3394d330e | ||
|
|
8ecd5eb54b | ||
|
|
813bfb5e99 | ||
|
|
402b8bf3d3 | ||
|
|
7372f4902a | ||
|
|
da77e1abc9 | ||
|
|
7352ccf301 | ||
|
|
5909a99b3e | ||
|
|
a2ca2b9575 | ||
|
|
0ba4c9d442 | ||
|
|
57557ec53b | ||
|
|
5da714b61e | ||
|
|
a6df874f9c | ||
|
|
2dc13c14dc | ||
|
|
2c449b4c9d | ||
|
|
0939dac591 | ||
|
|
2c95eff3ad | ||
|
|
7ff5718488 | ||
|
|
55c3321c3b | ||
|
|
0cdf29a4bc | ||
|
|
f7dcf85cbe | ||
|
|
39b9767258 | ||
|
|
2fa920a88b | ||
|
|
5a3923e842 | ||
|
|
2bb8d88e16 | ||
|
|
279c6d07e5 | ||
|
|
67eda2ba05 | ||
|
|
91b7440c7e | ||
|
|
1f3480ada1 | ||
|
|
1066cda295 | ||
|
|
585f9149d1 | ||
|
|
b6a3a119a1 | ||
|
|
a37a3a94ce | ||
|
|
16f0f58fcd | ||
|
|
d51a6a3483 | ||
|
|
471c574dc6 | ||
|
|
018fb9ac05 | ||
|
|
e1b8ab362c | ||
|
|
00d2f19dec | ||
|
|
a59817b8a7 | ||
|
|
4e8a531bb1 | ||
|
|
40c0412c63 | ||
|
|
5457d033d4 | ||
|
|
1deb071ada | ||
|
|
40d0b61939 | ||
|
|
0c6f96c2eb | ||
|
|
6e552c3f90 | ||
|
|
64ac989d07 | ||
|
|
11964610e1 | ||
|
|
bfff83d9de | ||
|
|
0efa0dcb21 | ||
|
|
944cd2f09f | ||
|
|
01bf5db292 | ||
|
|
0b76f8d984 | ||
|
|
b52671541e | ||
|
|
e09103966e | ||
|
|
9937009eab | ||
|
|
9bd06e360e | ||
|
|
2bea6d11f1 | ||
|
|
264886c511 | ||
|
|
697cb9bb95 | ||
|
|
7ec3c5a30c | ||
|
|
4fbd599058 | ||
|
|
c1071bf82e | ||
|
|
26c681313e | ||
|
|
9379a8971a | ||
|
|
175317c04d | ||
|
|
56a9107e7c | ||
|
|
fe0211557b | ||
|
|
b84e8a5793 | ||
|
|
041e3b99fd | ||
|
|
b9cf225e13 | ||
|
|
684ae6f21f | ||
|
|
2543863092 | ||
|
|
daca8cfd6c | ||
|
|
4d3db426c6 | ||
|
|
24fd1e1fc0 | ||
|
|
d5697799c8 | ||
|
|
e57c5e0fec | ||
|
|
aecfbff2e8 | ||
|
|
5f07e2155a | ||
|
|
5a8cebeea7 | ||
|
|
802f213873 | ||
|
|
d036fd8b1f | ||
|
|
79d7c5c682 | ||
|
|
36ac1adb3b | ||
|
|
4cd4e21bc6 | ||
|
|
6489ecb59e | ||
|
|
7c9fe7b560 | ||
|
|
d97454370c | ||
|
|
4e8aaf7f11 | ||
|
|
cb79110ef9 | ||
|
|
ccce5c8217 | ||
|
|
ad9adab6dc | ||
|
|
9c91e44feb | ||
|
|
0acbd90ac4 | ||
|
|
b2cdfbf8bc | ||
|
|
fb9c3fa30a | ||
|
|
8b1ecbc3af | ||
|
|
66cbd5450e | ||
|
|
13be84efdb | ||
|
|
f867b645ab | ||
|
|
27077f2f0a | ||
|
|
095f9a2aee | ||
|
|
a965c079fb | ||
|
|
31d2ff15e2 | ||
|
|
492a28b227 | ||
|
|
15b50a277e | ||
|
|
2a85ec5f97 | ||
|
|
268c3cb461 | ||
|
|
401249f1ac | ||
|
|
6bea0b205d | ||
|
|
e9be0b2074 | ||
|
|
c78cf79474 | ||
|
|
72d7f7ccb7 | ||
|
|
44e8364877 | ||
|
|
b4b6c7e106 | ||
|
|
e6e3c27cc9 | ||
|
|
d64ce893ad | ||
|
|
d66695fb30 | ||
|
|
c4eb458e03 | ||
|
|
f80a520f31 | ||
|
|
7af6ed945d | ||
|
|
c15c3d8f78 | ||
|
|
fab68704d2 | ||
|
|
acd9de7cee | ||
|
|
6a2b798acd | ||
|
|
8caaabda21 | ||
|
|
aa1233f47a | ||
|
|
a652d89a2d | ||
|
|
0ac144c992 | ||
|
|
67ae86557a | ||
|
|
c6e1bdf28d | ||
|
|
3ddc31337e | ||
|
|
7a2084a438 | ||
|
|
48c8173ebd | ||
|
|
930c058704 | ||
|
|
4f0826fb6f | ||
|
|
af9496be2e | ||
|
|
0f817acd30 | ||
|
|
efef5b1eff | ||
|
|
7fdc80e073 | ||
|
|
2622e50d40 | ||
|
|
5c1f9b64af | ||
|
|
f9dc31959d | ||
|
|
56507bb395 | ||
|
|
0903dd01e9 | ||
|
|
2eab58b853 | ||
|
|
a6a2351e06 | ||
|
|
de80deeb0b | ||
|
|
2f28691de3 | ||
|
|
9238db6d2b | ||
|
|
c1663093c0 | ||
|
|
0e87cac48e | ||
|
|
8d5364482b | ||
|
|
0effa7abff | ||
|
|
127fa2645c | ||
|
|
f4959fc875 | ||
|
|
dd3c075b1b | ||
|
|
4a9eed99b4 | ||
|
|
27ef7812f8 | ||
|
|
f10c1824c1 | ||
|
|
e29ee5f5fc | ||
|
|
891364e5ff | ||
|
|
a9f3f42289 | ||
|
|
f3771a22a5 | ||
|
|
31e4a9d8ee | ||
|
|
75ef30918c | ||
|
|
6871b1c6dc | ||
|
|
c42906a395 | ||
|
|
92ef9d0629 | ||
|
|
1e15e6b57f | ||
|
|
d1beabdd0d | ||
|
|
e074506faf | ||
|
|
670f4799d7 | ||
|
|
04eddb884c | ||
|
|
fc1d4fffaf | ||
|
|
d79f834750 | ||
|
|
b765ebad81 | ||
|
|
61be39191a | ||
|
|
e87b54a057 | ||
|
|
57c3853fb6 | ||
|
|
08f5abc6b2 | ||
|
|
cfe7834b95 | ||
|
|
1a8d88440e | ||
|
|
99c4becfae | ||
|
|
2daac8eef4 | ||
|
|
40e633b36f | ||
|
|
94a5e625be | ||
|
|
840ff99288 | ||
|
|
fbe242a2e2 | ||
|
|
ae880b4ee8 | ||
|
|
33b413042a | ||
|
|
9cff5875c7 | ||
|
|
80c787a79f | ||
|
|
9e76838fe4 | ||
|
|
75f5a89d2c | ||
|
|
c4d8bd4744 | ||
|
|
0fb7f04d6c | ||
|
|
01476b98cb | ||
|
|
d2ae506e8e | ||
|
|
65dfcd5e0a | ||
|
|
c3d869cd6c | ||
|
|
6b256c7cbb | ||
|
|
b73225de21 | ||
|
|
c3dd82b5ba | ||
|
|
735d0d41b8 | ||
|
|
291875509d | ||
|
|
d81b3fa05a | ||
|
|
07f7b0a943 | ||
|
|
b4b72a95ee | ||
|
|
57981bc329 | ||
|
|
43ceae8f46 | ||
|
|
3642dc3a57 | ||
|
|
0bf9d144f9 | ||
|
|
4de4256085 | ||
|
|
f589bdeb27 | ||
|
|
b3cdfdea49 | ||
|
|
f329c3b760 | ||
|
|
1aa4f62937 | ||
|
|
ceb49b322c | ||
|
|
6e145a4ccf | ||
|
|
900a020610 | ||
|
|
ab4a76022b | ||
|
|
a6caeff9d6 | ||
|
|
649209dc2b | ||
|
|
3e56ffab7c | ||
|
|
79e9a3d320 | ||
|
|
ce2b02624d | ||
|
|
80681fa56a | ||
|
|
8e8868e4a0 | ||
|
|
b671a7a679 | ||
|
|
6d7b38c02f | ||
|
|
3e6889145e | ||
|
|
60378b79af | ||
|
|
5a30e94472 | ||
|
|
cdfe1db094 | ||
|
|
99a98596ae | ||
|
|
7947109731 | ||
|
|
ab13556e8c | ||
|
|
df7aec2506 | ||
|
|
f07cb7379d | ||
|
|
cbb167f231 | ||
|
|
49e6225b7d | ||
|
|
5f0630a550 | ||
|
|
525f6d2a44 | ||
|
|
6e113683af | ||
|
|
99e69b93e5 | ||
|
|
53e952f1fd | ||
|
|
1b265e22e0 | ||
|
|
3dd0f5c328 | ||
|
|
7160b68bce | ||
|
|
827fbfb14a | ||
|
|
e678e66acd | ||
|
|
f2d7b0c75a | ||
|
|
7c3912b2e3 | ||
|
|
06f49ed5c2 | ||
|
|
39a6c88595 | ||
|
|
9145d4b1de | ||
|
|
9fb9c4937a | ||
|
|
9d3e3dd017 | ||
|
|
e8458444c6 | ||
|
|
2d065de342 | ||
|
|
7c3a368562 | ||
|
|
e2ff03e4ba | ||
|
|
a510328015 | ||
|
|
97440f9500 | ||
|
|
781fddddc6 | ||
|
|
7fcc651020 | ||
|
|
4c719ad706 | ||
|
|
207716e0f3 | ||
|
|
7ab523f5a2 | ||
|
|
2de28abb19 | ||
|
|
a26f7e6b51 | ||
|
|
d728e4f9a4 | ||
|
|
79dd7c14b2 | ||
|
|
c1320238ae | ||
|
|
4b10b7c44b | ||
|
|
49d328403d | ||
|
|
aa10bc7966 | ||
|
|
91eacf5af7 | ||
|
|
6543ffb790 | ||
|
|
e4290e6d33 | ||
|
|
de34bf094c | ||
|
|
a52a59b524 | ||
|
|
30b2b43814 | ||
|
|
1f4024879c | ||
|
|
d949f69a4b | ||
|
|
6cdad9b4ee | ||
|
|
d727b1a14b | ||
|
|
a6c6fb18cf | ||
|
|
069e57b3a4 | ||
|
|
df59effa09 | ||
|
|
09c9c94916 | ||
|
|
30ad68af20 | ||
|
|
237905c956 | ||
|
|
b4d870c3be | ||
|
|
213a290511 | ||
|
|
2fa65d4846 | ||
|
|
661a7f6f38 | ||
|
|
14e663d663 | ||
|
|
c7a6ab3c89 | ||
|
|
b8270effa2 | ||
|
|
e068e21672 | ||
|
|
9f7ddad73c | ||
|
|
a40a1f899d | ||
|
|
99bc161b86 | ||
|
|
85b642bd77 | ||
|
|
cd2236d018 | ||
|
|
b732ae829f | ||
|
|
cc66b2ef89 | ||
|
|
b4b5373697 | ||
|
|
ceb34240df | ||
|
|
03533d095c | ||
|
|
dab908a49c | ||
|
|
223c47fae2 | ||
|
|
f6ed30852b | ||
|
|
dc5c975a88 | ||
|
|
bd695bc74c | ||
|
|
e165101b27 | ||
|
|
096e5ffb93 | ||
|
|
fb8e595da3 | ||
|
|
868d8aa057 | ||
|
|
dd09413e47 | ||
|
|
c6a5a75d35 | ||
|
|
6a881ae708 | ||
|
|
279938afe5 | ||
|
|
75db215d3f | ||
|
|
a00f5a7aa3 | ||
|
|
85f450e7d0 | ||
|
|
f40ec9f0d2 | ||
|
|
a3403c3f3a | ||
|
|
1d35d74e85 | ||
|
|
dceaf60897 | ||
|
|
f761e4d033 | ||
|
|
9745e7668e | ||
|
|
5b97514b1f | ||
|
|
be02b9a457 | ||
|
|
7739c00cf9 | ||
|
|
3019a29b9f | ||
|
|
176a029bc7 | ||
|
|
66c6b9aa0d | ||
|
|
c27db29466 | ||
|
|
b027c1c6cc | ||
|
|
29136050db | ||
|
|
89aade4b2d | ||
|
|
0770a21d2a | ||
|
|
92cb4a02a7 | ||
|
|
335d810155 | ||
|
|
18bb5a5ac0 | ||
|
|
f29e1b82f7 | ||
|
|
0b5ee4ea04 | ||
|
|
6201d7c5bb | ||
|
|
a5e0c3f856 | ||
|
|
3913c81855 | ||
|
|
e170d20edc | ||
|
|
beedb1e931 | ||
|
|
684785eb64 | ||
|
|
bddf1110b4 | ||
|
|
83ec745644 | ||
|
|
12590207fa | ||
|
|
cc64d1fd9c | ||
|
|
6fcdfd7d41 | ||
|
|
a57fd02a4c | ||
|
|
3b54ae451e | ||
|
|
ab4c17ab7d | ||
|
|
4b26c2c11d | ||
|
|
18007107d8 | ||
|
|
85349d3a95 | ||
|
|
01d7657e7b | ||
|
|
75abc3e063 | ||
|
|
77a038ab44 | ||
|
|
2e610493c6 | ||
|
|
afb8a5faa2 | ||
|
|
79aa383c63 | ||
|
|
2ae97ff2da | ||
|
|
2b712207c6 | ||
|
|
71bc6b548a | ||
|
|
64bbbc3d8c | ||
|
|
7e88d111f9 | ||
|
|
7929766814 | ||
|
|
bc795fc99a | ||
|
|
5d80f8e195 | ||
|
|
a00bcade5f | ||
|
|
8491ec4183 | ||
|
|
dbba71bc7c | ||
|
|
aea81c2920 | ||
|
|
4e7821c809 | ||
|
|
2930c1477f | ||
|
|
3c498bd4a9 | ||
|
|
2136283065 | ||
|
|
24d58dd47f | ||
|
|
2d10510a15 | ||
|
|
26a525844b | ||
|
|
c85d34c34a | ||
|
|
eed4c06fb7 | ||
|
|
36da88f98f | ||
|
|
ca276a9a1b | ||
|
|
fd5fe341b1 | ||
|
|
a80ec31887 | ||
|
|
a8925a8d58 | ||
|
|
9257c0d177 | ||
|
|
6b763d59db | ||
|
|
76962605ed | ||
|
|
1ab1091e19 | ||
|
|
49daadcd84 | ||
|
|
9cbf128f18 | ||
|
|
436bb9b070 | ||
|
|
ca50d0db07 | ||
|
|
b95407f6ea | ||
|
|
02f6ca0ef2 | ||
|
|
88b98c8fb4 | ||
|
|
70e4b4a781 | ||
|
|
fed561d5ba | ||
|
|
a99b9869c0 | ||
|
|
4f3f81fef6 | ||
|
|
c88387694b | ||
|
|
1c0c3054fa | ||
|
|
6c1ee0c854 | ||
|
|
d0f04b0168 | ||
|
|
527564ab94 | ||
|
|
fc57c1be85 | ||
|
|
2a798d7e1b | ||
|
|
0ff0f704b9 | ||
|
|
723362b610 | ||
|
|
bf7843fc32 | ||
|
|
e7674ba0e7 | ||
|
|
21e6e7f669 | ||
|
|
c9c2cd825e | ||
|
|
73a7ebf5f3 | ||
|
|
b4655e6409 | ||
|
|
ce0788c104 | ||
|
|
955446c9f7 | ||
|
|
b3bb0ea404 | ||
|
|
6c52bec12d | ||
|
|
01c8643d1d | ||
|
|
3b35126fbc | ||
|
|
3ec12f0866 | ||
|
|
2d4b956b48 | ||
|
|
f966445355 | ||
|
|
c0ebddedbe | ||
|
|
ee6d162d47 | ||
|
|
b529164f3a | ||
|
|
1c8a8dbcb1 | ||
|
|
55b26db935 | ||
|
|
5ebc924b20 | ||
|
|
af3f6a8b7d | ||
|
|
134436f064 | ||
|
|
a690135d15 | ||
|
|
abd79fc6fa | ||
|
|
e975ad2950 | ||
|
|
510498a3e6 | ||
|
|
55dccdf598 | ||
|
|
58997c32bd | ||
|
|
54580450d3 | ||
|
|
c0395c6371 | ||
|
|
2dc947ff91 | ||
|
|
792c50518e | ||
|
|
400c31435c | ||
|
|
68fd30d2f9 | ||
|
|
e4a6e1c724 | ||
|
|
802e20160c | ||
|
|
0055894917 | ||
|
|
c46e60321a | ||
|
|
072c4ad387 | ||
|
|
f1ac4b1eff | ||
|
|
c2de4ee220 | ||
|
|
c53fe3ffda | ||
|
|
105a991779 | ||
|
|
e831e8007b | ||
|
|
01d53b32fa | ||
|
|
dee42c559a | ||
|
|
33c28f7591 | ||
|
|
03c0009414 | ||
|
|
6f4c373d58 | ||
|
|
5a27cb9e54 | ||
|
|
48a018e7a3 | ||
|
|
39fc24685f | ||
|
|
ba68b03aeb | ||
|
|
fb10506117 | ||
|
|
1d667ea546 | ||
|
|
db626777be | ||
|
|
8660563901 | ||
|
|
29f6d09c4e | ||
|
|
b3f4e5bb0b | ||
|
|
04f1d562b7 | ||
|
|
c610cabe53 | ||
|
|
14ac6afcd8 | ||
|
|
eea03f48bb | ||
|
|
bd10bef5d1 | ||
|
|
cb3202f9f9 | ||
|
|
3fb736981e | ||
|
|
802a4afce0 | ||
|
|
991b452a42 | ||
|
|
2a642ba5c7 | ||
|
|
7621999979 | ||
|
|
004449150a | ||
|
|
3e719dfaf6 | ||
|
|
10bed2997c | ||
|
|
fa9fce971f | ||
|
|
ec8796ebde | ||
|
|
e985684812 | ||
|
|
fac83d44b6 | ||
|
|
2b9bfdcfa1 | ||
|
|
7ccf354afb | ||
|
|
e57f0a0905 | ||
|
|
2e391ee3ef | ||
|
|
a4330ec239 | ||
|
|
b5335e6e62 | ||
|
|
ee35dcd356 | ||
|
|
8efba0c8ec | ||
|
|
55d27cbd91 | ||
|
|
46a8e8d4b8 | ||
|
|
51faf60eab | ||
|
|
23cec0568b | ||
|
|
cbbf157690 | ||
|
|
1cfd42cc7d | ||
|
|
8ff06039bd | ||
|
|
38b3168eae | ||
|
|
43fc2ecc82 | ||
|
|
0b389f2da6 | ||
|
|
749ba058fd | ||
|
|
d83112c174 | ||
|
|
cab24d6f3d | ||
|
|
572470939a | ||
|
|
4cbf65ceba | ||
|
|
cf7d5e8e47 | ||
|
|
f7455fdfb3 | ||
|
|
98929f5465 | ||
|
|
4be20db30b | ||
|
|
ff57240475 | ||
|
|
650941acc9 | ||
|
|
da93ab2ed4 | ||
|
|
4540775fdd | ||
|
|
a909cf4986 | ||
|
|
e18858ab09 | ||
|
|
5878be5cb8 | ||
|
|
17d4cba0c9 | ||
|
|
e30182268d | ||
|
|
a5df0b44d7 | ||
|
|
8d640a6a29 | ||
|
|
e6163eb49c | ||
|
|
4b779f9b59 | ||
|
|
9b83346ed3 | ||
|
|
26d3646c1b | ||
|
|
f7724b30d9 | ||
|
|
c28cfb92ac | ||
|
|
b5e5bedb58 | ||
|
|
2a5a1fc73b | ||
|
|
4b97cb3a36 | ||
|
|
25412a3400 | ||
|
|
c61af0cce9 | ||
|
|
e1bd6275d1 | ||
|
|
edee386ec8 | ||
|
|
325cc85ad4 | ||
|
|
8e3217e8a4 | ||
|
|
f5aa751bba | ||
|
|
9457d1246f | ||
|
|
5761a2d249 | ||
|
|
4cb9531d6a | ||
|
|
79c16ef58e | ||
|
|
95454a1733 | ||
|
|
4c2d7c9275 | ||
|
|
de15b99c40 | ||
|
|
81b1848021 | ||
|
|
00d84e0920 | ||
|
|
caedc3be67 | ||
|
|
212cd6d5fc | ||
|
|
2cc1f18bc9 | ||
|
|
32482904a9 | ||
|
|
be18a4992c | ||
|
|
95370a906d | ||
|
|
aec2c7b832 | ||
|
|
e52d6d7530 | ||
|
|
a1720fd6e3 | ||
|
|
19f484ec72 | ||
|
|
e29ae919ad | ||
|
|
895caad872 | ||
|
|
d87f4e347c | ||
|
|
a502f38360 | ||
|
|
fa5ccb3e9f | ||
|
|
fc24fad31c | ||
|
|
d53566acbc | ||
|
|
9c41cd5f34 | ||
|
|
b28277027e | ||
|
|
3c191f6815 | ||
|
|
96a5ef5392 | ||
|
|
5fe2e5e573 | ||
|
|
40056535c1 | ||
|
|
bc7203e720 | ||
|
|
bac87d5bdf | ||
|
|
1441fca816 | ||
|
|
272d529aee | ||
|
|
9dc26d68f8 | ||
|
|
5e3d92693f | ||
|
|
6dec38360f | ||
|
|
b0abbc3a24 | ||
|
|
22afa40eb0 | ||
|
|
6e8001e206 | ||
|
|
59a4ab5ada | ||
|
|
861829db28 | ||
|
|
2453f6f20c | ||
|
|
27b6172257 | ||
|
|
00a431e5b3 | ||
|
|
cc9dd4e34e | ||
|
|
ca8e3e73a9 | ||
|
|
c4a1a23756 | ||
|
|
6571977c97 | ||
|
|
e0b19c5520 | ||
|
|
1ea27f2805 | ||
|
|
a9775ee6ab | ||
|
|
057ab87d30 | ||
|
|
957aa9aeef | ||
|
|
ae5da8e9ac | ||
|
|
c45ed09112 | ||
|
|
42522ddd43 | ||
|
|
1ecf80a2f7 | ||
|
|
93c3684eb1 | ||
|
|
84cc5c7f4b | ||
|
|
8a0724498c | ||
|
|
f9976955cf | ||
|
|
8ecb802662 | ||
|
|
568dff92bb | ||
|
|
bbc186136d | ||
|
|
c2253a9532 | ||
|
|
3309b8d49b | ||
|
|
c0f65a8c01 | ||
|
|
d2fcc358ad | ||
|
|
c049453773 | ||
|
|
e4b95d2666 | ||
|
|
02d8d0a0a5 | ||
|
|
48e6cd8b01 | ||
|
|
9fb292898e | ||
|
|
f00ea139a9 | ||
|
|
d08f0790ea | ||
|
|
15c1ff069c | ||
|
|
061b70f461 | ||
|
|
6671a77e34 | ||
|
|
1cbbde8fb6 | ||
|
|
f958b9660b | ||
|
|
a250b07a34 | ||
|
|
b3a06ecbc0 | ||
|
|
12a8465047 | ||
|
|
15f6269ad4 | ||
|
|
e6ed2fafa0 | ||
|
|
9c37a19c7f | ||
|
|
27a3f4bfce | ||
|
|
5def286b2a | ||
|
|
b5aa280711 | ||
|
|
dbba00715c | ||
|
|
ec8bcb8c16 | ||
|
|
8d5249c9ba | ||
|
|
6cc9989653 | ||
|
|
fe59dc5f29 | ||
|
|
fd7b8c84e0 | ||
|
|
5bea192dd7 | ||
|
|
596306b12d | ||
|
|
c5dea1ee17 | ||
|
|
9c09fea06b | ||
|
|
bc28f6f06e | ||
|
|
3b1a43de1a | ||
|
|
76b044ae99 | ||
|
|
27b4be77a9 | ||
|
|
af6cf29b85 | ||
|
|
039a04b3e2 | ||
|
|
ceba505e4e | ||
|
|
ed556b5c63 | ||
|
|
6f34ccec75 | ||
|
|
feb79f3d56 | ||
|
|
35704d508d | ||
|
|
721ea4323b | ||
|
|
8e36f5ebdc | ||
|
|
28d2d44de1 | ||
|
|
df650aab6b | ||
|
|
8b9b0f7a0d | ||
|
|
fd0ac87506 | ||
|
|
dbb15ab25d | ||
|
|
83d753c036 | ||
|
|
fb249c5775 | ||
|
|
4415622c99 | ||
|
|
479a70c65f | ||
|
|
a3e73ee1a3 | ||
|
|
6ab9b9f1e6 | ||
|
|
9bdded3450 | ||
|
|
83c48a33a3 | ||
|
|
d7661bd5ef | ||
|
|
250b62f1e6 | ||
|
|
e70cc67bc1 | ||
|
|
e2c7458cc6 | ||
|
|
75aee31101 | ||
|
|
5b4cf4265b | ||
|
|
a4ffe6caa3 | ||
|
|
a315c900f9 | ||
|
|
258f83b2ca | ||
|
|
14051159d0 | ||
|
|
b3ee631aa6 | ||
|
|
daf2fdecec | ||
|
|
d3a0d0a181 | ||
|
|
21d77c738d | ||
|
|
f31d8747f9 | ||
|
|
cf45b90266 | ||
|
|
10ec2a1818 | ||
|
|
49469131c2 | ||
|
|
2dbb377f91 | ||
|
|
539176a2e1 | ||
|
|
dd7db993bc | ||
|
|
b31d805597 | ||
|
|
e987383708 | ||
|
|
07853b9c62 | ||
|
|
eb677b8581 | ||
|
|
0db220c79d | ||
|
|
73d1d2a1f2 | ||
|
|
185fc97786 | ||
|
|
146205eba6 | ||
|
|
d76811ac9a | ||
|
|
343f9e02a6 | ||
|
|
56d6efd7c6 | ||
|
|
211738b616 | ||
|
|
9e16d2c109 | ||
|
|
62f0f65d05 | ||
|
|
1ec02f0b23 | ||
|
|
c25845e724 | ||
|
|
b0a7fd45e6 | ||
|
|
213b6e51d3 | ||
|
|
e1f47f0079 | ||
|
|
91ed44e45a | ||
|
|
9a0f991da6 | ||
|
|
7cc35b68c2 | ||
|
|
525fb59dd2 | ||
|
|
ab74459790 | ||
|
|
ec745ebb25 | ||
|
|
49a31c0cf7 | ||
|
|
ba9b251007 | ||
|
|
b686dbb897 | ||
|
|
5ce2fa9ab9 | ||
|
|
78c93de6ce | ||
|
|
3d4aa157cb | ||
|
|
bfc4484715 | ||
|
|
dfbb139273 | ||
|
|
e18ae9cbfc | ||
|
|
8782faff18 | ||
|
|
bcde578a5d | ||
|
|
12609ea7dd | ||
|
|
abc671eb04 | ||
|
|
288cef9ccf | ||
|
|
c6f122e392 | ||
|
|
73de0369a1 | ||
|
|
2d7c1c6756 | ||
|
|
64b51e3e2b | ||
|
|
b5e1a0732c | ||
|
|
27879297e4 | ||
|
|
b3babbef60 | ||
|
|
7909fab83f | ||
|
|
9d0015949a | ||
|
|
bfabd546fd | ||
|
|
c2796cd1c7 | ||
|
|
72ffdecfa6 | ||
|
|
d765407dbe | ||
|
|
f6c8898506 | ||
|
|
8f0817c621 | ||
|
|
2ffb058e61 | ||
|
|
a7f1c177c5 | ||
|
|
023e63dfb1 | ||
|
|
e63ac279c1 | ||
|
|
8b7f15cdca | ||
|
|
a263658bdb | ||
|
|
3f4a6ce0d2 | ||
|
|
4ee2b18d97 | ||
|
|
2d4e3cf77e | ||
|
|
5f3f056703 | ||
|
|
c933973249 | ||
|
|
7ee2335810 | ||
|
|
3c064fb4af | ||
|
|
6629cb4adb |
+10
-8
@@ -15,8 +15,10 @@ install:
|
||||
- msmpisdk.msi /passive
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install METIS
|
||||
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
|
||||
# Install METIS, use a mirror because the original source server is not always
|
||||
# up. Original url:
|
||||
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz
|
||||
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
|
||||
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd metis-5.1.0
|
||||
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
|
||||
@@ -26,17 +28,17 @@ install:
|
||||
- cd ..
|
||||
|
||||
# Install hypre
|
||||
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/V2-10-0b.tar.gz'
|
||||
- 7z x V2-10-0b.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2-10-0b
|
||||
- cmake -H. -Bbuild -DHYPRE_USING_FEI=OFF -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/v2.19.0.tar.gz'
|
||||
- 7z x v2.19.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.19.0/src
|
||||
- cmake -H. -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- cmake --build build
|
||||
- cmake --build build --target install
|
||||
- cd ..
|
||||
- cd ../..
|
||||
|
||||
# MFEM
|
||||
before_build:
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2-10-0b\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2-10-0b\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_DIR=%cd%\hypre-2.19.0\src\hypre -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE
|
||||
|
||||
build_script:
|
||||
|
||||
+12
-14
@@ -163,20 +163,6 @@ miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_*
|
||||
|
||||
miniapps/hypsys/build
|
||||
miniapps/hypsys/errors.txt
|
||||
miniapps/hypsys/grid*
|
||||
miniapps/hypsys/hypsys
|
||||
miniapps/hypsys/initial*
|
||||
miniapps/hypsys/output
|
||||
miniapps/hypsys/phypsys
|
||||
miniapps/hypsys/pressure*
|
||||
miniapps/hypsys/results
|
||||
miniapps/hypsys/scripts/gridfunc-scatter
|
||||
miniapps/hypsys/ultimate*
|
||||
miniapps/hypsys/velocity*
|
||||
miniapps/hypsys/various
|
||||
|
||||
miniapps/meshing/mobius-strip
|
||||
miniapps/meshing/klein-bottle
|
||||
miniapps/meshing/toroid
|
||||
@@ -258,12 +244,24 @@ miniapps/navier/navier_3dfoc
|
||||
miniapps/navier/tgv_out*.txt
|
||||
miniapps/navier/*_output
|
||||
|
||||
miniapps/adjoint/cvsRoberts_ASAi_dns
|
||||
miniapps/adjoint/adjoint_advection_diffusion
|
||||
|
||||
# Unit test binary and outputs
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
tests/unit/punit_tests
|
||||
tests/unit/sedov_tests_*
|
||||
tests/unit/psedov_tests_*
|
||||
tests/unit/tmop_tests_*
|
||||
tests/unit/ptmop_tests_*
|
||||
tests/unit/cube.mesh
|
||||
tests/unit/star.mesh
|
||||
tests/unit/blade.mesh
|
||||
tests/unit/square01.mesh
|
||||
tests/unit/toroid-hex.mesh
|
||||
tests/unit/beam-hex-nurbs.mesh
|
||||
tests/unit/square-disc-nurbs.mesh
|
||||
|
||||
# Test script output
|
||||
tests/scripts/*.err
|
||||
|
||||
+98
-33
@@ -11,11 +11,20 @@
|
||||
|
||||
language: cpp
|
||||
|
||||
os: linux
|
||||
dist: bionic
|
||||
|
||||
stages:
|
||||
- checks
|
||||
- tests
|
||||
- optional
|
||||
|
||||
env:
|
||||
global:
|
||||
- HYPRE_ARCHIVE=v2.19.0.tar.gz
|
||||
HYPRE_URL=https://github.com/hypre-space/hypre/archive/$HYPRE_ARCHIVE
|
||||
HYPRE_TOP_DIR=hypre-2.19.0
|
||||
|
||||
jobs:
|
||||
include:
|
||||
|
||||
@@ -28,6 +37,7 @@ jobs:
|
||||
|
||||
- stage: checks
|
||||
os: linux
|
||||
dist: xenial
|
||||
name: "code-style"
|
||||
addons:
|
||||
apt:
|
||||
@@ -46,9 +56,6 @@ jobs:
|
||||
packages:
|
||||
- doxygen
|
||||
- graphviz
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
env: MPI=YES
|
||||
script:
|
||||
- cd ${TRAVIS_BUILD_DIR}
|
||||
- cd tests/scripts
|
||||
@@ -63,13 +70,24 @@ jobs:
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
env: MPI=YES
|
||||
script:
|
||||
before_script:
|
||||
- cd ${TRAVIS_BUILD_DIR}
|
||||
- mpicxx -v
|
||||
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
|
||||
- make all -j3
|
||||
- make test-noclean
|
||||
script:
|
||||
- cd tests/scripts
|
||||
- ./runtest gitignore
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a Lib ..; rm -rf * ; mv ../libmetis.a ../Lib .;
|
||||
rm -f Lib/*.{c,o}
|
||||
|
||||
# ========================
|
||||
# Optional Checks/Tests
|
||||
@@ -78,6 +96,7 @@ jobs:
|
||||
|
||||
- stage: optional
|
||||
name: "branch-history"
|
||||
if: branch != next
|
||||
# need full git history for the binary/big files check
|
||||
git:
|
||||
depth: false
|
||||
@@ -106,6 +125,8 @@ jobs:
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
@@ -114,6 +135,8 @@ jobs:
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
@@ -137,9 +160,9 @@ jobs:
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=2
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -168,9 +191,9 @@ jobs:
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=2
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -193,16 +216,16 @@ jobs:
|
||||
- cd ${TRAVIS_BUILD_DIR}/build
|
||||
- cmake ..
|
||||
-DMFEM_USE_MPI=ON
|
||||
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../hypre-2.10.0b/src/hypre
|
||||
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../$HYPRE_TOP_DIR/src/hypre
|
||||
-DMFEM_MPI_NP=$NPROCS
|
||||
- make -j3 mfem examples
|
||||
- cd ${TRAVIS_BUILD_DIR}/build/tests/unit
|
||||
- make -j3
|
||||
- ctest --output-on-failure
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -218,27 +241,43 @@ jobs:
|
||||
# - parallel
|
||||
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
osx_image: xcode11.2
|
||||
compiler: clang
|
||||
name: "Mac: Serial + Debug"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
osx_image: xcode11.2
|
||||
compiler: clang
|
||||
name: "Mac: Serial"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
osx_image: xcode11.2
|
||||
compiler: clang
|
||||
name: "Mac: Parallel + Debug"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
@@ -246,9 +285,9 @@ jobs:
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
@@ -257,9 +296,13 @@ jobs:
|
||||
rm -f Lib/*.{c,o}
|
||||
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
osx_image: xcode11.2
|
||||
compiler: clang
|
||||
name: "Mac: Parallel"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
@@ -267,9 +310,9 @@ jobs:
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
@@ -283,14 +326,19 @@ before_install:
|
||||
# brew install open-mpi;
|
||||
# fi
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.1:
|
||||
# Disable ccache while building dependencies that are cached:
|
||||
- echo "before \$PATH = $PATH";
|
||||
export PATH=${PATH//\/usr\/lib\/ccache:/};
|
||||
echo "after \$PATH = $PATH"
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.6:
|
||||
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
|
||||
mkdir -p $HOME/builds && cd $HOME/builds &&
|
||||
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.1.tar.bz2 &&
|
||||
wget https://download.open-mpi.org/release/open-mpi/v2.1/openmpi-2.1.6.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.6.tar.bz2 &&
|
||||
mkdir openmpi-build && cd openmpi-build &&
|
||||
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
|
||||
../openmpi-2.1.6/configure --prefix=$HOME/local-cached &&
|
||||
make -j3 all && make install;
|
||||
fi;
|
||||
PATH=$HOME/local-cached/bin:$PATH;
|
||||
@@ -335,26 +383,28 @@ install:
|
||||
|
||||
# hypre
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
rm -rf hypre-2.10.0b;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
if [ ! -e $HYPRE_TOP_DIR/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget $HYPRE_URL;
|
||||
rm -rf $HYPRE_TOP_DIR;
|
||||
tar xvzf $HYPRE_ARCHIVE;
|
||||
cd $HYPRE_TOP_DIR/src;
|
||||
./configure --disable-fortran CC=mpicc CXX=mpic++;
|
||||
make -j3;
|
||||
cd ../..;
|
||||
else
|
||||
echo "Reusing cached hypre-2.10.0b/";
|
||||
echo "Reusing cached $HYPRE_TOP_DIR/";
|
||||
fi;
|
||||
ln -s hypre-2.10.0b hypre;
|
||||
ln -s $HYPRE_TOP_DIR hypre;
|
||||
else
|
||||
echo "Serial build, not using hypre";
|
||||
fi
|
||||
|
||||
# METIS
|
||||
# METIS, use a mirror because the original source server is not always up.
|
||||
# Original url:
|
||||
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e metis-4.0/libmetis.a ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
wget https://mfem.github.io/tpls/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
|
||||
rm -rf metis-4.0;
|
||||
@@ -364,6 +414,18 @@ install:
|
||||
fi;
|
||||
fi
|
||||
|
||||
# Re-enable ccache on linux; enable ccache on mac os:
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
export PATH="/usr/lib/ccache:$PATH";
|
||||
else
|
||||
if [ $TRAVIS_OS_NAME == "osx" ]; then
|
||||
export PATH="/usr/local/opt/ccache/libexec:$PATH";
|
||||
fi;
|
||||
fi
|
||||
|
||||
- printf "which \$CC = "; which $CC;
|
||||
printf "which \$CXX = "; which $CXX
|
||||
|
||||
script:
|
||||
# Compiler
|
||||
- if [ $MPI == "YES" ]; then
|
||||
@@ -384,6 +446,9 @@ script:
|
||||
if [ "$CODECOV" == "YES" ]; then
|
||||
CPPFLAGS="--coverage -g";
|
||||
fi;
|
||||
if [ "$TRAVIS_OS_NAME" != "linux" ] || [ "$DEBUG" == "YES" ]; then
|
||||
CPPFLAGS+=" -pedantic -Wall -Werror";
|
||||
fi
|
||||
|
||||
# Configure the library
|
||||
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG $MAKE_CXX_FLAG
|
||||
|
||||
@@ -33,6 +33,9 @@ Meshing improvements
|
||||
for example the periodic-annulus-sector and periodic-torus-sector files in
|
||||
the data directory.
|
||||
|
||||
- Added complete action of the TMOP Integrator to account for the spatial
|
||||
derivatives of discrete and analytic targets.
|
||||
|
||||
Performance improvements
|
||||
------------------------
|
||||
- Added support for explicit vectorization in the high-performance templated
|
||||
@@ -41,12 +44,26 @@ Performance improvements
|
||||
- x86 (SSE/AVX/AVX2/AVX512),
|
||||
- Power8 & Power9 (VSX),
|
||||
- BG/Q (QPX).
|
||||
These are now enabled by default, and can be disabled with MFEM_USE_SIMD=NO.
|
||||
These are disabled by default, and can be enabled with MFEM_USE_SIMD=YES.
|
||||
See the new file linalg/simd.hpp and the new directory linalg/simd.
|
||||
|
||||
Improved GPU capabilities
|
||||
-------------------------
|
||||
- Added support for Chebyshev accelerated polynomial smoother on GPU.
|
||||
- The TMOP mesh optimization algorithms were extended to GPU:
|
||||
- QualityMetric #1, #2 and #7 are available in 2D, #302, #303 and #321 in 3D
|
||||
- Both AnalyticAdaptTC and DiscreteAdaptTC TargetConstructor are available
|
||||
- Kernels for normalization and limiting have been added
|
||||
- The AdvectorCG now also support AssemblyLevel::PARTIAL
|
||||
|
||||
- Optimized AMD/HIP kernel support.
|
||||
|
||||
- Added a Full Assembly mode compatible with Device kernel execution. This
|
||||
assembly level builds on top of the current Element Assembly kernels to
|
||||
compute a global sparse matrix. All integrators supported by element assembly
|
||||
are also supported by full assembly. See the '-fa' option in Example 9.
|
||||
|
||||
- Added support for BlockOperator on GPU. See the updated Example 5.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
@@ -71,7 +88,7 @@ Discretization improvements
|
||||
and, in the continuous field case, arbitrary mesh edges and faces.
|
||||
|
||||
- Added new coefficient and vector coefficient classes for QuadratureFunctions.
|
||||
Additionaly, new LinearForm integrators were also added which make use of
|
||||
Additionally, new LinearForm integrators were also added which make use of
|
||||
these new QuadratureFunction coefficient classes.
|
||||
|
||||
- Added support face integrals on the boundaries of NURBS meshes.
|
||||
@@ -88,6 +105,10 @@ Linear and nonlinear solvers
|
||||
and solution during the solving process of an IterativeSolver after every
|
||||
iteration.
|
||||
|
||||
- Added support for the CVODES package in SUNDIALS which provides ODE
|
||||
solvers with sensitivity analysis capabilities. See the CVODESSolver
|
||||
class and the new adjoint miniapps below.
|
||||
|
||||
- Block arrays of parallel matrices can now be merged into a single parallel
|
||||
matrix with the function HypreParMatrixFromBlocks. This could be useful for
|
||||
solving block systems with parallel direct solvers such as STRUMPACK.
|
||||
@@ -115,6 +136,14 @@ New and updated examples and miniapps
|
||||
equations of incompressible fluid dynamics. See the miniapps/navier directory
|
||||
for more details.
|
||||
|
||||
- Added a new miniapps/adjoint directory with two miniapps demonstrating how to
|
||||
solve adjoint problems in MFEM using the CVODES package in SUNDIALS. Both of
|
||||
these miniapps require the MFEM_USE_SUNDIALS configuration option.
|
||||
* The cvsRoberts_ASAi_dns miniapp solves a backward adjoint problem for a
|
||||
system of ODEs, evaluating both forward and adjoint quadratures in serial.
|
||||
* The adjoint_advection_diffusion miniapp solves a backward adjoint problem
|
||||
for an advection diffusion PDE, evaluating adjoint quadratures in parallel.
|
||||
|
||||
- Ported Example 11p to SLEPc, to demonstrate solving the Laplace eigenvalue
|
||||
equation with the shift-and-invert spectral transformation method.
|
||||
|
||||
@@ -125,11 +154,13 @@ New and updated examples and miniapps
|
||||
- Added a new meshing miniapp, Minimal Surface, which solves Plateau's problem:
|
||||
the Dirichlet problem for the minimal surface equation.
|
||||
|
||||
- Added partial assembly support to examples 4/4p and 5/5p, with diagonal
|
||||
- Added partial assembly support to Example 4/4p and Example 5/5p, with diagonal
|
||||
preconditioning.
|
||||
|
||||
- Added a new test problem in example 24/24p, demonstrating a mixed bilinear
|
||||
form for H(div) and L_2, with partial assembly support.
|
||||
- Added full assembly support in Example 9/9p.
|
||||
|
||||
- Added a new test problem in Example 24/24p, demonstrating a mixed bilinear
|
||||
form for H1, H(curl), H(div) and L_2, with partial assembly support.
|
||||
|
||||
- Added weak Dirichlet boundary conditions (Nitsche) to the NURBS miniapp.
|
||||
|
||||
@@ -137,6 +168,8 @@ New and updated examples and miniapps
|
||||
mesh based on element attributes. Any newly exposed boundary elements are
|
||||
assigned attribute numbers related to the trimmed element attributes.
|
||||
|
||||
- Added device support in Example 5/5p.
|
||||
|
||||
Improved testing
|
||||
----------------
|
||||
- Added a GitLab pipeline that automates PR testing on supercomputing systems
|
||||
|
||||
+2
-2
@@ -211,10 +211,10 @@ endif()
|
||||
# SUNDIALS
|
||||
if (MFEM_USE_SUNDIALS)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
find_package(SUNDIALS REQUIRED NVector_Serial CVODE ARKODE KINSOL)
|
||||
find_package(SUNDIALS REQUIRED NVector_Serial CVODES ARKODE KINSOL)
|
||||
else()
|
||||
find_package(SUNDIALS REQUIRED
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODES ARKODE KINSOL)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
||||
@@ -109,6 +109,7 @@ The MFEM source code has the following structure:
|
||||
├── linalg
|
||||
├── mesh
|
||||
├── miniapps
|
||||
│ ├── adjoint
|
||||
│ ├── common
|
||||
│ ├── electromagnetics
|
||||
│ ├── gslib
|
||||
|
||||
@@ -25,5 +25,6 @@ mfem_find_package(SUNDIALS SUNDIALS SUNDIALS_DIR
|
||||
ADD_COMPONENT NVector_ParHyp
|
||||
"include" nvector/nvector_parhyp.h "lib" sundials_nvecparhyp
|
||||
ADD_COMPONENT CVODE "include" cvode/cvode.h "lib" sundials_cvode
|
||||
ADD_COMPONENT CVODES "include" cvodes/cvodes.h "lib" sundials_cvodes
|
||||
ADD_COMPONENT ARKODE "include" arkode/arkode.h "lib" sundials_arkode
|
||||
ADD_COMPONENT KINSOL "include" kinsol/kinsol.h "lib" sundials_kinsol)
|
||||
|
||||
@@ -50,7 +50,7 @@ option(MFEM_USE_OCCA "Enable OCCA" OFF)
|
||||
option(MFEM_USE_RAJA "Enable RAJA" OFF)
|
||||
option(MFEM_USE_CEED "Enable CEED" OFF)
|
||||
option(MFEM_USE_UMPIRE "Enable Umpire" OFF)
|
||||
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" ON)
|
||||
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" OFF)
|
||||
option(MFEM_USE_ADIOS2 "Enable ADIOS2" OFF)
|
||||
|
||||
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
|
||||
@@ -88,6 +88,8 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
|
||||
|
||||
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
|
||||
|
||||
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
|
||||
# and modify cmake variables for hypre for sundials
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-5.0.0/instdir" CACHE PATH
|
||||
"Path to the SUNDIALS library.")
|
||||
# The following may be necessary, if SUNDIALS was built with KLU:
|
||||
|
||||
+12
-4
@@ -138,7 +138,8 @@ MFEM_USE_RAJA = NO
|
||||
MFEM_USE_OCCA = NO
|
||||
MFEM_USE_CEED = NO
|
||||
MFEM_USE_UMPIRE = NO
|
||||
MFEM_USE_SIMD = YES
|
||||
MFEM_USE_CAMP = NO
|
||||
MFEM_USE_SIMD = NO
|
||||
MFEM_USE_ADIOS2 = NO
|
||||
|
||||
# Compile and link options for zlib.
|
||||
@@ -189,10 +190,12 @@ OPENMP_LIB =
|
||||
POSIX_CLOCKS_LIB = -lrt
|
||||
|
||||
# SUNDIALS library configuration
|
||||
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
|
||||
# and modify cmake variables for hypre for sundials
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-5.0.0/instdir
|
||||
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
|
||||
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib64 -L$(SUNDIALS_DIR)/lib64\
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
-lsundials_arkode -lsundials_cvodes -lsundials_nvecserial -lsundials_kinsol
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
|
||||
@@ -339,9 +342,9 @@ GSLIB_DIR = @MFEM_DIR@/../gslib/build
|
||||
GSLIB_OPT = -I$(GSLIB_DIR)/include
|
||||
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
|
||||
|
||||
# CUDA library configuration (currently not needed)
|
||||
# CUDA library configuration
|
||||
CUDA_OPT =
|
||||
CUDA_LIB =
|
||||
CUDA_LIB = -lcusparse
|
||||
|
||||
# HIP library configuration (currently not needed)
|
||||
HIP_OPT =
|
||||
@@ -370,6 +373,11 @@ UMPIRE_DIR = @MFEM_DIR@/../umpire
|
||||
UMPIRE_OPT = -I$(UMPIRE_DIR)/include
|
||||
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -lumpire
|
||||
|
||||
# CAMP library configuration
|
||||
CAMP_DIR = @MFEM_DIR@/../camp
|
||||
CAMP_OPT = -I$(CAMP_DIR)/include
|
||||
CAMP_LIB = -L$(CAMP_DIR)/lib
|
||||
|
||||
# If YES, enable some informational messages
|
||||
VERBOSE = NO
|
||||
|
||||
|
||||
@@ -770,6 +770,7 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/examples/pumi \
|
||||
@MFEM_SOURCE_DIR@/examples/hiop \
|
||||
@MFEM_SOURCE_DIR@/examples/sundials \
|
||||
@MFEM_SOURCE_DIR@/miniapps/adjoint \
|
||||
@MFEM_SOURCE_DIR@/miniapps/common \
|
||||
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
|
||||
@MFEM_SOURCE_DIR@/miniapps/gslib \
|
||||
|
||||
@@ -88,8 +88,8 @@ namespace mfem {
|
||||
* - <a class="el" href="ex24p_8cpp_source.html">Example 24p</a>: parallel mixed finite element spaces and interpolators
|
||||
* - <a class="el" href="ex25_8cpp_source.html">Example 25</a>: simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex25p_8cpp_source.html">Example 25p</a>: parallel simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
*
|
||||
* <H4>SUNDIALS Examples</H4>
|
||||
* - Variants of Examples
|
||||
@@ -101,6 +101,9 @@ namespace mfem {
|
||||
* and
|
||||
* <a class="el" href="sundials_2ex16p_8cpp_source.html">16p</a>
|
||||
* demonstrating the use of MFEM's \link sundials.hpp SUNDIALS classes\endlink
|
||||
* - CVODES adjoint examples:
|
||||
* <a class="el" href="cvsRoberts__ASAi__dns_8cpp_source.html">serial ODE system</a>,
|
||||
* <a class="el" href="adjoint__advection__diffusion_8cpp_source.html">parallel advection-diffusion</a>
|
||||
*
|
||||
* <H4>PETSc Examples</H4>
|
||||
* - Variants of Examples
|
||||
@@ -140,7 +143,9 @@ namespace mfem {
|
||||
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
|
||||
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
|
||||
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
|
||||
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
|
||||
* - <a class="el" href="toroid_8cpp_source.html">Toroid</a>: generate simple toroidal meshes
|
||||
* - <a class="el" href="twist_8cpp_source.html">Twist</a>: generate simple periodic meshes
|
||||
@@ -157,7 +162,6 @@ namespace mfem {
|
||||
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
|
||||
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
|
||||
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
|
||||
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
|
||||
*
|
||||
|
||||
+34
-31
@@ -103,8 +103,8 @@ int main(int argc, char *argv[])
|
||||
// 3. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
|
||||
// the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
|
||||
// 4. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
|
||||
@@ -112,10 +112,10 @@ int main(int argc, char *argv[])
|
||||
// elements.
|
||||
{
|
||||
int ref_levels =
|
||||
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
|
||||
(int)floor(log(50000./mesh.GetNE())/log(2.)/dim);
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
mesh.UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -123,66 +123,70 @@ int main(int argc, char *argv[])
|
||||
// Lagrange finite elements of the specified order. If order < 1, we
|
||||
// instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (mesh->GetNodes())
|
||||
else if (mesh.GetNodes())
|
||||
{
|
||||
fec = mesh->GetNodes()->OwnFEC();
|
||||
fec = mesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
|
||||
FiniteElementSpace fespace(&mesh, fec);
|
||||
cout << "Number of finite element unknowns: "
|
||||
<< fespace->GetTrueVSize() << endl;
|
||||
<< fespace.GetTrueVSize() << endl;
|
||||
|
||||
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
|
||||
// In this example, the boundary conditions are defined by marking all
|
||||
// the boundary attributes from the mesh as essential (Dirichlet) and
|
||||
// converting them to a list of true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
if (mesh->bdr_attributes.Size())
|
||||
if (mesh.bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(mesh->bdr_attributes.Max());
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
|
||||
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
|
||||
// the basis functions in the finite element fespace.
|
||||
LinearForm *b = new LinearForm(fespace);
|
||||
LinearForm b(&fespace);
|
||||
ConstantCoefficient one(1.0);
|
||||
b->AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b->Assemble();
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b.Assemble();
|
||||
|
||||
// 8. Define the solution vector x as a finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
GridFunction x(fespace);
|
||||
GridFunction x(&fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 9. Set up the bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
|
||||
// domain integrator.
|
||||
BilinearForm *a = new BilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
BilinearForm a(&fespace);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
|
||||
// 10. Assemble the bilinear form and the corresponding linear system,
|
||||
// applying any necessary transformations such as: eliminating boundary
|
||||
// conditions, applying conforming constraints for non-conforming AMR,
|
||||
// static condensation, etc.
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
cout << "Size of linear system: " << A->Height() << endl;
|
||||
|
||||
@@ -203,9 +207,9 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
else // Jacobi preconditioning in partial assembly mode
|
||||
{
|
||||
if (UsesTensorBasis(*fespace))
|
||||
if (UsesTensorBasis(fespace))
|
||||
{
|
||||
OperatorJacobiSmoother M(*a, ess_tdof_list);
|
||||
OperatorJacobiSmoother M(a, ess_tdof_list);
|
||||
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
|
||||
}
|
||||
else
|
||||
@@ -215,13 +219,13 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 12. Recover the solution as a finite element grid function.
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// 13. Save the refined mesh and the solution. This output can be viewed later
|
||||
// using GLVis: "glvis -m refined.mesh -g sol.gf".
|
||||
ofstream mesh_ofs("refined.mesh");
|
||||
mesh_ofs.precision(8);
|
||||
mesh->Print(mesh_ofs);
|
||||
mesh.Print(mesh_ofs);
|
||||
ofstream sol_ofs("sol.gf");
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
@@ -233,15 +237,14 @@ int main(int argc, char *argv[])
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << *mesh << x << flush;
|
||||
sol_sock << "solution\n" << mesh << x << flush;
|
||||
}
|
||||
|
||||
// 15. Free the used memory.
|
||||
delete a;
|
||||
delete b;
|
||||
delete fespace;
|
||||
if (order > 0) { delete fec; }
|
||||
delete mesh;
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
+37
-35
@@ -112,8 +112,8 @@ int main(int argc, char *argv[])
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement. We choose
|
||||
@@ -121,23 +121,23 @@ int main(int argc, char *argv[])
|
||||
// more than 10,000 elements.
|
||||
{
|
||||
int ref_levels =
|
||||
(int)floor(log(10000./mesh->GetNE())/log(2.)/dim);
|
||||
(int)floor(log(10000./mesh.GetNE())/log(2.)/dim);
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
mesh.UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
mesh.Clear();
|
||||
{
|
||||
int par_ref_levels = 2;
|
||||
for (int l = 0; l < par_ref_levels; l++)
|
||||
{
|
||||
pmesh->UniformRefinement();
|
||||
pmesh.UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -145,13 +145,16 @@ int main(int argc, char *argv[])
|
||||
// use continuous Lagrange finite elements of the specified order. If
|
||||
// order < 1, we instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (pmesh->GetNodes())
|
||||
else if (pmesh.GetNodes())
|
||||
{
|
||||
fec = pmesh->GetNodes()->OwnFEC();
|
||||
fec = pmesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
@@ -160,9 +163,10 @@ int main(int argc, char *argv[])
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
|
||||
HYPRE_Int size = fespace->GlobalTrueVSize();
|
||||
ParFiniteElementSpace fespace(&pmesh, fec);
|
||||
HYPRE_Int size = fespace.GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element unknowns: " << size << endl;
|
||||
@@ -173,44 +177,44 @@ int main(int argc, char *argv[])
|
||||
// by marking all the boundary attributes from the mesh as essential
|
||||
// (Dirichlet) and converting them to a list of true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
if (pmesh->bdr_attributes.Size())
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(pmesh->bdr_attributes.Max());
|
||||
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 9. Set up the parallel linear form b(.) which corresponds to the
|
||||
// right-hand side of the FEM linear system, which in this case is
|
||||
// (1,phi_i) where phi_i are the basis functions in fespace.
|
||||
ParLinearForm *b = new ParLinearForm(fespace);
|
||||
ParLinearForm b(&fespace);
|
||||
ConstantCoefficient one(1.0);
|
||||
b->AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b->Assemble();
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b.Assemble();
|
||||
|
||||
// 10. Define the solution vector x as a parallel finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
ParGridFunction x(fespace);
|
||||
ParGridFunction x(&fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 11. Set up the parallel bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
|
||||
// domain integrator.
|
||||
ParBilinearForm *a = new ParBilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
ParBilinearForm a(&fespace);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
|
||||
// 12. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, static condensation, etc.
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
// 13. Solve the linear system A X = B.
|
||||
// * With full assembly, use the BoomerAMG preconditioner from hypre.
|
||||
@@ -218,9 +222,9 @@ int main(int argc, char *argv[])
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
{
|
||||
if (UsesTensorBasis(*fespace))
|
||||
if (UsesTensorBasis(fespace))
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -238,7 +242,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
@@ -249,7 +253,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh->Print(mesh_ofs);
|
||||
pmesh.Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
@@ -264,16 +268,14 @@ int main(int argc, char *argv[])
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << *pmesh << x << flush;
|
||||
sol_sock << "solution\n" << pmesh << x << flush;
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
delete a;
|
||||
delete b;
|
||||
delete fespace;
|
||||
if (order > 0) { delete fec; }
|
||||
delete pmesh;
|
||||
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
MPI_Finalize();
|
||||
|
||||
return 0;
|
||||
|
||||
+37
-28
@@ -13,6 +13,11 @@
|
||||
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
|
||||
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
|
||||
//
|
||||
// With partial assembly:
|
||||
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1 -pa
|
||||
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2 -pa
|
||||
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0 -pa
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define and
|
||||
// solve simple complex-valued linear systems. It implements three
|
||||
// variants of a damped harmonic oscillator:
|
||||
@@ -76,6 +81,7 @@ int main(int argc, char *argv[])
|
||||
bool visualization = 1;
|
||||
bool herm_conv = true;
|
||||
bool exact_sol = true;
|
||||
bool pa = false;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -106,6 +112,8 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
@@ -282,6 +290,7 @@ int main(int argc, char *argv[])
|
||||
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
|
||||
|
||||
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -318,6 +327,8 @@ int main(int argc, char *argv[])
|
||||
// -Grad(a Div) - omega^2 b + omega c
|
||||
//
|
||||
BilinearForm *pcOp = new BilinearForm(fespace);
|
||||
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -348,19 +359,8 @@ int main(int argc, char *argv[])
|
||||
Vector B, U;
|
||||
|
||||
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
|
||||
u = 0.0;
|
||||
U = 0.0;
|
||||
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
|
||||
{
|
||||
ComplexSparseMatrix * Asp =
|
||||
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
|
||||
|
||||
cout << "Size of linear system: "
|
||||
<< 2 * Asp->real().Width() << endl << endl;
|
||||
}
|
||||
cout << "Size of linear system: " << A->Width() << endl << endl;
|
||||
|
||||
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// preconditioner based on the appropriate sparse smoother.
|
||||
@@ -368,8 +368,8 @@ int main(int argc, char *argv[])
|
||||
Array<int> blockOffsets;
|
||||
blockOffsets.SetSize(3);
|
||||
blockOffsets[0] = 0;
|
||||
blockOffsets[1] = PCOp.Ptr()->Height();
|
||||
blockOffsets[2] = PCOp.Ptr()->Height();
|
||||
blockOffsets[1] = A->Height() / 2;
|
||||
blockOffsets[2] = A->Height() / 2;
|
||||
blockOffsets.PartialSum();
|
||||
|
||||
BlockDiagonalPreconditioner BDP(blockOffsets);
|
||||
@@ -377,22 +377,31 @@ int main(int argc, char *argv[])
|
||||
Operator * pc_r = NULL;
|
||||
Operator * pc_i = NULL;
|
||||
|
||||
double s = 1.0;
|
||||
switch (prob)
|
||||
if (pa)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
|
||||
s = -1.0;
|
||||
break;
|
||||
case 2:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
|
||||
default: break; // This should be unreachable
|
||||
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorHandle PCOp;
|
||||
pcOp->SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 2:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
default:
|
||||
break; // This should be unreachable
|
||||
}
|
||||
}
|
||||
double s = (prob != 1) ? 1.0 : -1.0;
|
||||
pc_i = new ScaledOperator(pc_r,
|
||||
(conv == ComplexOperator::HERMITIAN) ?
|
||||
s:-s);
|
||||
|
||||
+39
-28
@@ -13,6 +13,11 @@
|
||||
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
|
||||
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
|
||||
//
|
||||
// With partial assembly:
|
||||
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 1 -p 1 -pa
|
||||
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 1 -p 2 -pa
|
||||
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0 -pa
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define and
|
||||
// solve simple complex-valued linear systems. It implements three
|
||||
// variants of a damped harmonic oscillator:
|
||||
@@ -84,6 +89,7 @@ int main(int argc, char *argv[])
|
||||
bool visualization = 1;
|
||||
bool herm_conv = true;
|
||||
bool exact_sol = true;
|
||||
bool pa = false;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -116,6 +122,8 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
@@ -315,6 +323,7 @@ int main(int argc, char *argv[])
|
||||
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
|
||||
|
||||
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -351,6 +360,7 @@ int main(int argc, char *argv[])
|
||||
// -Grad(a Div) - omega^2 b + omega c
|
||||
//
|
||||
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
|
||||
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -382,19 +392,11 @@ int main(int argc, char *argv[])
|
||||
Vector B, U;
|
||||
|
||||
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
|
||||
u = 0.0;
|
||||
U = 0.0;
|
||||
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
ComplexHypreParMatrix * Ahyp =
|
||||
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
|
||||
|
||||
cout << "Size of linear system: "
|
||||
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
|
||||
<< 2 * fespace->GlobalTrueVSize() << endl << endl;
|
||||
}
|
||||
|
||||
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
|
||||
@@ -404,8 +406,8 @@ int main(int argc, char *argv[])
|
||||
Array<int> blockTrueOffsets;
|
||||
blockTrueOffsets.SetSize(3);
|
||||
blockTrueOffsets[0] = 0;
|
||||
blockTrueOffsets[1] = PCOp.Ptr()->Height();
|
||||
blockTrueOffsets[2] = PCOp.Ptr()->Height();
|
||||
blockTrueOffsets[1] = A->Height() / 2;
|
||||
blockTrueOffsets[2] = A->Height() / 2;
|
||||
blockTrueOffsets.PartialSum();
|
||||
|
||||
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
|
||||
@@ -413,25 +415,34 @@ int main(int argc, char *argv[])
|
||||
Operator * pc_r = NULL;
|
||||
Operator * pc_i = NULL;
|
||||
|
||||
switch (prob)
|
||||
if (pa)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
break;
|
||||
case 2:
|
||||
if (dim == 2 )
|
||||
{
|
||||
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
else
|
||||
{
|
||||
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
break;
|
||||
case 2:
|
||||
if (dim == 2 )
|
||||
{
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
else
|
||||
{
|
||||
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
}
|
||||
pc_i = new ScaledOperator(pc_r,
|
||||
(conv == ComplexOperator::HERMITIAN) ?
|
||||
|
||||
+86
-7
@@ -7,6 +7,7 @@
|
||||
// ex24 -m ../data/beam-tet.mesh
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 1
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 2
|
||||
// ex24 -m ../data/escher.mesh
|
||||
// ex24 -m ../data/escher.mesh -o 2
|
||||
// ex24 -m ../data/fichera.mesh
|
||||
@@ -24,12 +25,13 @@
|
||||
// ex24 -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code illustrates usage of mixed finite element
|
||||
// spaces, with two variants:
|
||||
// spaces, with three variants:
|
||||
//
|
||||
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
|
||||
// 2) (div v, q) for v in H(div) tested against q in L_2
|
||||
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
|
||||
// 3) (div v, q) for v in H(div) tested against q in L_2
|
||||
//
|
||||
// Using different approaches, we project the gradient or
|
||||
// Using different approaches, we project the gradient, curl, or
|
||||
// divergence to the appropriate space.
|
||||
//
|
||||
// We recommend viewing examples 1, 3, and 5 before viewing this
|
||||
@@ -45,8 +47,11 @@ using namespace mfem;
|
||||
double p_exact(const Vector &x);
|
||||
void gradp_exact(const Vector &, Vector &);
|
||||
double div_gradp_exact(const Vector &x);
|
||||
void v_exact(const Vector &x, Vector &v);
|
||||
void curlv_exact(const Vector &x, Vector &cv);
|
||||
|
||||
int dim;
|
||||
double freq = 1.0, kappa;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
@@ -83,6 +88,7 @@ int main(int argc, char *argv[])
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
@@ -119,10 +125,15 @@ int main(int argc, char *argv[])
|
||||
trial_fec = new H1_FECollection(order, dim);
|
||||
test_fec = new ND_FECollection(order, dim);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
trial_fec = new ND_FECollection(order, dim);
|
||||
test_fec = new RT_FECollection(order-1, dim);
|
||||
}
|
||||
else
|
||||
{
|
||||
trial_fec = new RT_FECollection(order - 1, dim);
|
||||
test_fec = new L2_FECollection(order - 1, dim);
|
||||
trial_fec = new RT_FECollection(order-1, dim);
|
||||
test_fec = new L2_FECollection(order-1, dim);
|
||||
}
|
||||
|
||||
FiniteElementSpace trial_fes(mesh, trial_fec);
|
||||
@@ -136,6 +147,12 @@ int main(int argc, char *argv[])
|
||||
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
|
||||
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
|
||||
endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: "
|
||||
@@ -150,12 +167,18 @@ int main(int argc, char *argv[])
|
||||
GridFunction x(&test_fes);
|
||||
FunctionCoefficient p_coef(p_exact);
|
||||
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
|
||||
VectorFunctionCoefficient v_coef(sdim, v_exact);
|
||||
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
|
||||
FunctionCoefficient divgradp_coef(div_gradp_exact);
|
||||
|
||||
if (prob == 0)
|
||||
{
|
||||
gftrial.ProjectCoefficient(p_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
gftrial.ProjectCoefficient(v_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
gftrial.ProjectCoefficient(gradp_coef);
|
||||
@@ -179,6 +202,11 @@ int main(int argc, char *argv[])
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
|
||||
}
|
||||
else
|
||||
{
|
||||
a.AddDomainIntegrator(new MassIntegrator(one));
|
||||
@@ -244,6 +272,10 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
dlo.AddDomainInterpolator(new GradientInterpolator());
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
dlo.AddDomainInterpolator(new CurlInterpolator());
|
||||
}
|
||||
else
|
||||
{
|
||||
dlo.AddDomainInterpolator(new DivergenceInterpolator());
|
||||
@@ -258,6 +290,10 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
exact_proj.ProjectCoefficient(gradp_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
exact_proj.ProjectCoefficient(curlv_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
exact_proj.ProjectCoefficient(divgradp_coef);
|
||||
@@ -276,10 +312,23 @@ int main(int argc, char *argv[])
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
|
||||
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
" ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
double errSol = x.ComputeL2Error(curlv_coef);
|
||||
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
|
||||
double errProj = exact_proj.ComputeL2Error(curlv_coef);
|
||||
|
||||
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in H(div): "
|
||||
"|| E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
int order_quad = max(2, 2*order+1);
|
||||
@@ -295,7 +344,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
|
||||
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v "
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
@@ -371,3 +420,33 @@ double div_gradp_exact(const Vector &x)
|
||||
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void v_exact(const Vector &x, Vector &v)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(2));
|
||||
v(2) = sin(kappa * x(0));
|
||||
}
|
||||
else
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(0));
|
||||
if (x.Size() == 3) { v(2) = 0.0; }
|
||||
}
|
||||
}
|
||||
|
||||
void curlv_exact(const Vector &x, Vector &cv)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
cv(0) = -kappa * cos(kappa * x(2));
|
||||
cv(1) = -kappa * cos(kappa * x(0));
|
||||
cv(2) = -kappa * cos(kappa * x(1));
|
||||
}
|
||||
else
|
||||
{
|
||||
cv = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
+93
-11
@@ -6,7 +6,8 @@
|
||||
// mpirun -np 4 ex24p -m ../data/square-disc.mesh -o 2
|
||||
// mpirun -np 4 ex24p -m ../data/beam-tet.mesh
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -p 1 -pa
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 1
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 2
|
||||
// mpirun -np 4 ex24p -m ../data/escher.mesh
|
||||
// mpirun -np 4 ex24p -m ../data/escher.mesh -o 2
|
||||
// mpirun -np 4 ex24p -m ../data/fichera.mesh
|
||||
@@ -24,12 +25,13 @@
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code illustrates usage of mixed finite element
|
||||
// spaces, with two variants:
|
||||
// spaces, with three variants:
|
||||
//
|
||||
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
|
||||
// 2) (div v, q) for v in H(div) tested against q in L_2
|
||||
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
|
||||
// 3) (div v, q) for v in H(div) tested against q in L_2
|
||||
//
|
||||
// Using different approaches, we project the gradient or
|
||||
// Using different approaches, we project the gradient, curl, or
|
||||
// divergence to the appropriate space.
|
||||
//
|
||||
// We recommend viewing examples 1, 3, and 5 before viewing this
|
||||
@@ -45,8 +47,11 @@ using namespace mfem;
|
||||
double p_exact(const Vector &x);
|
||||
void gradp_exact(const Vector &, Vector &);
|
||||
double div_gradp_exact(const Vector &x);
|
||||
void v_exact(const Vector &x, Vector &v);
|
||||
void curlv_exact(const Vector &x, Vector &cv);
|
||||
|
||||
int dim;
|
||||
double freq = 1.0, kappa;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
@@ -96,6 +101,7 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
@@ -147,10 +153,15 @@ int main(int argc, char *argv[])
|
||||
trial_fec = new H1_FECollection(order, dim);
|
||||
test_fec = new ND_FECollection(order, dim);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
trial_fec = new ND_FECollection(order, dim);
|
||||
test_fec = new RT_FECollection(order-1, dim);
|
||||
}
|
||||
else
|
||||
{
|
||||
trial_fec = new RT_FECollection(order - 1, dim);
|
||||
test_fec = new L2_FECollection(order - 1, dim);
|
||||
trial_fec = new RT_FECollection(order-1, dim);
|
||||
test_fec = new L2_FECollection(order-1, dim);
|
||||
}
|
||||
|
||||
ParFiniteElementSpace trial_fes(pmesh, trial_fec);
|
||||
@@ -166,6 +177,12 @@ int main(int argc, char *argv[])
|
||||
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
|
||||
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
|
||||
endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: "
|
||||
@@ -181,12 +198,18 @@ int main(int argc, char *argv[])
|
||||
ParGridFunction x(&test_fes);
|
||||
FunctionCoefficient p_coef(p_exact);
|
||||
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
|
||||
VectorFunctionCoefficient v_coef(sdim, v_exact);
|
||||
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
|
||||
FunctionCoefficient divgradp_coef(div_gradp_exact);
|
||||
|
||||
if (prob == 0)
|
||||
{
|
||||
gftrial.ProjectCoefficient(p_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
gftrial.ProjectCoefficient(v_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
gftrial.ProjectCoefficient(gradp_coef);
|
||||
@@ -210,6 +233,11 @@ int main(int argc, char *argv[])
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
|
||||
}
|
||||
else
|
||||
{
|
||||
a.AddDomainIntegrator(new MassIntegrator(one));
|
||||
@@ -293,6 +321,10 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
dlo.AddDomainInterpolator(new GradientInterpolator());
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
dlo.AddDomainInterpolator(new CurlInterpolator());
|
||||
}
|
||||
else
|
||||
{
|
||||
dlo.AddDomainInterpolator(new DivergenceInterpolator());
|
||||
@@ -307,6 +339,10 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
exact_proj.ProjectCoefficient(gradp_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
exact_proj.ProjectCoefficient(curlv_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
exact_proj.ProjectCoefficient(divgradp_coef);
|
||||
@@ -324,14 +360,30 @@ int main(int argc, char *argv[])
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
|
||||
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl)"
|
||||
": || E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad"
|
||||
" p ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
double errSol = x.ComputeL2Error(curlv_coef);
|
||||
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
|
||||
double errProj = exact_proj.ComputeL2Error(curlv_coef);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in "
|
||||
"H(div): || E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
int order_quad = max(2, 2*order+1);
|
||||
@@ -350,7 +402,7 @@ int main(int argc, char *argv[])
|
||||
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
|
||||
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
" ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
@@ -436,3 +488,33 @@ double div_gradp_exact(const Vector &x)
|
||||
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void v_exact(const Vector &x, Vector &v)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(2));
|
||||
v(2) = sin(kappa * x(0));
|
||||
}
|
||||
else
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(0));
|
||||
if (x.Size() == 3) { v(2) = 0.0; }
|
||||
}
|
||||
}
|
||||
|
||||
void curlv_exact(const Vector &x, Vector &cv)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
cv(0) = -kappa * cos(kappa * x(2));
|
||||
cv(1) = -kappa * cos(kappa * x(0));
|
||||
cv(2) = -kappa * cos(kappa * x(1));
|
||||
}
|
||||
else
|
||||
{
|
||||
cv = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
+15
-23
@@ -389,27 +389,22 @@ int main(int argc, char *argv[])
|
||||
// applying any necessary transformations such as: assembly, eliminating
|
||||
// boundary conditions, applying conforming constraints for
|
||||
// non-conforming AMR, etc.
|
||||
a.Assemble();
|
||||
a.Assemble(0);
|
||||
|
||||
OperatorHandle Ah;
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
// 13. Transform to monolithic SparseMatrix
|
||||
SparseMatrix *A = Ah.As<ComplexSparseMatrix>()->GetSystemMatrix();
|
||||
|
||||
cout << "Size of linear system: " << A->Height() << endl;
|
||||
|
||||
// 14. Solve using a direct or an iterative solver
|
||||
// 13. Solve using a direct or an iterative solver
|
||||
#ifdef MFEM_USE_SUITESPARSE
|
||||
{
|
||||
UMFPackSolver solver(*A);
|
||||
solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
solver.Mult(B, X);
|
||||
ComplexUMFPackSolver csolver(*A.As<ComplexSparseMatrix>());
|
||||
csolver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
csolver.SetPrintLevel(1);
|
||||
csolver.Mult(B, X);
|
||||
}
|
||||
#else
|
||||
|
||||
// 14a. Set up the Bilinear form a(.,.) for the preconditioner
|
||||
// 13a. Set up the Bilinear form a(.,.) for the preconditioner
|
||||
//
|
||||
// In Comp
|
||||
// Domain: 1/mu (Curl E, Curl F) + omega^2 * epsilon (E,F)
|
||||
@@ -437,10 +432,10 @@ int main(int argc, char *argv[])
|
||||
|
||||
prec.Assemble();
|
||||
|
||||
OperatorHandle PCOpAh;
|
||||
OperatorPtr PCOpAh;
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// 14b. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// 13b. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// preconditioner based on the Gauss-Seidel sparse smoother.
|
||||
Array<int> offsets(3);
|
||||
offsets[0] = 0;
|
||||
@@ -467,17 +462,15 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
#endif
|
||||
|
||||
// 15. Recover the solution as a finite element grid function and compute the
|
||||
// 14. Recover the solution as a finite element grid function and compute the
|
||||
// errors if the exact solution is known.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// If exact is known compute the error
|
||||
if (exact_known)
|
||||
{
|
||||
ComplexGridFunction x_gf(fespace);
|
||||
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
|
||||
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
|
||||
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
|
||||
int order_quad = max(2, 2 * order + 1);
|
||||
const IntegrationRule *irs[Geometry::NumGeom];
|
||||
for (int i = 0; i < Geometry::NumGeom; ++i)
|
||||
@@ -506,7 +499,7 @@ int main(int argc, char *argv[])
|
||||
<< sqrt(L2Error_Re*L2Error_Re + L2Error_Im*L2Error_Im) << "\n\n";
|
||||
}
|
||||
|
||||
// 16. Save the refined mesh and the solution. This output can be viewed
|
||||
// 15. Save the refined mesh and the solution. This output can be viewed
|
||||
// later using GLVis: "glvis -m mesh -g sol".
|
||||
{
|
||||
ofstream mesh_ofs("ex25.mesh");
|
||||
@@ -521,7 +514,7 @@ int main(int argc, char *argv[])
|
||||
x.imag().Save(sol_i_ofs);
|
||||
}
|
||||
|
||||
// 17. Send the solution by socket to a GLVis server.
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
// Define visualization keys for GLVis (see GLVis documentation)
|
||||
@@ -572,8 +565,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 18. Free the used memory.
|
||||
delete A;
|
||||
// 17. Free the used memory.
|
||||
delete pml;
|
||||
delete fespace;
|
||||
delete fec;
|
||||
|
||||
+7
-16
@@ -419,21 +419,15 @@ int main(int argc, char *argv[])
|
||||
// constraints for non-conforming AMR, etc.
|
||||
a.Assemble();
|
||||
|
||||
OperatorHandle Ah;
|
||||
OperatorPtr Ah;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
|
||||
// 15. Transform to monolithic HypreParMatrix
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Size of linear system: " << A->GetGlobalNumRows() << endl;
|
||||
}
|
||||
|
||||
// 16. Solve using a direct or an iterative solver
|
||||
// 15. Solve using a direct or an iterative solver
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
{
|
||||
// Transform to monolithic HypreParMatrix
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
SuperLURowLocMatrix SA(*A);
|
||||
SuperLUSolver superlu(MPI_COMM_WORLD);
|
||||
superlu.SetPrintStatistics(false);
|
||||
@@ -441,9 +435,9 @@ int main(int argc, char *argv[])
|
||||
superlu.SetColumnPermutation(superlu::PARMETIS);
|
||||
superlu.SetOperator(SA);
|
||||
superlu.Mult(B, X);
|
||||
delete A;
|
||||
}
|
||||
#else
|
||||
|
||||
// 16a. Set up the parallel Bilinear form a(.,.) for the preconditioner
|
||||
//
|
||||
// In Comp
|
||||
@@ -472,7 +466,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
prec.Assemble();
|
||||
|
||||
OperatorHandle PCOpAh;
|
||||
OperatorPtr PCOpAh;
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// 16b. Define and apply a parallel GMRES solver for AU=B with a block
|
||||
@@ -496,7 +490,7 @@ int main(int argc, char *argv[])
|
||||
gmres.SetMaxIter(2000);
|
||||
gmres.SetRelTol(1e-5);
|
||||
gmres.SetAbsTol(0.0);
|
||||
gmres.SetOperator(*A);
|
||||
gmres.SetOperator(*Ah);
|
||||
gmres.SetPreconditioner(BlockAMS);
|
||||
gmres.Mult(B, X);
|
||||
}
|
||||
@@ -509,10 +503,8 @@ int main(int argc, char *argv[])
|
||||
// If exact is known compute the error
|
||||
if (exact_known)
|
||||
{
|
||||
ParComplexGridFunction x_gf(fespace);
|
||||
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
|
||||
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
|
||||
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
|
||||
int order_quad = max(2, 2 * order + 1);
|
||||
const IntegrationRule *irs[Geometry::NumGeom];
|
||||
for (int i = 0; i < Geometry::NumGeom; ++i)
|
||||
@@ -629,7 +621,6 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 20. Free the used memory.
|
||||
delete A;
|
||||
delete pml;
|
||||
delete fespace;
|
||||
delete fec;
|
||||
|
||||
+36
-17
@@ -11,6 +11,12 @@
|
||||
// ex5 -m ../data/escher.mesh
|
||||
// ex5 -m ../data/fichera.mesh
|
||||
//
|
||||
// Device sample runs:
|
||||
// ex5 -m ../data/star.mesh -pa -d cuda
|
||||
// ex5 -m ../data/star.mesh -pa -d raja-cuda
|
||||
// ex5 -m ../data/star.mesh -pa -d raja-omp
|
||||
// ex5 -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple 2D/3D mixed Darcy problem
|
||||
// corresponding to the saddle point system
|
||||
// k*u + grad p = f
|
||||
@@ -50,6 +56,7 @@ int main(int argc, char *argv[])
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int order = 1;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
@@ -59,6 +66,8 @@ int main(int argc, char *argv[])
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
@@ -70,13 +79,18 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
|
||||
// 2. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
device.Print();
|
||||
|
||||
// 3. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
|
||||
// the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 3. Refine the mesh to increase the resolution. In this example we do
|
||||
// 4. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
|
||||
// largest number that gives a final mesh with no more than 10,000
|
||||
// elements.
|
||||
@@ -89,7 +103,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Define a finite element space on the mesh. Here we use the
|
||||
// 5. Define a finite element space on the mesh. Here we use the
|
||||
// Raviart-Thomas finite elements of the specified order.
|
||||
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
|
||||
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
|
||||
@@ -97,7 +111,7 @@ int main(int argc, char *argv[])
|
||||
FiniteElementSpace *R_space = new FiniteElementSpace(mesh, hdiv_coll);
|
||||
FiniteElementSpace *W_space = new FiniteElementSpace(mesh, l2_coll);
|
||||
|
||||
// 5. Define the BlockStructure of the problem, i.e. define the array of
|
||||
// 6. Define the BlockStructure of the problem, i.e. define the array of
|
||||
// offsets for each variable. The last component of the Array is the sum
|
||||
// of the dimensions of each block.
|
||||
Array<int> block_offsets(3); // number of variables + 1
|
||||
@@ -112,7 +126,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "dim(R+W) = " << block_offsets.Last() << "\n";
|
||||
std::cout << "***********************************************************\n";
|
||||
|
||||
// 6. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
// 7. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
ConstantCoefficient k(1.0);
|
||||
|
||||
VectorFunctionCoefficient fcoeff(dim, fFun);
|
||||
@@ -122,25 +136,28 @@ int main(int argc, char *argv[])
|
||||
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
|
||||
FunctionCoefficient pcoeff(pFun_ex);
|
||||
|
||||
// 7. Allocate memory (x, rhs) for the analytical solution and the right hand
|
||||
// 8. Allocate memory (x, rhs) for the analytical solution and the right hand
|
||||
// side. Define the GridFunction u,p for the finite element solution and
|
||||
// linear forms fform and gform for the right hand side. The data
|
||||
// allocated by x and rhs are passed as a reference to the grid functions
|
||||
// (u,p) and the linear forms (fform, gform).
|
||||
BlockVector x(block_offsets), rhs(block_offsets);
|
||||
MemoryType mt = device.GetMemoryType();
|
||||
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
|
||||
|
||||
LinearForm *fform(new LinearForm);
|
||||
fform->Update(R_space, rhs.GetBlock(0), 0);
|
||||
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
|
||||
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
|
||||
fform->Assemble();
|
||||
fform->SyncAliasMemory(rhs);
|
||||
|
||||
LinearForm *gform(new LinearForm);
|
||||
gform->Update(W_space, rhs.GetBlock(1), 0);
|
||||
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
|
||||
gform->Assemble();
|
||||
gform->SyncAliasMemory(rhs);
|
||||
|
||||
// 8. Assemble the finite element matrices for the Darcy operator
|
||||
// 9. Assemble the finite element matrices for the Darcy operator
|
||||
//
|
||||
// D = [ M B^T ]
|
||||
// [ B 0 ]
|
||||
@@ -185,7 +202,7 @@ int main(int argc, char *argv[])
|
||||
darcyOp.SetBlock(1,0, &B);
|
||||
}
|
||||
|
||||
// 9. Construct the operators for preconditioner
|
||||
// 10. Construct the operators for preconditioner
|
||||
//
|
||||
// P = [ diag(M) 0 ]
|
||||
// [ 0 B diag(M)^-1 B^T ]
|
||||
@@ -202,10 +219,11 @@ int main(int argc, char *argv[])
|
||||
if (pa)
|
||||
{
|
||||
mVarf->AssembleDiagonal(Md);
|
||||
auto Md_host = Md.HostRead();
|
||||
Vector invMd(mVarf->Height());
|
||||
for (int i=0; i<mVarf->Height(); ++i)
|
||||
{
|
||||
invMd(i) = 1.0 / Md(i);
|
||||
invMd(i) = 1.0 / Md_host[i];
|
||||
}
|
||||
|
||||
Vector BMBt_diag(bVarf->Height());
|
||||
@@ -246,7 +264,7 @@ int main(int argc, char *argv[])
|
||||
darcyPrec.SetDiagonalBlock(0, invM);
|
||||
darcyPrec.SetDiagonalBlock(1, invS);
|
||||
|
||||
// 10. Solve the linear system with MINRES.
|
||||
// 11. Solve the linear system with MINRES.
|
||||
// Check the norm of the unpreconditioned residual.
|
||||
int maxIter(1000);
|
||||
double rtol(1.e-6);
|
||||
@@ -263,6 +281,7 @@ int main(int argc, char *argv[])
|
||||
solver.SetPrintLevel(1);
|
||||
x = 0.0;
|
||||
solver.Mult(rhs, x);
|
||||
if (device.IsEnabled()) { x.HostRead(); }
|
||||
chrono.Stop();
|
||||
|
||||
if (solver.GetConverged())
|
||||
@@ -273,7 +292,7 @@ int main(int argc, char *argv[])
|
||||
<< " iterations. Residual norm is " << solver.GetFinalNorm() << ".\n";
|
||||
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
|
||||
|
||||
// 11. Create the grid functions u and p. Compute the L2 error norms.
|
||||
// 12. Create the grid functions u and p. Compute the L2 error norms.
|
||||
GridFunction u, p;
|
||||
u.MakeRef(R_space, x.GetBlock(0), 0);
|
||||
p.MakeRef(W_space, x.GetBlock(1), 0);
|
||||
@@ -293,7 +312,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "|| u_h - u_ex || / || u_ex || = " << err_u / norm_u << "\n";
|
||||
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
|
||||
|
||||
// 12. Save the mesh and the solution. This output can be viewed later using
|
||||
// 13. Save the mesh and the solution. This output can be viewed later using
|
||||
// GLVis: "glvis -m ex5.mesh -g sol_u.gf" or "glvis -m ex5.mesh -g
|
||||
// sol_p.gf".
|
||||
{
|
||||
@@ -310,13 +329,13 @@ int main(int argc, char *argv[])
|
||||
p.Save(p_ofs);
|
||||
}
|
||||
|
||||
// 13. Save data in the VisIt format
|
||||
// 14. Save data in the VisIt format
|
||||
VisItDataCollection visit_dc("Example5", mesh);
|
||||
visit_dc.RegisterField("velocity", &u);
|
||||
visit_dc.RegisterField("pressure", &p);
|
||||
visit_dc.Save();
|
||||
|
||||
// 14. Save data in the ParaView format
|
||||
// 15. Save data in the ParaView format
|
||||
ParaViewDataCollection paraview_dc("Example5", mesh);
|
||||
paraview_dc.SetPrefixPath("ParaView");
|
||||
paraview_dc.SetLevelsOfDetail(order);
|
||||
@@ -328,7 +347,7 @@ int main(int argc, char *argv[])
|
||||
paraview_dc.RegisterField("pressure",&p);
|
||||
paraview_dc.Save();
|
||||
|
||||
// 15. Send the solution by socket to a GLVis server.
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
@@ -341,7 +360,7 @@ int main(int argc, char *argv[])
|
||||
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
|
||||
}
|
||||
|
||||
// 16. Free the used memory.
|
||||
// 17. Free the used memory.
|
||||
delete fform;
|
||||
delete gform;
|
||||
delete invM;
|
||||
|
||||
+42
-21
@@ -11,6 +11,12 @@
|
||||
// mpirun -np 4 ex5p -m ../data/escher.mesh
|
||||
// mpirun -np 4 ex5p -m ../data/fichera.mesh
|
||||
//
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d cuda
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-cuda
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-omp
|
||||
// mpirun -np 4 ex5p -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple 2D/3D mixed Darcy problem
|
||||
// corresponding to the saddle point system
|
||||
// k*u + grad p = f
|
||||
@@ -60,6 +66,7 @@ int main(int argc, char *argv[])
|
||||
int order = 1;
|
||||
bool par_format = false;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
bool adios2 = false;
|
||||
|
||||
@@ -75,6 +82,8 @@ int main(int argc, char *argv[])
|
||||
"Format to use when saving the results for VisIt.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
@@ -96,13 +105,18 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
// 3. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
if (myid == 0) { device.Print(); }
|
||||
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 4. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement. We choose
|
||||
// 'ref_levels' to be the largest number that gives a final mesh with no
|
||||
// more than 10,000 elements, unless the user specifies it as input.
|
||||
@@ -118,7 +132,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
|
||||
@@ -131,7 +145,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// 7. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// use the Raviart-Thomas finite elements of the specified order.
|
||||
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
|
||||
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
|
||||
@@ -151,7 +165,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "***********************************************************\n";
|
||||
}
|
||||
|
||||
// 7. Define the two BlockStructure of the problem. block_offsets is used
|
||||
// 8. Define the two BlockStructure of the problem. block_offsets is used
|
||||
// for Vector based on dof (like ParGridFunction or ParLinearForm),
|
||||
// block_trueOffstes is used for Vector based on trueDof (HypreParVector
|
||||
// for the rhs and solution of the linear system). The offsets computed
|
||||
@@ -168,7 +182,7 @@ int main(int argc, char *argv[])
|
||||
block_trueOffsets[2] = W_space->TrueVSize();
|
||||
block_trueOffsets.PartialSum();
|
||||
|
||||
// 8. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
// 9. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
ConstantCoefficient k(1.0);
|
||||
|
||||
VectorFunctionCoefficient fcoeff(dim, fFun);
|
||||
@@ -178,25 +192,30 @@ int main(int argc, char *argv[])
|
||||
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
|
||||
FunctionCoefficient pcoeff(pFun_ex);
|
||||
|
||||
// 9. Define the parallel grid function and parallel linear forms, solution
|
||||
// vector and rhs.
|
||||
BlockVector x(block_offsets), rhs(block_offsets);
|
||||
BlockVector trueX(block_trueOffsets), trueRhs(block_trueOffsets);
|
||||
// 10. Define the parallel grid function and parallel linear forms, solution
|
||||
// vector and rhs.
|
||||
MemoryType mt = device.GetMemoryType();
|
||||
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
|
||||
BlockVector trueX(block_trueOffsets, mt), trueRhs(block_trueOffsets, mt);
|
||||
|
||||
ParLinearForm *fform(new ParLinearForm);
|
||||
fform->Update(R_space, rhs.GetBlock(0), 0);
|
||||
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
|
||||
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
|
||||
fform->Assemble();
|
||||
fform->SyncAliasMemory(rhs);
|
||||
fform->ParallelAssemble(trueRhs.GetBlock(0));
|
||||
trueRhs.GetBlock(0).SyncAliasMemory(trueRhs);
|
||||
|
||||
ParLinearForm *gform(new ParLinearForm);
|
||||
gform->Update(W_space, rhs.GetBlock(1), 0);
|
||||
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
|
||||
gform->Assemble();
|
||||
gform->SyncAliasMemory(rhs);
|
||||
gform->ParallelAssemble(trueRhs.GetBlock(1));
|
||||
trueRhs.GetBlock(1).SyncAliasMemory(trueRhs);
|
||||
|
||||
// 10. Assemble the finite element matrices for the Darcy operator
|
||||
// 11. Assemble the finite element matrices for the Darcy operator
|
||||
//
|
||||
// D = [ M B^T ]
|
||||
// [ B 0 ]
|
||||
@@ -249,7 +268,7 @@ int main(int argc, char *argv[])
|
||||
darcyOp->SetBlock(1,0, B);
|
||||
}
|
||||
|
||||
// 11. Construct the operators for preconditioner
|
||||
// 12. Construct the operators for preconditioner
|
||||
//
|
||||
// P = [ diag(M) 0 ]
|
||||
// [ 0 B diag(M)^-1 B^T ]
|
||||
@@ -266,10 +285,11 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
Md_PA.SetSize(R_space->GetTrueVSize());
|
||||
mVarf->AssembleDiagonal(Md_PA);
|
||||
auto Md_host = Md_PA.HostRead();
|
||||
Vector invMd(Md_PA.Size());
|
||||
for (int i=0; i<Md_PA.Size(); ++i)
|
||||
{
|
||||
invMd(i) = 1.0 / Md_PA(i);
|
||||
invMd(i) = 1.0 / Md_host[i];
|
||||
}
|
||||
|
||||
Vector BMBt_diag(W_space->GetTrueVSize());
|
||||
@@ -302,7 +322,7 @@ int main(int argc, char *argv[])
|
||||
darcyPr->SetDiagonalBlock(0, invM);
|
||||
darcyPr->SetDiagonalBlock(1, invS);
|
||||
|
||||
// 12. Solve the linear system with MINRES.
|
||||
// 13. Solve the linear system with MINRES.
|
||||
// Check the norm of the unpreconditioned residual.
|
||||
int maxIter(pa ? 1000 : 500);
|
||||
double rtol(1.e-6);
|
||||
@@ -319,6 +339,7 @@ int main(int argc, char *argv[])
|
||||
solver.SetPrintLevel(verbose);
|
||||
trueX = 0.0;
|
||||
solver.Mult(trueRhs, trueX);
|
||||
if (device.IsEnabled()) { trueX.HostRead(); }
|
||||
chrono.Stop();
|
||||
|
||||
if (verbose)
|
||||
@@ -332,7 +353,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
|
||||
}
|
||||
|
||||
// 13. Extract the parallel grid function corresponding to the finite element
|
||||
// 14. Extract the parallel grid function corresponding to the finite element
|
||||
// approximation X. This is the local solution on each processor. Compute
|
||||
// L2 error norms.
|
||||
ParGridFunction *u(new ParGridFunction);
|
||||
@@ -360,7 +381,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
|
||||
}
|
||||
|
||||
// 14. Save the refined mesh and the solution in parallel. This output can be
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can be
|
||||
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol_*".
|
||||
{
|
||||
ostringstream mesh_name, u_name, p_name;
|
||||
@@ -381,7 +402,7 @@ int main(int argc, char *argv[])
|
||||
p->Save(p_ofs);
|
||||
}
|
||||
|
||||
// 15. Save data in the VisIt format
|
||||
// 16. Save data in the VisIt format
|
||||
VisItDataCollection visit_dc("Example5-Parallel", pmesh);
|
||||
visit_dc.RegisterField("velocity", u);
|
||||
visit_dc.RegisterField("pressure", p);
|
||||
@@ -390,7 +411,7 @@ int main(int argc, char *argv[])
|
||||
DataCollection::PARALLEL_FORMAT);
|
||||
visit_dc.Save();
|
||||
|
||||
// 16. Save data in the ParaView format
|
||||
// 17. Save data in the ParaView format
|
||||
ParaViewDataCollection paraview_dc("Example5P", pmesh);
|
||||
paraview_dc.SetPrefixPath("ParaView");
|
||||
paraview_dc.SetLevelsOfDetail(order);
|
||||
@@ -402,7 +423,7 @@ int main(int argc, char *argv[])
|
||||
paraview_dc.RegisterField("pressure",p);
|
||||
paraview_dc.Save();
|
||||
|
||||
// 17. Optionally output a BP (binary pack) file using ADIOS2. This can be
|
||||
// 18. Optionally output a BP (binary pack) file using ADIOS2. This can be
|
||||
// visualized with the ParaView VTX reader.
|
||||
#ifdef MFEM_USE_ADIOS2
|
||||
if (adios2)
|
||||
@@ -422,7 +443,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
#endif
|
||||
|
||||
// 18. Send the solution by socket to a GLVis server.
|
||||
// 19. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
@@ -442,7 +463,7 @@ int main(int argc, char *argv[])
|
||||
<< endl;
|
||||
}
|
||||
|
||||
// 19. Free the used memory.
|
||||
// 20. Free the used memory.
|
||||
delete fform;
|
||||
delete gform;
|
||||
delete u;
|
||||
|
||||
+19
-17
@@ -20,8 +20,11 @@
|
||||
// Device sample runs:
|
||||
// ex9 -pa
|
||||
// ex9 -ea
|
||||
// ex9 -fa
|
||||
// ex9 -pa -m ../data/periodic-cube.mesh
|
||||
// ex9 -pa -m ../data/periodic-cube.mesh -d cuda
|
||||
// ex9 -ea -m ../data/periodic-cube.mesh -d cuda
|
||||
// ex9 -fa -m ../data/periodic-cube.mesh -d cuda
|
||||
//
|
||||
// Description: This example code solves the time-dependent advection equation
|
||||
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
|
||||
@@ -144,6 +147,7 @@ int main(int argc, char *argv[])
|
||||
int order = 3;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
bool fa = false;
|
||||
const char *device_config = "cpu";
|
||||
int ode_solver_type = 4;
|
||||
double t_final = 10.0;
|
||||
@@ -170,6 +174,8 @@ int main(int argc, char *argv[])
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
|
||||
"--no-full-assembly", "Enable Full Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
|
||||
@@ -254,7 +260,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
// 5. Define the discontinuous DG finite element space of the given
|
||||
// polynomial order on the refined mesh.
|
||||
DG_FECollection fec(order, dim, BasisType::Positive);
|
||||
DG_FECollection fec(order, dim, BasisType::GaussLobatto);
|
||||
FiniteElementSpace fes(&mesh, &fec);
|
||||
|
||||
cout << "Number of unknowns: " << fes.GetVSize() << endl;
|
||||
@@ -278,6 +284,11 @@ int main(int argc, char *argv[])
|
||||
m.SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
k.SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
}
|
||||
else if (fa)
|
||||
{
|
||||
m.SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
k.SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
}
|
||||
m.AddDomainIntegrator(new MassIntegrator);
|
||||
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k.AddInteriorFaceIntegrator(
|
||||
@@ -378,10 +389,6 @@ int main(int argc, char *argv[])
|
||||
// iterations, ti, with a time-step dt).
|
||||
FE_Evolution adv(m, k, b);
|
||||
|
||||
Vector masses(u.Size());
|
||||
m.SpMat().Mult(u, masses);
|
||||
double mass = masses.Sum();
|
||||
|
||||
double t = 0.0;
|
||||
adv.SetTime(t);
|
||||
ode_solver->Init(adv);
|
||||
@@ -428,9 +435,6 @@ int main(int argc, char *argv[])
|
||||
u.Save(osol);
|
||||
}
|
||||
|
||||
m.SpMat().Mult(u, masses);
|
||||
cout << "Mass difference:" << abs(mass - masses.Sum()) << endl;
|
||||
|
||||
// 10. Free the used memory.
|
||||
delete ode_solver;
|
||||
delete pd;
|
||||
@@ -444,21 +448,19 @@ int main(int argc, char *argv[])
|
||||
FE_Evolution::FE_Evolution(BilinearForm &_M, BilinearForm &_K, const Vector &_b)
|
||||
: TimeDependentOperator(_M.Height()), M(_M), K(_K), b(_b), z(_M.Height())
|
||||
{
|
||||
bool pa = M.GetAssemblyLevel() == AssemblyLevel::PARTIAL;
|
||||
bool ea = M.GetAssemblyLevel() == AssemblyLevel::ELEMENT;
|
||||
Array<int> ess_tdof_list;
|
||||
if (pa || ea)
|
||||
if (M.GetAssemblyLevel() == AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
M_prec = new DSmoother(M.SpMat());
|
||||
M_solver.SetOperator(M.SpMat());
|
||||
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
|
||||
}
|
||||
else
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(M, ess_tdof_list);
|
||||
M_solver.SetOperator(M);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
else
|
||||
{
|
||||
M_prec = new DSmoother(M.SpMat());
|
||||
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
|
||||
M_solver.SetOperator(M.SpMat());
|
||||
}
|
||||
M_solver.SetPreconditioner(*M_prec);
|
||||
M_solver.iterative_mode = false;
|
||||
M_solver.SetRelTol(1e-9);
|
||||
|
||||
+24
-15
@@ -21,8 +21,11 @@
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex9p -pa
|
||||
// mpirun -np 4 ex9p -ea
|
||||
// mpirun -np 4 ex9p -fa
|
||||
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh
|
||||
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh -d cuda
|
||||
// mpirun -np 4 ex9p -ea -m ../data/periodic-cube.mesh -d cuda
|
||||
// mpirun -np 4 ex9p -fa -m ../data/periodic-cube.mesh -d cuda
|
||||
//
|
||||
// Description: This example code solves the time-dependent advection equation
|
||||
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
|
||||
@@ -164,6 +167,7 @@ int main(int argc, char *argv[])
|
||||
int order = 3;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
bool fa = false;
|
||||
const char *device_config = "cpu";
|
||||
int ode_solver_type = 4;
|
||||
double t_final = 10.0;
|
||||
@@ -193,6 +197,8 @@ int main(int argc, char *argv[])
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
|
||||
"--no-full-assembly", "Enable Full Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
|
||||
@@ -329,6 +335,12 @@ int main(int argc, char *argv[])
|
||||
m->SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
k->SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
}
|
||||
else if (fa)
|
||||
{
|
||||
m->SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
k->SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
}
|
||||
|
||||
m->AddDomainIntegrator(new MassIntegrator);
|
||||
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k->AddInteriorFaceIntegrator(
|
||||
@@ -565,29 +577,21 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
|
||||
M_solver(_M.ParFESpace()->GetComm()),
|
||||
z(_M.Height())
|
||||
{
|
||||
bool pa = _M.GetAssemblyLevel()==AssemblyLevel::PARTIAL;
|
||||
bool ea = _M.GetAssemblyLevel()==AssemblyLevel::ELEMENT;
|
||||
|
||||
if (pa || ea)
|
||||
{
|
||||
M.Reset(&_M, false);
|
||||
K.Reset(&_K, false);
|
||||
}
|
||||
else
|
||||
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
M.Reset(_M.ParallelAssemble(), true);
|
||||
K.Reset(_K.ParallelAssemble(), true);
|
||||
}
|
||||
else
|
||||
{
|
||||
M.Reset(&_M, false);
|
||||
K.Reset(&_K, false);
|
||||
}
|
||||
|
||||
M_solver.SetOperator(*M);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
if (pa || ea)
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
else
|
||||
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
HypreParMatrix &M_mat = *M.As<HypreParMatrix>();
|
||||
HypreParMatrix &K_mat = *K.As<HypreParMatrix>();
|
||||
@@ -596,6 +600,11 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
|
||||
|
||||
dg_solver = new DG_Solver(M_mat, K_mat, *_M.FESpace());
|
||||
}
|
||||
else
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
|
||||
M_solver.SetPreconditioner(*M_prec);
|
||||
M_solver.iterative_mode = false;
|
||||
|
||||
@@ -50,10 +50,43 @@ set(SRCS
|
||||
fespacehierarchy.cpp
|
||||
nonlininteg_vectorconvection.cpp
|
||||
quadinterpolator.cpp
|
||||
quadinterpolator_det.cpp
|
||||
quadinterpolator_eval_by_nodes.cpp
|
||||
quadinterpolator_eval_by_vdim.cpp
|
||||
quadinterpolator_grad_by_nodes.cpp
|
||||
quadinterpolator_grad_by_vdim.cpp
|
||||
quadinterpolator_grad_phys_by_nodes.cpp
|
||||
quadinterpolator_grad_phys_by_vdim.cpp
|
||||
quadinterpolator_face.cpp
|
||||
restriction.cpp
|
||||
staticcond.cpp
|
||||
tmop.cpp
|
||||
tmop_pa.cpp
|
||||
tmop_pa_h2d.cpp
|
||||
tmop_pa_h2d_c0.cpp
|
||||
tmop_pa_h2m.cpp
|
||||
tmop_pa_h2m_c0.cpp
|
||||
tmop_pa_h2s.cpp
|
||||
tmop_pa_h2s_c0.cpp
|
||||
tmop_pa_h3d.cpp
|
||||
tmop_pa_h3d_c0.cpp
|
||||
tmop_pa_h3m.cpp
|
||||
tmop_pa_h3m_c0.cpp
|
||||
tmop_pa_h3s.cpp
|
||||
tmop_pa_h3s_c0.cpp
|
||||
tmop_pa_jp2.cpp
|
||||
tmop_pa_jp3.cpp
|
||||
tmop_pa_jt2_tc.cpp
|
||||
tmop_pa_jt3_datc.cpp
|
||||
tmop_pa_jt3_tc.cpp
|
||||
tmop_pa_p2.cpp
|
||||
tmop_pa_p2_c0.cpp
|
||||
tmop_pa_p3.cpp
|
||||
tmop_pa_p3_c0.cpp
|
||||
tmop_pa_w2.cpp
|
||||
tmop_pa_w2_c0.cpp
|
||||
tmop_pa_w3.cpp
|
||||
tmop_pa_w3_c0.cpp
|
||||
tmop_tools.cpp
|
||||
gslib.cpp
|
||||
transfer.cpp
|
||||
@@ -83,7 +116,10 @@ set(HDRS
|
||||
nonlinearform_ext.hpp
|
||||
nonlininteg.hpp
|
||||
quadinterpolator.hpp
|
||||
quadinterpolator_eval.hpp
|
||||
quadinterpolator_face.hpp
|
||||
quadinterpolator_grad.hpp
|
||||
quadinterpolator_grad_phys.hpp
|
||||
restriction.hpp
|
||||
fespacehierarchy.hpp
|
||||
staticcond.hpp
|
||||
@@ -96,6 +132,7 @@ set(HDRS
|
||||
tfespace.hpp
|
||||
tintrules.hpp
|
||||
tmop.hpp
|
||||
tmop_pa.hpp
|
||||
tmop_tools.hpp
|
||||
gslib.hpp
|
||||
transfer.hpp
|
||||
|
||||
+16
-14
@@ -76,7 +76,7 @@ BilinearForm::BilinearForm(FiniteElementSpace * f)
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::FULL;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
@@ -94,7 +94,7 @@ BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
|
||||
precompute_sparsity = ps;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::FULL;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
|
||||
@@ -121,9 +121,10 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
assembly = assembly_level;
|
||||
switch (assembly)
|
||||
{
|
||||
case AssemblyLevel::LEGACYFULL:
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
// ext = new FABilinearFormExtension(this);
|
||||
// Use the original BilinearForm implementation for now
|
||||
ext = new FABilinearFormExtension(this);
|
||||
break;
|
||||
case AssemblyLevel::ELEMENT:
|
||||
ext = new EABilinearFormExtension(this);
|
||||
@@ -143,7 +144,7 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
void BilinearForm::EnableStaticCondensation()
|
||||
{
|
||||
delete static_cond;
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
static_cond = NULL;
|
||||
MFEM_WARNING("Static condensation not supported for this assembly level");
|
||||
@@ -168,7 +169,7 @@ void BilinearForm::EnableHybridization(FiniteElementSpace *constr_space,
|
||||
const Array<int> &ess_tdof_list)
|
||||
{
|
||||
delete hybridization;
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
delete constr_integ;
|
||||
hybridization = NULL;
|
||||
@@ -223,7 +224,7 @@ MatrixInverse * BilinearForm::Inverse() const
|
||||
|
||||
void BilinearForm::Finalize (int skip_zeros)
|
||||
{
|
||||
if (assembly == AssemblyLevel::FULL)
|
||||
if (assembly == AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
if (!static_cond) { mat->Finalize(skip_zeros); }
|
||||
if (mat_e) { mat_e->Finalize(skip_zeros); }
|
||||
@@ -639,8 +640,7 @@ void BilinearForm::AssembleDiagonal(Vector &diag) const
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
|
||||
"matrix and use SparseMatrix::GetDiag?");
|
||||
mat->GetDiag(diag);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1083,7 +1083,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
|
||||
mat = NULL;
|
||||
mat_e = NULL;
|
||||
extern_bfs = 0;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
@@ -1108,7 +1108,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
|
||||
bbfi_marker = mbf->bbfi_marker;
|
||||
btfbfi_marker = mbf->btfbfi_marker;
|
||||
|
||||
assembly = AssemblyLevel::FULL;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
@@ -1121,6 +1121,8 @@ void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
assembly = assembly_level;
|
||||
switch (assembly)
|
||||
{
|
||||
case AssemblyLevel::LEGACYFULL:
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
// ext = new FAMixedBilinearFormExtension(this);
|
||||
// Use the original BilinearForm implementation for now
|
||||
@@ -1191,7 +1193,7 @@ void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
|
||||
|
||||
MatrixInverse * MixedBilinearForm::Inverse() const
|
||||
{
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
|
||||
return NULL;
|
||||
@@ -1204,7 +1206,7 @@ MatrixInverse * MixedBilinearForm::Inverse() const
|
||||
|
||||
void MixedBilinearForm::Finalize (int skip_zeros)
|
||||
{
|
||||
if (assembly == AssemblyLevel::FULL)
|
||||
if (assembly == AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
mat -> Finalize (skip_zeros);
|
||||
}
|
||||
@@ -1481,7 +1483,7 @@ void MixedBilinearForm::AssembleDiagonal_ADAt(const Vector &D,
|
||||
|
||||
void MixedBilinearForm::ConformingAssemble()
|
||||
{
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
|
||||
return;
|
||||
|
||||
@@ -29,8 +29,11 @@ namespace mfem
|
||||
form classes derived from Operator. */
|
||||
enum class AssemblyLevel
|
||||
{
|
||||
/// Fully assembled form, i.e. a global sparse matrix in MFEM, Hypre or PETSC
|
||||
/// format.
|
||||
/// Legacy fully assembled form, i.e. a global sparse matrix in MFEM, Hypre
|
||||
/// or PETSC format. This assembly is ALWAYS performed on the host.
|
||||
LEGACYFULL = 0,
|
||||
/// Fully assembled form, i.e. a global sparse matrix in MFEM format. This
|
||||
/// assembly is compatible with device execution.
|
||||
FULL,
|
||||
/// Form assembled at element level, which computes and stores dense element
|
||||
/// matrices.
|
||||
@@ -119,7 +122,7 @@ protected:
|
||||
static_cond = NULL; hybridization = NULL;
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
+188
-36
@@ -15,6 +15,7 @@
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "libceed/ceed.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -115,7 +116,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
|
||||
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
|
||||
|
||||
const int iSz = integrators.Size();
|
||||
if (elem_restrict)
|
||||
if (elem_restrict && !DeviceCanUseCeed())
|
||||
{
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
@@ -292,7 +293,8 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
|
||||
// Data and methods for element-assembled bilinear forms
|
||||
EABilinearFormExtension::EABilinearFormExtension(BilinearForm *form)
|
||||
: PABilinearFormExtension(form)
|
||||
: PABilinearFormExtension(form),
|
||||
factorize_face_terms(form->FESpace()->IsDGSpace())
|
||||
{
|
||||
}
|
||||
|
||||
@@ -347,6 +349,17 @@ void EABilinearFormExtension::Assemble()
|
||||
{
|
||||
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr);
|
||||
}
|
||||
|
||||
if (factorize_face_terms && int_face_restrict_lex)
|
||||
{
|
||||
auto restFint = dynamic_cast<const L2FaceRestriction&>(*int_face_restrict_lex);
|
||||
restFint.AddFaceMatricesToElementMatrices(ea_data_int, ea_data);
|
||||
}
|
||||
if (factorize_face_terms && bdr_face_restrict_lex)
|
||||
{
|
||||
auto restFbdr = dynamic_cast<const L2FaceRestriction&>(*bdr_face_restrict_lex);
|
||||
restFbdr.AddFaceMatricesToElementMatrices(ea_data_bdr, ea_data);
|
||||
}
|
||||
}
|
||||
|
||||
void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
@@ -399,24 +412,27 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
|
||||
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
if (!factorize_face_terms)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
res += A_int(i, j, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
}
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
@@ -443,7 +459,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
// Treatment of boundary faces
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (bdr_face_restrict_lex && bFISz>0)
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, faceBdrX);
|
||||
@@ -522,24 +538,27 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
|
||||
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
if (!factorize_face_terms)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
res += A_int(j, i, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
}
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
@@ -566,7 +585,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
// Treatment of boundary faces
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (bdr_face_restrict_lex && bFISz>0)
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, faceBdrX);
|
||||
@@ -595,6 +614,139 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
// Data and methods for fully-assembled bilinear forms
|
||||
FABilinearFormExtension::FABilinearFormExtension(BilinearForm *form)
|
||||
: EABilinearFormExtension(form),
|
||||
mat(form->FESpace()->GetVSize(),form->FESpace()->GetVSize(),0),
|
||||
face_mat(form->FESpace()->GetVSize(),0,0),
|
||||
use_face_mat(false)
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if ( ParFiniteElementSpace* pfes =
|
||||
dynamic_cast<ParFiniteElementSpace*>(form->FESpace()) )
|
||||
{
|
||||
if (pfes->IsDGSpace())
|
||||
{
|
||||
use_face_mat = true;
|
||||
pfes->ExchangeFaceNbrData();
|
||||
face_mat.SetWidth(pfes->GetFaceNbrVSize());
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::Assemble()
|
||||
{
|
||||
EABilinearFormExtension::Assemble();
|
||||
FiniteElementSpace &fes = *a->FESpace();
|
||||
if (fes.IsDGSpace())
|
||||
{
|
||||
const L2ElementRestriction *restE =
|
||||
static_cast<const L2ElementRestriction*>(elem_restrict);
|
||||
const L2FaceRestriction *restF =
|
||||
static_cast<const L2FaceRestriction*>(int_face_restrict_lex);
|
||||
// 1. Fill I
|
||||
// 1.1 Increment with restE
|
||||
restE->FillI(mat);
|
||||
// 1.2 Increment with restF
|
||||
if (restF) { restF->FillI(mat, face_mat); }
|
||||
// 1.3 Sum the non-zeros in I
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
int cpt = 0;
|
||||
const int vd = fes.GetVDim();
|
||||
const int ndofs = ne*elemDofs*vd;
|
||||
for (int i = 0; i < ndofs; i++)
|
||||
{
|
||||
const int nnz = h_I[i];
|
||||
h_I[i] = cpt;
|
||||
cpt += nnz;
|
||||
}
|
||||
const int nnz = cpt;
|
||||
h_I[ndofs] = nnz;
|
||||
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
|
||||
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
|
||||
if (use_face_mat && restF)
|
||||
{
|
||||
auto h_I_face = face_mat.HostReadWriteI();
|
||||
int cpt = 0;
|
||||
for (int i = 0; i < ndofs; i++)
|
||||
{
|
||||
const int nnz = h_I_face[i];
|
||||
h_I_face[i] = cpt;
|
||||
cpt += nnz;
|
||||
}
|
||||
const int nnz_face = cpt;
|
||||
h_I_face[ndofs] = nnz_face;
|
||||
face_mat.GetMemoryJ().New(nnz_face,
|
||||
face_mat.GetMemoryJ().GetMemoryType());
|
||||
face_mat.GetMemoryData().New(nnz_face,
|
||||
face_mat.GetMemoryData().GetMemoryType());
|
||||
}
|
||||
// 2. Fill J and Data
|
||||
// 2.1 Fill J and Data with Elem ea_data
|
||||
restE->FillJAndData(ea_data, mat);
|
||||
// 2.2 Fill J and Data with Face ea_data_ext
|
||||
if (restF) { restF->FillJAndData(ea_data_ext, mat, face_mat); }
|
||||
// 2.3 Shift indirections in I back to original
|
||||
auto I = mat.HostReadWriteI();
|
||||
for (int i = ndofs; i > 0; i--)
|
||||
{
|
||||
I[i] = I[i-1];
|
||||
}
|
||||
I[0] = 0;
|
||||
if (use_face_mat && restF)
|
||||
{
|
||||
auto I_face = face_mat.HostReadWriteI();
|
||||
for (int i = ndofs; i > 0; i--)
|
||||
{
|
||||
I_face[i] = I_face[i-1];
|
||||
}
|
||||
I_face[0] = 0;
|
||||
}
|
||||
}
|
||||
else // continuous Galerkin case
|
||||
{
|
||||
const ElementRestriction &rest =
|
||||
static_cast<const ElementRestriction&>(*elem_restrict);
|
||||
rest.FillSparseMatrix(ea_data, mat);
|
||||
}
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
mat.Mult(x, y);
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (const ParFiniteElementSpace *pfes =
|
||||
dynamic_cast<const ParFiniteElementSpace*>(testFes))
|
||||
{
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
|
||||
const_cast<Vector&>(x),0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
Vector &shared_x = x_gf.FaceNbrData();
|
||||
if (shared_x.Size()) { face_mat.AddMult(shared_x, y); }
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
mat.MultTranspose(x, y);
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (const ParFiniteElementSpace *pfes =
|
||||
dynamic_cast<const ParFiniteElementSpace*>(testFes))
|
||||
{
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
|
||||
const_cast<Vector&>(x),0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
Vector &shared_x = x_gf.FaceNbrData();
|
||||
if (shared_x.Size()) { face_mat.AddMultTranspose(shared_x, y); }
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
|
||||
: Operator(form->Height(), form->Width()), a(form)
|
||||
{
|
||||
|
||||
+19
-21
@@ -62,27 +62,6 @@ public:
|
||||
virtual void Update() = 0;
|
||||
};
|
||||
|
||||
/** @brief Data and methods for fully-assembled bilinear forms.
|
||||
Not yet implemented! Use the BilinearForm Class instead. */
|
||||
class FABilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
public:
|
||||
FABilinearFormExtension(BilinearForm *form)
|
||||
: BilinearFormExtension(form) { }
|
||||
|
||||
/// TODO
|
||||
void Assemble() {}
|
||||
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A) {}
|
||||
void FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
Vector &x, Vector &b,
|
||||
OperatorHandle &A, Vector &X, Vector &B,
|
||||
int copy_interior = 0) {}
|
||||
void Mult(const Vector &x, Vector &y) const {}
|
||||
void MultTranspose(const Vector &x, Vector &y) const {}
|
||||
void Update() {}
|
||||
~FABilinearFormExtension() {}
|
||||
};
|
||||
|
||||
/// Data and methods for partially-assembled bilinear forms
|
||||
class PABilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
@@ -119,10 +98,12 @@ class EABilinearFormExtension : public PABilinearFormExtension
|
||||
protected:
|
||||
int ne;
|
||||
int elemDofs;
|
||||
// The element matrices are stored row major
|
||||
Vector ea_data;
|
||||
int nf_int, nf_bdr;
|
||||
int faceDofs;
|
||||
Vector ea_data_int, ea_data_ext, ea_data_bdr;
|
||||
bool factorize_face_terms;
|
||||
|
||||
public:
|
||||
EABilinearFormExtension(BilinearForm *form);
|
||||
@@ -132,6 +113,23 @@ public:
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
/// Data and methods for fully-assembled bilinear forms
|
||||
class FABilinearFormExtension : public EABilinearFormExtension
|
||||
{
|
||||
private:
|
||||
SparseMatrix mat;
|
||||
/// face_mat handles parallelism for DG face terms.
|
||||
SparseMatrix face_mat;
|
||||
bool use_face_mat;
|
||||
|
||||
public:
|
||||
FABilinearFormExtension(BilinearForm *form);
|
||||
|
||||
void Assemble();
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
/// Data and methods for matrix-free bilinear forms NOT YET IMPLEMENTED.
|
||||
class MFBilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
|
||||
+32
-2
@@ -1685,6 +1685,22 @@ protected:
|
||||
{
|
||||
trial_fe.CalcPhysCurlShape(Trans, shape);
|
||||
}
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes);
|
||||
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
private:
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
|
||||
const DofToQuad *mapsOtest; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsCtest; ///< Not owned. DOF-to-quad map, closed.
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, dofs1D, dofs1Dtest,quad1D, testType, trialType, coeffDim;
|
||||
};
|
||||
|
||||
/** Class for integrating the bilinear form a(u,v) := (Q u, curl v) in 3D and
|
||||
@@ -1724,6 +1740,20 @@ protected:
|
||||
{
|
||||
test_fe.CalcPhysCurlShape(Trans, shape);
|
||||
}
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes);
|
||||
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
private:
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, dofs1D, quad1D, testType, trialType, coeffDim;
|
||||
};
|
||||
|
||||
/** Class for integrating the bilinear form a(u,v) := - (Q u, grad v) in either
|
||||
@@ -1924,7 +1954,7 @@ public:
|
||||
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe);
|
||||
|
||||
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
|
||||
void SetupPA(const FiniteElementSpace &fes);
|
||||
};
|
||||
|
||||
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
|
||||
@@ -2000,7 +2030,7 @@ public:
|
||||
const FiniteElement &test_fe,
|
||||
ElementTransformation &Trans);
|
||||
|
||||
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
|
||||
void SetupPA(const FiniteElementSpace &fes);
|
||||
};
|
||||
|
||||
/** Mass integrator (u, v) restricted to the boundary of a domain */
|
||||
|
||||
@@ -32,7 +32,7 @@ static void EAConvectionAssemble1D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -54,7 +54,7 @@ static void EAConvectionAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Bj[k1] * D(k1, e) * r_Gi[k1];
|
||||
}
|
||||
A(i1, j1, e) = val;
|
||||
A(i1, j1, e) += val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -76,7 +76,7 @@ static void EAConvectionAssemble2D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -121,7 +121,7 @@ static void EAConvectionAssemble2D(const int NE,
|
||||
* r_B[k1][j1]* r_B[k2][j2];
|
||||
}
|
||||
}
|
||||
A(i1, i2, j1, j2, e) = val;
|
||||
A(i1, i2, j1, j2, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -145,7 +145,7 @@ static void EAConvectionAssemble3D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -191,7 +191,7 @@ static void EAConvectionAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
A(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
A(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,10 @@
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
#include "restriction.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
@@ -68,47 +72,53 @@ static void PAConvectionSetup3D(const int Q1D,
|
||||
const double alpha,
|
||||
Vector &op)
|
||||
{
|
||||
const int NQ = Q1D*Q1D*Q1D;
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const bool const_v = vel.Size() == 3;
|
||||
auto V =
|
||||
const_v ? Reshape(vel.Read(), 3,1,1) : Reshape(vel.Read(), 3,NQ,NE);
|
||||
auto y = Reshape(op.Write(), NQ, 3, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
const auto V = const_v ?
|
||||
Reshape(vel.Read(), 3,1,1,1,1) :
|
||||
Reshape(vel.Read(), 3,Q1D,Q1D,Q1D,NE);
|
||||
auto y = Reshape(op.Write(), Q1D,Q1D,Q1D,3,NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double J11 = J(q,0,0,e);
|
||||
const double J21 = J(q,1,0,e);
|
||||
const double J31 = J(q,2,0,e);
|
||||
const double J12 = J(q,0,1,e);
|
||||
const double J22 = J(q,1,1,e);
|
||||
const double J32 = J(q,2,1,e);
|
||||
const double J13 = J(q,0,2,e);
|
||||
const double J23 = J(q,1,2,e);
|
||||
const double J33 = J(q,2,2,e);
|
||||
const double w = alpha * W[q];
|
||||
const double v0 = const_v ? V(0,0,0) : V(0,q,e);
|
||||
const double v1 = const_v ? V(1,0,0) : V(1,q,e);
|
||||
const double v2 = const_v ? V(2,0,0) : V(2,q,e);
|
||||
const double wx = w * v0;
|
||||
const double wy = w * v1;
|
||||
const double wz = w * v2;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// q . J^{-1} = q . adj(J)
|
||||
y(q,0,e) = wx * A11 + wy * A12 + wz * A13;
|
||||
y(q,1,e) = wx * A21 + wy * A22 + wz * A23;
|
||||
y(q,2,e) = wx * A31 + wy * A32 + wz * A33;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double w = alpha * W(qx,qy,qz);
|
||||
const double v0 = const_v ? V(0,0,0,0,0) : V(0,qx,qy,qz,e);
|
||||
const double v1 = const_v ? V(1,0,0,0,0) : V(1,qx,qy,qz,e);
|
||||
const double v2 = const_v ? V(2,0,0,0,0) : V(2,qx,qy,qz,e);
|
||||
const double wx = w * v0;
|
||||
const double wy = w * v1;
|
||||
const double wz = w * v2;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// q . J^{-1} = q . adj(J)
|
||||
y(qx,qy,qz,0,e) = wx * A11 + wy * A12 + wz * A13;
|
||||
y(qx,qy,qz,1,e) = wx * A21 + wy * A22 + wz * A23;
|
||||
y(qx,qy,qz,2,e) = wx * A31 + wy * A32 + wz * A33;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -184,8 +194,8 @@ void PAConvectionApply2D(const int ne,
|
||||
Gu[dy][qx] = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dy][dx];
|
||||
Bu[dy][qx] += bx * x;
|
||||
Gu[dy][qx] += gx * x;
|
||||
@@ -202,8 +212,8 @@ void PAConvectionApply2D(const int ne,
|
||||
BGu[qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
GBu[qy][qx] += gx * Bu[dy][qx];
|
||||
BGu[qy][qx] += bx * Gu[dy][qx];
|
||||
}
|
||||
@@ -232,7 +242,7 @@ void PAConvectionApply2D(const int ne,
|
||||
BDGu[dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BDGu[dy][qx] += w * DGu[qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -244,7 +254,7 @@ void PAConvectionApply2D(const int ne,
|
||||
double BBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBDGu += w * BDGu[dy][qx];
|
||||
}
|
||||
y(dx,dy,e) += BBDGu;
|
||||
@@ -310,7 +320,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[tidz][dy][dx];
|
||||
const double x = u[tidz][dy][dx];
|
||||
Bu[tidz][dy][qx] += bx * x;
|
||||
Gu[tidz][dy][qx] += gx * x;
|
||||
}
|
||||
@@ -327,8 +337,8 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
BGu[tidz][qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
GBu[tidz][qy][qx] += gx * Bu[tidz][dy][qx];
|
||||
BGu[tidz][qy][qx] += bx * Gu[tidz][dy][qx];
|
||||
}
|
||||
@@ -359,7 +369,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
BDGu[tidz][dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BDGu[tidz][dy][qx] += w * DGu[tidz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -372,7 +382,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
double BBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBDGu += w * BDGu[tidz][dy][qx];
|
||||
}
|
||||
y(dx,dy,e) += BBDGu;
|
||||
@@ -436,8 +446,8 @@ void PAConvectionApply3D(const int ne,
|
||||
Gu[dz][dy][qx] = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dz][dy][dx];
|
||||
Bu[dz][dy][qx] += bx * x;
|
||||
Gu[dz][dy][qx] += gx * x;
|
||||
@@ -459,8 +469,8 @@ void PAConvectionApply3D(const int ne,
|
||||
BGu[dz][qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
BBu[dz][qy][qx] += bx * Bu[dz][dy][qx];
|
||||
GBu[dz][qy][qx] += gx * Bu[dz][dy][qx];
|
||||
BGu[dz][qy][qx] += bx * Gu[dz][dy][qx];
|
||||
@@ -482,8 +492,8 @@ void PAConvectionApply3D(const int ne,
|
||||
BBGu[qz][qy][qx] = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
GBBu[qz][qy][qx] += gx * BBu[dz][qy][qx];
|
||||
BGBu[qz][qy][qx] += bx * GBu[dz][qy][qx];
|
||||
BBGu[qz][qy][qx] += bx * BGu[dz][qy][qx];
|
||||
@@ -521,7 +531,7 @@ void PAConvectionApply3D(const int ne,
|
||||
BDGu[dz][qy][qx] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double w = Bt(dz,qz);
|
||||
const double w = Bt(dz,qz);
|
||||
BDGu[dz][qy][qx] += w * DGu[qz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -537,7 +547,7 @@ void PAConvectionApply3D(const int ne,
|
||||
BBDGu[dz][dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BBDGu[dz][dy][qx] += w * BDGu[dz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -552,7 +562,7 @@ void PAConvectionApply3D(const int ne,
|
||||
double BBBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBBDGu += w * BBDGu[dz][dy][qx];
|
||||
}
|
||||
y(dx,dy,dz,e) += BBBDGu;
|
||||
@@ -625,8 +635,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double Gu_ = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dz][dy][dx];
|
||||
Bu_ += bx * x;
|
||||
Gu_ += gx * x;
|
||||
@@ -651,8 +661,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BGu_ = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
BBu_ += bx * Bu[dz][dy][qx];
|
||||
GBu_ += gx * Bu[dz][dy][qx];
|
||||
BGu_ += bx * Gu[dz][dy][qx];
|
||||
@@ -678,8 +688,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBGu_ = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
GBBu_ += gx * BBu[dz][qy][qx];
|
||||
BGBu_ += bx * GBu[dz][qy][qx];
|
||||
BBGu_ += bx * BGu[dz][qy][qx];
|
||||
@@ -721,7 +731,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BDGu_ = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double w = Bt(dz,qz);
|
||||
const double w = Bt(dz,qz);
|
||||
BDGu_ += w * DGu[qz][qy][qx];
|
||||
}
|
||||
BDGu[dz][qy][qx] = BDGu_;
|
||||
@@ -739,7 +749,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBDGu_ = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BBDGu_ += w * BDGu[dz][qy][qx];
|
||||
}
|
||||
BBDGu[dz][dy][qx] = BBDGu_;
|
||||
@@ -756,7 +766,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBBDGu += w * BBDGu[dz][dy][qx];
|
||||
}
|
||||
y(dx,dy,dz,e) = BBBDGu;
|
||||
@@ -766,6 +776,117 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
|
||||
static void QEvalVGF2D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto X = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto C = Reshape(y_, VDIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
MFEM_SHARED double DD[NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
mfem::kernels::LoadX<MD1,NBZ>(e,D1D,c,X,DD);
|
||||
mfem::kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,DD,DQ);
|
||||
mfem::kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
double G;
|
||||
mfem::kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,G);
|
||||
C(c,qx,qy,e) = G;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
|
||||
static void QEvalVGF3D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto X = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto C = Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
MFEM_SHARED double DDD[MD1*MD1*MD1];
|
||||
MFEM_SHARED double DDQ[MD1*MD1*MQ1];
|
||||
MFEM_SHARED double DQQ[MD1*MQ1*MQ1];
|
||||
MFEM_SHARED double QQQ[MQ1*MQ1*MQ1];
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
mfem::kernels::LoadX<MD1>(e,D1D,c,X,DDD);
|
||||
mfem::kernels::EvalX<MD1,MQ1>(D1D,Q1D,B,DDD,DDQ);
|
||||
mfem::kernels::EvalY<MD1,MQ1>(D1D,Q1D,B,DDQ,DQQ);
|
||||
mfem::kernels::EvalZ<MD1,MQ1>(D1D,Q1D,B,DQQ,QQQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
double G;
|
||||
mfem::kernels::PullEval<MQ1>(qx,qy,qz,QQQ,G);
|
||||
C(c,qx,qy,qz,e) = G;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assumes tensor-product elements
|
||||
@@ -778,16 +899,104 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
const int nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode, temp_type);
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
|
||||
pa_data.SetSize(symmDims * nq * ne, temp_type);
|
||||
Vector vel;
|
||||
if (VectorConstantCoefficient *cQ = dynamic_cast<VectorConstantCoefficient*>(Q))
|
||||
if (VectorConstantCoefficient *cQ =
|
||||
dynamic_cast<VectorConstantCoefficient*>(Q))
|
||||
{
|
||||
vel = cQ->GetVec();
|
||||
}
|
||||
else if (VectorGridFunctionCoefficient *vgfQ =
|
||||
dynamic_cast<VectorGridFunctionCoefficient*>(Q))
|
||||
{
|
||||
Vector xe;
|
||||
vel.SetSize(dim * nq * ne, temp_type);
|
||||
|
||||
const GridFunction *gf = vgfQ->GetGridFunction();
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
const FiniteElementSpace &gf_fes = *gf->FESpace();
|
||||
|
||||
const int vdim = gf_fes.GetVDim();
|
||||
const Operator *R = gf_fes.GetElementRestriction(ordering);
|
||||
const FiniteElement &el_gf = *gf_fes.GetFE(0);
|
||||
const DofToQuad *maps_gf = &el_gf.GetDofToQuad(*ir, mode);
|
||||
const int D1D = maps_gf->ndof;
|
||||
const int Q1D = maps_gf->nqpt;
|
||||
|
||||
MFEM_VERIFY(R,"");
|
||||
MFEM_VERIFY(vdim == dim, "");
|
||||
MFEM_VERIFY(dim==2 || dim==3,"");
|
||||
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
xe.UseDevice(true);
|
||||
R->Mult(*gf, xe);
|
||||
|
||||
const auto B = maps_gf->B.Read();
|
||||
const auto x = xe.Read();
|
||||
auto y = vel.Write();
|
||||
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: QEvalVGF2D<2,2,2>(ne,B,x,y); break;
|
||||
case 0x33: QEvalVGF2D<2,3,3>(ne,B,x,y); break;
|
||||
case 0x34: QEvalVGF2D<2,3,4>(ne,B,x,y); break;
|
||||
default:
|
||||
{
|
||||
constexpr int MAX_DQ = 8;
|
||||
MFEM_VERIFY(D1D <= MAX_DQ, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_DQ, "");
|
||||
QEvalVGF2D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x23: QEvalVGF3D<3,2,3>(ne,B,x,y); break;
|
||||
case 0x34: QEvalVGF3D<3,3,4>(ne,B,x,y); break;
|
||||
case 0x35: QEvalVGF3D<3,3,5>(ne,B,x,y); break;
|
||||
case 0x46: QEvalVGF3D<3,4,6>(ne,B,x,y); break;
|
||||
case 0x48: QEvalVGF3D<3,4,8>(ne,B,x,y); break;
|
||||
default:
|
||||
{
|
||||
constexpr int MAX_DQ = 6;
|
||||
MFEM_VERIFY(D1D <= MAX_DQ, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_DQ, "");
|
||||
QEvalVGF3D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (VectorQuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<VectorQuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == dim * nq * ne,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
|
||||
qFun.Read();
|
||||
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
vel.SetSize(dim * nq * ne);
|
||||
@@ -827,9 +1036,12 @@ static void PAConvectionApply(const int dim,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPAConvectionApply2D<2,2,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x33: return SmemPAConvectionApply2D<3,3,3>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x44: return SmemPAConvectionApply2D<4,4,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x33: return SmemPAConvectionApply2D<3,3,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x34: return SmemPAConvectionApply2D<3,4,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x44: return SmemPAConvectionApply2D<4,4,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x46: return SmemPAConvectionApply2D<4,6,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x55: return SmemPAConvectionApply2D<5,5,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x58: return SmemPAConvectionApply2D<5,8,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x66: return SmemPAConvectionApply2D<6,6,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x77: return SmemPAConvectionApply2D<7,7,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x88: return SmemPAConvectionApply2D<8,8,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
@@ -842,8 +1054,12 @@ static void PAConvectionApply(const int dim,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPAConvectionApply3D<2,3>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x24: return SmemPAConvectionApply3D<2,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x26: return SmemPAConvectionApply3D<2,6>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x34: return SmemPAConvectionApply3D<3,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x35: return SmemPAConvectionApply3D<3,5>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x45: return SmemPAConvectionApply3D<4,5>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x48: return SmemPAConvectionApply3D<4,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x56: return SmemPAConvectionApply3D<5,6>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x67: return SmemPAConvectionApply3D<6,7>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x78: return SmemPAConvectionApply3D<7,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
|
||||
@@ -167,6 +167,19 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
|
||||
r.SetSize(1);
|
||||
r(0) = c_rho->constant;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* c_rho =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(rho))
|
||||
{
|
||||
const QuadratureFunction &qFun = c_rho->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == nq * nf,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
r.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
r.SetSize(nq * nf);
|
||||
@@ -200,6 +213,20 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
|
||||
{
|
||||
vel = c_u->GetVec();
|
||||
}
|
||||
else if (VectorQuadratureFunctionCoefficient* c_u =
|
||||
dynamic_cast<VectorQuadratureFunctionCoefficient*>(u))
|
||||
{
|
||||
// Assumed to be in lexicographical ordering
|
||||
const QuadratureFunction &qFun = c_u->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == dim * nq * nf,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
vel.SetSize(dim * nq * nf);
|
||||
|
||||
@@ -31,7 +31,7 @@ static void EADiffusionAssemble1D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -53,7 +53,7 @@ static void EADiffusionAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Gj[k1] * D(k1, e) * r_Gi[k1];
|
||||
}
|
||||
A(i1, j1, e) = val;
|
||||
A(i1, j1, e) += val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -75,7 +75,7 @@ static void EADiffusionAssemble2D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -120,7 +120,7 @@ static void EADiffusionAssemble2D(const int NE,
|
||||
+ gbi * D11 * gbj;
|
||||
}
|
||||
}
|
||||
A(i1, i2, j1, j2, e) = val;
|
||||
A(i1, i2, j1, j2, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -144,7 +144,7 @@ static void EADiffusionAssemble3D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 6, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -208,7 +208,7 @@ static void EADiffusionAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
A(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
A(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+314
-228
@@ -170,47 +170,53 @@ static void PADiffusionSetup3D(const int Q1D,
|
||||
const Vector &c,
|
||||
Vector &d)
|
||||
{
|
||||
const int NQ = Q1D*Q1D*Q1D;
|
||||
const bool const_c = c.Size() == 1;
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
|
||||
auto D = Reshape(d.Write(), NQ, 6, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
|
||||
Reshape(c.Read(), Q1D,Q1D,Q1D,NE);
|
||||
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, 6, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double J11 = J(q,0,0,e);
|
||||
const double J21 = J(q,1,0,e);
|
||||
const double J31 = J(q,2,0,e);
|
||||
const double J12 = J(q,0,1,e);
|
||||
const double J22 = J(q,1,1,e);
|
||||
const double J32 = J(q,2,1,e);
|
||||
const double J13 = J(q,0,2,e);
|
||||
const double J23 = J(q,1,2,e);
|
||||
const double J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0) : C(q,e);
|
||||
const double c_detJ = W[q] * coeff / detJ;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
D(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
D(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
D(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
D(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
D(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
D(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
|
||||
const double c_detJ = W(qx,qy,qz) * coeff / detJ;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
D(qx,qy,qz,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
D(qx,qy,qz,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
D(qx,qy,qz,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
D(qx,qy,qz,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
D(qx,qy,qz,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
D(qx,qy,qz,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -253,8 +259,7 @@ static void PADiffusionSetup(const int dim,
|
||||
}
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
const bool force)
|
||||
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
@@ -263,7 +268,7 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed() && !force)
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
if (ceedDataPtr) { delete ceedDataPtr; }
|
||||
CeedData* ptr = new CeedData();
|
||||
@@ -271,17 +276,16 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
InitCeedCoeff(Q, ptr);
|
||||
return CeedPADiffusionAssemble(fes, *ir, *ptr);
|
||||
}
|
||||
#else
|
||||
MFEM_CONTRACT_VAR(force);
|
||||
#endif
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
const int nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
|
||||
@@ -296,6 +300,19 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = cQ->constant;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == ne*nq,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
coeff.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
coeff.SetSize(nq * ne);
|
||||
@@ -723,6 +740,7 @@ static void PADiffusionAssembleDiagonal(const int dim,
|
||||
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,B,G,D,Y);
|
||||
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,B,G,D,Y);
|
||||
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,B,G,D,Y);
|
||||
case 0x46: return SmemPADiffusionDiagonal3D<4,6>(NE,B,G,D,Y);
|
||||
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,B,G,D,Y);
|
||||
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,B,G,D,Y);
|
||||
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,B,G,D,Y);
|
||||
@@ -736,9 +754,17 @@ static void PADiffusionAssembleDiagonal(const int dim,
|
||||
|
||||
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
|
||||
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G, pa_data, diag);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAssembleDiagonalPA(ceedDataPtr, diag);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G, pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1307,7 +1333,33 @@ static void PADiffusionApply3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Diffusion Apply 3D kernel
|
||||
// Half of B and G are stored in shared to get B, Bt, G and Gt.
|
||||
// Indices computation for SmemPADiffusionApply3D.
|
||||
static MFEM_HOST_DEVICE inline int qi(const int q, const int d, const int Q)
|
||||
{
|
||||
return (q<=d) ? q : Q-1-q;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int dj(const int q, const int d, const int D)
|
||||
{
|
||||
return (q<=d) ? d : D-1-d;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int qk(const int q, const int d, const int Q)
|
||||
{
|
||||
return (q<=d) ? Q-1-q : q;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int dl(const int q, const int d, const int D)
|
||||
{
|
||||
return (q<=d) ? D-1-d : d;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline double sign(const int q, const int d)
|
||||
{
|
||||
return (q<=d) ? -1.0 : 1.0;
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void SmemPADiffusionApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
@@ -1320,28 +1372,27 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= M1D, "");
|
||||
MFEM_VERIFY(Q1D <= M1Q, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, 6, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
|
||||
MFEM_SHARED double sBG[2][MQ1*MD1];
|
||||
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
|
||||
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
|
||||
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
|
||||
double (*Gt)[MQ1] = (double (*)[MQ1]) (sBG+1);
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
MFEM_SHARED double sBG[MQ1*MD1];
|
||||
double (*B)[MD1] = (double (*)[MD1]) sBG;
|
||||
double (*G)[MD1] = (double (*)[MD1]) sBG;
|
||||
double (*Bt)[MQ1] = (double (*)[MQ1]) sBG;
|
||||
double (*Gt)[MQ1] = (double (*)[MQ1]) sBG;
|
||||
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
|
||||
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
|
||||
@@ -1359,108 +1410,127 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
|
||||
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
|
||||
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
X[dz][dy][dx] = x(dx,dy,dz,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
G[q][d] = g(q,d);
|
||||
}
|
||||
const int i = qi(qx,dy,Q1D);
|
||||
const int j = dj(qx,dy,D1D);
|
||||
const int k = qk(qx,dy,Q1D);
|
||||
const int l = dl(qx,dy,D1D);
|
||||
B[i][j] = b(qx,dy);
|
||||
G[k][l] = g(qx,dy) * sign(qx,dy);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
double u[D1D], v[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double coords = X[dz][dy][dx];
|
||||
u += coords * B[qx][dx];
|
||||
v += coords * G[qx][dx];
|
||||
}
|
||||
DDQ0[dz][dy][qx] = u;
|
||||
DDQ1[dz][dy][qx] = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ1[dz][dy][qx] * B[qy][dy];
|
||||
v += DDQ0[dz][dy][qx] * G[qy][dy];
|
||||
w += DDQ0[dz][dy][qx] * B[qy][dy];
|
||||
}
|
||||
DQQ0[dz][qy][qx] = u;
|
||||
DQQ1[dz][qy][qx] = v;
|
||||
DQQ2[dz][qy][qx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
const int i = qi(qx,dx,Q1D);
|
||||
const int j = dj(qx,dx,D1D);
|
||||
const int k = qk(qx,dx,Q1D);
|
||||
const int l = dl(qx,dx,D1D);
|
||||
const double s = sign(qx,dx);
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ0[dz][qy][qx] * B[qz][dz];
|
||||
v += DQQ1[dz][qy][qx] * B[qz][dz];
|
||||
w += DQQ2[dz][qy][qx] * G[qz][dz];
|
||||
const double coords = X[dz][dy][dx];
|
||||
u[dz] += coords * B[i][j];
|
||||
v[dz] += coords * G[k][l] * s;
|
||||
}
|
||||
QQQ0[qz][qy][qx] = u;
|
||||
QQQ1[qz][qy][qx] = v;
|
||||
QQQ2[qz][qy][qx] = w;
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
DDQ0[dz][dy][qx] = u[dz];
|
||||
DDQ1[dz][dy][qx] = v[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
double u[D1D], v[D1D], w[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = w[dz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
|
||||
const double O11 = d(q,0,e);
|
||||
const double O12 = d(q,1,e);
|
||||
const double O13 = d(q,2,e);
|
||||
const double O22 = d(q,3,e);
|
||||
const double O23 = d(q,4,e);
|
||||
const double O33 = d(q,5,e);
|
||||
const double gX = QQQ0[qz][qy][qx];
|
||||
const double gY = QQQ1[qz][qy][qx];
|
||||
const double gZ = QQQ2[qz][qy][qx];
|
||||
const int i = qi(qy,dy,Q1D);
|
||||
const int j = dj(qy,dy,D1D);
|
||||
const int k = qk(qy,dy,Q1D);
|
||||
const int l = dl(qy,dy,D1D);
|
||||
const double s = sign(qy,dy);
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++)
|
||||
{
|
||||
u[dz] += DDQ1[dz][dy][qx] * B[i][j];
|
||||
v[dz] += DDQ0[dz][dy][qx] * G[k][l] * s;
|
||||
w[dz] += DDQ0[dz][dy][qx] * B[i][j];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++)
|
||||
{
|
||||
DQQ0[dz][qy][qx] = u[dz];
|
||||
DQQ1[dz][qy][qx] = v[dz];
|
||||
DQQ2[dz][qy][qx] = w[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
const int i = qi(qz,dz,Q1D);
|
||||
const int j = dj(qz,dz,D1D);
|
||||
const int k = qk(qz,dz,Q1D);
|
||||
const int l = dl(qz,dz,D1D);
|
||||
const double s = sign(qz,dz);
|
||||
u[qz] += DQQ0[dz][qy][qx] * B[i][j];
|
||||
v[qz] += DQQ1[dz][qy][qx] * B[i][j];
|
||||
w[qz] += DQQ2[dz][qy][qx] * G[k][l] * s;
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
const double O11 = d(qx,qy,qz,0,e);
|
||||
const double O12 = d(qx,qy,qz,1,e);
|
||||
const double O13 = d(qx,qy,qz,2,e);
|
||||
const double O22 = d(qx,qy,qz,3,e);
|
||||
const double O23 = d(qx,qy,qz,4,e);
|
||||
const double O33 = d(qx,qy,qz,5,e);
|
||||
const double gX = u[qz];
|
||||
const double gY = v[qz];
|
||||
const double gZ = w[qz];
|
||||
QQQ0[qz][qy][qx] = (O11*gX) + (O12*gY) + (O13*gZ);
|
||||
QQQ1[qz][qy][qx] = (O12*gX) + (O22*gY) + (O23*gZ);
|
||||
QQQ2[qz][qy][qx] = (O13*gX) + (O23*gY) + (O33*gZ);
|
||||
@@ -1468,78 +1538,112 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (tidz == 0)
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
Bt[d][q] = b(q,d);
|
||||
Gt[d][q] = g(q,d);
|
||||
}
|
||||
const int i = qi(q,d,Q1D);
|
||||
const int j = dj(q,d,D1D);
|
||||
const int k = qk(q,d,Q1D);
|
||||
const int l = dl(q,d,D1D);
|
||||
Bt[j][i] = b(q,d);
|
||||
Gt[l][k] = g(q,d) * sign(q,d);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += QQQ0[qz][qy][qx] * Gt[dx][qx];
|
||||
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
|
||||
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
|
||||
}
|
||||
QQD0[qz][qy][dx] = u;
|
||||
QQD1[qz][qy][dx] = v;
|
||||
QQD2[qz][qy][dx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += QQD0[qz][qy][dx] * Bt[dy][qy];
|
||||
v += QQD1[qz][qy][dx] * Gt[dy][qy];
|
||||
w += QQD2[qz][qy][dx] * Bt[dy][qy];
|
||||
}
|
||||
QDD0[qz][dy][dx] = u;
|
||||
QDD1[qz][dy][dx] = v;
|
||||
QDD2[qz][dy][dx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
const int i = qi(qx,dx,Q1D);
|
||||
const int j = dj(qx,dx,D1D);
|
||||
const int k = qk(qx,dx,Q1D);
|
||||
const int l = dl(qx,dx,D1D);
|
||||
const double s = sign(qx,dx);
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u += QDD0[qz][dy][dx] * Bt[dz][qz];
|
||||
v += QDD1[qz][dy][dx] * Bt[dz][qz];
|
||||
w += QDD2[qz][dy][dx] * Gt[dz][qz];
|
||||
u[qz] += QQQ0[qz][qy][qx] * Gt[l][k] * s;
|
||||
v[qz] += QQQ1[qz][qy][qx] * Bt[j][i];
|
||||
w[qz] += QQQ2[qz][qy][qx] * Bt[j][i];
|
||||
}
|
||||
y(dx,dy,dz,e) += (u + v + w);
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD0[qz][qy][dx] = u[qz];
|
||||
QQD1[qz][qy][dx] = v[qz];
|
||||
QQD2[qz][qy][dx] = w[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const int i = qi(qy,dy,Q1D);
|
||||
const int j = dj(qy,dy,D1D);
|
||||
const int k = qk(qy,dy,Q1D);
|
||||
const int l = dl(qy,dy,D1D);
|
||||
const double s = sign(qy,dy);
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u[qz] += QQD0[qz][qy][dx] * Bt[j][i];
|
||||
v[qz] += QQD1[qz][qy][dx] * Gt[l][k] * s;
|
||||
w[qz] += QQD2[qz][qy][dx] * Bt[j][i];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QDD0[qz][dy][dx] = u[qz];
|
||||
QDD1[qz][dy][dx] = v[qz];
|
||||
QDD2[qz][dy][dx] = w[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u[D1D], v[D1D], w[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz) { u[dz] = v[dz] = w[dz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const int i = qi(qz,dz,Q1D);
|
||||
const int j = dj(qz,dz,D1D);
|
||||
const int k = qk(qz,dz,Q1D);
|
||||
const int l = dl(qz,dz,D1D);
|
||||
const double s = sign(qz,dz);
|
||||
u[dz] += QDD0[qz][dy][dx] * Bt[j][i];
|
||||
v[dz] += QDD1[qz][dy][dx] * Bt[j][i];
|
||||
w[dz] += QDD2[qz][dy][dx] * Gt[l][k] * s;
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
y(dx,dy,dz,e) += (u[dz] + v[dz] + w[dz]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1574,9 +1678,11 @@ static void PADiffusionApply(const int dim,
|
||||
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
const int ID = (D1D << 4 ) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
switch (ID)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,D,X,Y);
|
||||
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,D,X,Y);
|
||||
@@ -1589,11 +1695,13 @@ static void PADiffusionApply(const int dim,
|
||||
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
switch (ID)
|
||||
{
|
||||
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,D,X,Y);
|
||||
case 0x24: return SmemPADiffusionApply3D<2,4>(NE,B,G,D,X,Y);
|
||||
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,D,X,Y);
|
||||
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,D,X,Y);
|
||||
case 0x46: return SmemPADiffusionApply3D<4,6>(NE,B,G,D,X,Y);
|
||||
@@ -1614,29 +1722,7 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorSyncArray(ceedDataPtr->v, mem);
|
||||
CeedAddMultPA(ceedDataPtr, x, y);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
|
||||
+1330
-30
File diff suppressed because it is too large
Load Diff
+11
-1
@@ -114,6 +114,8 @@ void PAHdivMassApply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
|
||||
@@ -238,6 +240,7 @@ void PAHdivMassAssembleDiagonal2D(const int D1D,
|
||||
Vector &_diag)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
|
||||
@@ -614,6 +617,8 @@ static void PADivDivApply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bot = Reshape(_Bot.Read(), D1D-1, Q1D);
|
||||
@@ -977,6 +982,7 @@ static void PADivDivAssembleDiagonal2D(const int D1D,
|
||||
Vector &_diag)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
|
||||
@@ -1400,6 +1406,8 @@ static void PAHdivL2Apply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
|
||||
@@ -1666,6 +1674,8 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto L2Bo = Reshape(_L2Bo.Read(), Q1D, L2D1D);
|
||||
auto Gct = Reshape(_Gct.Read(), D1D, Q1D);
|
||||
@@ -1724,7 +1734,7 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
|
||||
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
double aX[HDIV_MAX_D1D];
|
||||
double aX[MAX_D1D];
|
||||
|
||||
int osc = 0;
|
||||
for (int c = 0; c < VDIM; ++c) // loop over x, y components
|
||||
|
||||
@@ -30,7 +30,7 @@ static void EAMassAssemble1D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -52,7 +52,7 @@ static void EAMassAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Bi[k1] * r_Bj[k1] * D(k1, e);
|
||||
}
|
||||
M(i1, j1, e) = val;
|
||||
M(i1, j1, e) += val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -72,7 +72,7 @@ static void EAMassAssemble2D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -114,7 +114,7 @@ static void EAMassAssemble2D(const int NE,
|
||||
* s_D[k1][k2];
|
||||
}
|
||||
}
|
||||
M(i1, i2, j1, j2, e) = val;
|
||||
M(i1, i2, j1, j2, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -136,7 +136,7 @@ static void EAMassAssemble3D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -189,7 +189,7 @@ static void EAMassAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
M(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
M(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+101
-57
@@ -23,8 +23,9 @@ namespace mfem
|
||||
|
||||
// PA Mass Assemble kernel
|
||||
|
||||
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
{
|
||||
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
@@ -33,7 +34,7 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
ElementTransformation *T = mesh->GetElementTransformation(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed() && !force)
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
if (ceedDataPtr) { delete ceedDataPtr; }
|
||||
CeedData* ptr = new CeedData();
|
||||
@@ -45,27 +46,57 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetMesh()->GetNE();
|
||||
nq = ir->GetNPoints();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
|
||||
GeometricFactors::JACOBIANS);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const int flags = GeometricFactors::JACOBIANS |
|
||||
GeometricFactors::COORDINATES;
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
geom = mesh->GetGeometricFactors(*ir, flags, mode, temp_type);
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
Vector coeff;
|
||||
Vector *coeff{nullptr};
|
||||
bool own_coeff{true};
|
||||
if (Q == nullptr)
|
||||
{
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = 1.0;
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(1);
|
||||
(*coeff)(0) = 1.0;
|
||||
}
|
||||
else if (ConstantCoefficient* cQ = dynamic_cast<ConstantCoefficient*>(Q))
|
||||
{
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = cQ->constant;
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(1);
|
||||
(*coeff)(0) = cQ->constant;
|
||||
}
|
||||
else if (QuadratureCoefficient* cQ = dynamic_cast<QuadratureCoefficient*>(Q))
|
||||
{
|
||||
coeff = cQ->Data();
|
||||
own_coeff = false;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == nq * ne,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
coeff->MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
coeff.SetSize(nq * ne);
|
||||
auto C = Reshape(coeff.HostWrite(), nq, ne);
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(nq * ne);
|
||||
auto C = Reshape(coeff->HostWrite(), nq, ne);
|
||||
for (int e = 0; e < ne; ++e)
|
||||
{
|
||||
ElementTransformation& T = *fes.GetElementTransformation(e);
|
||||
@@ -80,11 +111,11 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
const bool const_c = coeff->Size() == 1;
|
||||
auto w = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
|
||||
auto C =
|
||||
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
|
||||
const_c ? Reshape(coeff->Read(), 1,1) : Reshape(coeff->Read(), NQ,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
@@ -103,28 +134,43 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
if (dim==3)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
auto W = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
|
||||
auto C =
|
||||
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ,NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
const int Q1D = quad1D;
|
||||
const bool const_c = coeff->Size() == 1;
|
||||
const auto W = Reshape(ir->GetWeights().Read(),Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(geom->J.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const auto C = const_c ?
|
||||
Reshape(coeff->Read(), 1,1,1,1) :
|
||||
Reshape(coeff->Read(), Q1D,Q1D,Q1D,NE);
|
||||
auto V = Reshape(pa_data.Write(), Q1D,Q1D,Q1D,NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
|
||||
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
|
||||
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0) : C(q,e);
|
||||
v(q,e) = W[q] * coeff * detJ;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
|
||||
V(qx,qy,qz,e) = W(qx,qy,qz) * coeff * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
if (own_coeff) { delete coeff; }
|
||||
}
|
||||
|
||||
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
@@ -426,8 +472,12 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
|
||||
case 0x24: return SmemPAMassAssembleDiagonal3D<2,4>(NE,B,D,Y);
|
||||
case 0x26: return SmemPAMassAssembleDiagonal3D<2,6>(NE,B,D,Y);
|
||||
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
|
||||
case 0x35: return SmemPAMassAssembleDiagonal3D<3,5>(NE,B,D,Y);
|
||||
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
|
||||
case 0x48: return SmemPAMassAssembleDiagonal3D<4,8>(NE,B,D,Y);
|
||||
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
|
||||
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
|
||||
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
|
||||
@@ -440,8 +490,16 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
|
||||
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
|
||||
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAssembleDiagonalPA(ceedDataPtr, diag);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -639,6 +697,7 @@ static void SmemPAMassApply2D(const int NE,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
@@ -902,6 +961,7 @@ static void SmemPAMassApply3D(const int NE,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
@@ -1151,10 +1211,13 @@ static void PAMassApply(const int dim,
|
||||
case 0x24: return SmemPAMassApply2D<2,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply2D<3,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply2D<3,5,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply2D<3,6,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply2D<4,6,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply2D<4,8,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x57: return SmemPAMassApply2D<5,7,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x58: return SmemPAMassApply2D<5,8,2>(NE,B,Bt,D,X,Y);
|
||||
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
|
||||
@@ -1162,6 +1225,7 @@ static void PAMassApply(const int dim,
|
||||
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
mfem::out << "Unknown 2D kernel 0x" << std::hex << id << std::endl;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
@@ -1170,7 +1234,9 @@ static void PAMassApply(const int dim,
|
||||
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
|
||||
case 0x24: return SmemPAMassApply3D<2,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply3D<3,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply3D<3,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x37: return SmemPAMassApply3D<3,7>(NE,B,Bt,D,X,Y);
|
||||
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply3D<4,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply3D<4,8>(NE,B,Bt,D,X,Y);
|
||||
@@ -1182,8 +1248,8 @@ static void PAMassApply(const int dim,
|
||||
case 0x9A: return SmemPAMassApply3D<9,10>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
mfem::out << "Unknown 3D kernel 0x" << std::hex << id << std::endl;
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
@@ -1192,29 +1258,7 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorSyncArray(ceedDataPtr->v, mem);
|
||||
CeedAddMultPA(ceedDataPtr, x, y);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
|
||||
@@ -25,7 +25,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
if (ne == 0) { return; }
|
||||
const int dofs = fes.GetFE(0)->GetDof();
|
||||
auto A = Reshape(ea_data_tmp.Write(), dofs, dofs, ne);
|
||||
auto AT = Reshape(ea_data.Write(), dofs, dofs, ne);
|
||||
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < dofs; i++)
|
||||
|
||||
@@ -9,12 +9,14 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void PAHcurlSetup2D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
@@ -22,6 +24,7 @@ void PAHcurlSetup2D(const int Q1D,
|
||||
Vector &op);
|
||||
|
||||
void PAHcurlSetup3D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
@@ -172,16 +175,36 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
|
||||
|
||||
Vector coeff(ne * nq);
|
||||
const int coeffDim = VQ ? VQ->GetVDim() : 1;
|
||||
|
||||
Vector coeff(coeffDim * ne * nq);
|
||||
coeff = 1.0;
|
||||
if (Q)
|
||||
auto coeffh = Reshape(coeff.HostWrite(), coeffDim, nq, ne);
|
||||
if (Q || VQ)
|
||||
{
|
||||
Vector D(VQ ? coeffDim : 0);
|
||||
if (VQ)
|
||||
{
|
||||
MFEM_VERIFY(coeffDim == dim, "");
|
||||
}
|
||||
|
||||
for (int e=0; e<ne; ++e)
|
||||
{
|
||||
ElementTransformation *tr = mesh->GetElementTransformation(e);
|
||||
for (int p=0; p<nq; ++p)
|
||||
{
|
||||
coeff[p + (e * nq)] = Q->Eval(*tr, ir->IntPoint(p));
|
||||
if (VQ)
|
||||
{
|
||||
VQ->Eval(D, *tr, ir->IntPoint(p));
|
||||
for (int i=0; i<coeffDim; ++i)
|
||||
{
|
||||
coeffh(i, p, e) = D[i];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
coeffh(0, p, e) = Q->Eval(*tr, ir->IntPoint(p));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -190,12 +213,12 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
|
||||
if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
|
||||
{
|
||||
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup3D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
|
||||
{
|
||||
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup2D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (el->GetDerivType() == mfem::FiniteElement::DIV && dim == 3)
|
||||
@@ -348,12 +371,12 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
|
||||
// Use the same setup functions as VectorFEMassIntegrator.
|
||||
if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
|
||||
{
|
||||
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup3D(quad1D, 1, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
|
||||
{
|
||||
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup2D(quad1D, 1, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
// Implementation of Coefficient class
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
@@ -21,6 +22,13 @@ namespace mfem
|
||||
|
||||
using namespace std;
|
||||
|
||||
double QuadratureCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
auto coeff = mfem::Reshape(qData->HostRead(), nip, NE);
|
||||
return coeff(ip.index, T.ElementNo);
|
||||
}
|
||||
|
||||
double PWConstCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
|
||||
+31
-1
@@ -30,7 +30,10 @@ class ParMesh;
|
||||
/** @brief Base class Coefficients that optionally depend on space and time.
|
||||
These are used by the BilinearFormIntegrator, LinearFormIntegrator, and
|
||||
NonlinearFormIntegrator classes to represent the physical coefficients in
|
||||
the PDEs that are being discretized. */
|
||||
the PDEs that are being discretized. This class can also be used in a more
|
||||
general way to represent functions that don't necessarily belong to a FE
|
||||
space, e.g., to project onto GridFunctions to use as initial conditions,
|
||||
exact solutions, etc. See, e.g., ex4 or ex22 for these uses. */
|
||||
class Coefficient
|
||||
{
|
||||
protected:
|
||||
@@ -84,6 +87,33 @@ public:
|
||||
{ return (constant); }
|
||||
};
|
||||
|
||||
|
||||
/// class for quadrature coefficient
|
||||
class QuadratureCoefficient : public Coefficient
|
||||
{
|
||||
|
||||
private:
|
||||
const int nip;
|
||||
const int NE;
|
||||
public:
|
||||
Vector *qData{nullptr};
|
||||
|
||||
//Set external data
|
||||
QuadratureCoefficient(Vector *Data, int in_nip, int in_NE)
|
||||
: qData(Data), nip(in_nip), NE(in_NE)
|
||||
{ }
|
||||
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
|
||||
Vector *Data()
|
||||
{
|
||||
return qData;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
/// class for piecewise constant coefficient
|
||||
/** @brief A piecewise constant coefficient with the constants keyed
|
||||
off the element attribute numbers. */
|
||||
class PWConstCoefficient : public Coefficient
|
||||
|
||||
+135
-53
@@ -342,11 +342,10 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
int ci)
|
||||
{
|
||||
FiniteElementSpace * fes = blfr->FESpace();
|
||||
|
||||
int vsize = fes->GetVSize();
|
||||
|
||||
// Allocate temporary vectors
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
|
||||
// Extract the real and imaginary parts of the input vectors
|
||||
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
|
||||
@@ -360,8 +359,7 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC) { b_i *= -1.0; }
|
||||
|
||||
int tvsize = fes->GetTrueVSize();
|
||||
SparseMatrix * A_r = nullptr;
|
||||
SparseMatrix * A_i = nullptr;
|
||||
OperatorHandle A_r, A_i;
|
||||
|
||||
X.SetSize(2 * tvsize);
|
||||
B.SetSize(2 * tvsize);
|
||||
@@ -374,42 +372,39 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
|
||||
if (RealInteg())
|
||||
{
|
||||
A_r = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
|
||||
b_0 = b_r;
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_r, X_0, B_0, ci);
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, A_r, X_0, B_0, ci);
|
||||
X_r = X_0; B_r = B_0;
|
||||
|
||||
b_0 = b_i;
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_r, X_0, B_0, ci);
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, A_r, X_0, B_0, ci);
|
||||
X_i = X_0; B_i = B_0;
|
||||
|
||||
if (ImagInteg())
|
||||
{
|
||||
A_i = new SparseMatrix;
|
||||
blfi->SetDiagonalPolicy(mfem::Matrix::DiagonalPolicy::DIAG_ZERO);
|
||||
|
||||
b_0 = 0.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, false);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, false);
|
||||
B_r -= B_0;
|
||||
|
||||
b_0 = 0.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, false);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, false);
|
||||
B_i += B_0;
|
||||
}
|
||||
}
|
||||
else if (ImagInteg())
|
||||
{
|
||||
A_i = new SparseMatrix;
|
||||
blfi->SetDiagonalPolicy(diag_policy);
|
||||
|
||||
b_0 = b_i;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, ci);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, ci);
|
||||
X_r = X_0; B_i = B_0;
|
||||
|
||||
b_0 = b_r; b_0 *= -1.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, ci);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, ci);
|
||||
X_i = X_0; B_r = B_0; B_r *= -1.0;
|
||||
}
|
||||
else
|
||||
@@ -417,16 +412,55 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify RHS and offdiagonal blocks (imaginary parts of the matrix) to
|
||||
// conform with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
int j = ess_tdof_list[k];
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
|
||||
{
|
||||
B_i *= -1.0;
|
||||
b_i *= -1.0;
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
ComplexSparseMatrix * A_sp;
|
||||
A_sp = new ComplexSparseMatrix(A_r, A_i, true, true, conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
|
||||
A_i.Type() == Operator::MFEM_SPARSEMAT )
|
||||
{
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
|
||||
A_i.As<SparseMatrix>(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
ComplexOperator * A_op =
|
||||
new ComplexOperator(A_r.Ptr(),
|
||||
A_i.Ptr(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -434,31 +468,60 @@ SesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
OperatorHandle &A)
|
||||
|
||||
{
|
||||
SparseMatrix * A_r = nullptr;
|
||||
SparseMatrix * A_i = nullptr;
|
||||
|
||||
OperatorHandle A_r, A_i;
|
||||
if (RealInteg())
|
||||
{
|
||||
A_r = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
blfr->FormSystemMatrix(ess_tdof_list, *A_r);
|
||||
blfr->FormSystemMatrix(ess_tdof_list, A_r);
|
||||
}
|
||||
if (ImagInteg())
|
||||
{
|
||||
A_i = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
blfi->FormSystemMatrix(ess_tdof_list, *A_i);
|
||||
blfi->SetDiagonalPolicy(RealInteg() ?
|
||||
mfem::Matrix::DiagonalPolicy::DIAG_ZERO :
|
||||
diag_policy);
|
||||
blfi->FormSystemMatrix(ess_tdof_list, A_i);
|
||||
}
|
||||
if (!RealInteg() && !ImagInteg())
|
||||
{
|
||||
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r, A_i, true, true, conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
|
||||
A_i.Type() == Operator::MFEM_SPARSEMAT )
|
||||
{
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
|
||||
A_i.As<SparseMatrix>(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
ComplexOperator * A_op =
|
||||
new ComplexOperator(A_r.Ptr(),
|
||||
A_i.Ptr(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -646,7 +709,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
|
||||
tdof_offsets = new HYPRE_Int[n+1];
|
||||
|
||||
for (int i=0; i<=n; i++)
|
||||
for (int i = 0; i <= n; i++)
|
||||
{
|
||||
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
|
||||
}
|
||||
@@ -654,7 +717,8 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
|
||||
|
||||
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
ParLinearForm *plf_r, ParLinearForm *plf_i,
|
||||
ParLinearForm *plf_r,
|
||||
ParLinearForm *plf_i,
|
||||
ComplexOperator::Convention
|
||||
convention)
|
||||
: Vector(2*(pfes->GetVSize())),
|
||||
@@ -670,7 +734,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
|
||||
tdof_offsets = new HYPRE_Int[n+1];
|
||||
|
||||
for (int i=0; i<=n; i++)
|
||||
for (int i = 0; i <= n; i++)
|
||||
{
|
||||
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
|
||||
}
|
||||
@@ -817,7 +881,8 @@ ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
{}
|
||||
|
||||
ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
ParBilinearForm *pbfr, ParBilinearForm *pbfi,
|
||||
ParBilinearForm *pbfr,
|
||||
ParBilinearForm *pbfi,
|
||||
ComplexOperator::Convention convention)
|
||||
: conv(convention),
|
||||
pblfr(new ParBilinearForm(pf,pbfr)),
|
||||
@@ -913,9 +978,10 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
int vsize = pfes->GetVSize();
|
||||
|
||||
// Allocate temporary vectors
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
|
||||
// Extract the real and imaginary parts of the input vectors
|
||||
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
|
||||
Vector x_r(x.GetData(), vsize);
|
||||
Vector x_i(&(x.GetData())[vsize], vsize);
|
||||
|
||||
@@ -974,25 +1040,34 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
// Modify RHS and offdiagonal blocks (Imaginary parts of the matrix) to
|
||||
// conform with standard essential BC treatment i.e. zero out rows and
|
||||
// columns and place ones on the diagonal.
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
// Modify RHS to conform with standard essential BC treatment
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
int j=ess_tdof_list[k];
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if ( A_i.Type() == Operator::Hypre_ParCSR )
|
||||
{
|
||||
HypreParMatrix * Ah; A_i.Get(Ah);
|
||||
int n = ess_tdof_list.Size();
|
||||
hypre_ParCSRMatrix * Aih =
|
||||
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
|
||||
for (int k=0; k<n; k++)
|
||||
HypreParMatrix * Ah;
|
||||
A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix *Aih = *Ah;
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
int j=ess_tdof_list[k];
|
||||
int j = ess_tdof_list[k];
|
||||
Aih->diag->data[Aih->diag->i[j]] = 0.0;
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
|
||||
@@ -1000,6 +1075,7 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
B_i *= -1.0;
|
||||
b_i *= -1.0;
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
if ( A_r.Type() == Operator::Hypre_ParCSR ||
|
||||
@@ -1023,6 +1099,8 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -1043,25 +1121,27 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
// Modify offdiagonal blocks (Imaginary parts of the matrix) to conform with
|
||||
// standard essential BC treatment i.e. zero out rows and columns and place
|
||||
// ones on the diagonal.
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if ( A_i.Type() == Operator::Hypre_ParCSR )
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
int j;
|
||||
|
||||
HypreParMatrix * Ah; A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix * Aih =
|
||||
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
|
||||
for (int k=0; k<n; k++)
|
||||
HypreParMatrix * Ah;
|
||||
A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix * Aih = *Ah;
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
j=ess_tdof_list[k];
|
||||
int j = ess_tdof_list[k];
|
||||
Aih->diag->data[Aih->diag->i[j]] = 0.0;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
@@ -1087,6 +1167,8 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
|
||||
+31
-1
@@ -219,6 +219,21 @@ public:
|
||||
void SetConvention(const ComplexOperator::Convention &
|
||||
convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::FULL (default)
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
blfr->SetAssemblyLevel(assembly_level);
|
||||
blfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
BilinearForm & real() { return *blfr; }
|
||||
BilinearForm & imag() { return *blfi; }
|
||||
const BilinearForm & real() const { return *blfr; }
|
||||
@@ -478,7 +493,7 @@ public:
|
||||
/** Class for a parallel sesquilinear form
|
||||
|
||||
A sesquilinear form is a generalization of a bilinear form to complex-valued
|
||||
fields. Sesquilinear forms are linear in the second argument but but the
|
||||
fields. Sesquilinear forms are linear in the second argument but the
|
||||
first argument involves a complex conjugate in the sense that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
@@ -524,6 +539,21 @@ public:
|
||||
void SetConvention(const ComplexOperator::Convention &
|
||||
convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::FULL (default)
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
pblfr->SetAssemblyLevel(assembly_level);
|
||||
pblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
ParBilinearForm & real() { return *pblfr; }
|
||||
ParBilinearForm & imag() { return *pblfi; }
|
||||
const ParBilinearForm & real() const { return *pblfr; }
|
||||
|
||||
+40
-8
@@ -415,9 +415,6 @@ void VisItDataCollection::SetMesh(MPI_Comm comm, Mesh *new_mesh)
|
||||
void VisItDataCollection::RegisterField(const std::string& name,
|
||||
GridFunction *gf)
|
||||
{
|
||||
DataCollection::RegisterField(name, gf);
|
||||
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim());
|
||||
|
||||
int LOD = 1;
|
||||
if (gf->FESpace()->GetNURBSext())
|
||||
{
|
||||
@@ -431,6 +428,27 @@ void VisItDataCollection::RegisterField(const std::string& name,
|
||||
}
|
||||
}
|
||||
|
||||
DataCollection::RegisterField(name, gf);
|
||||
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD);
|
||||
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
|
||||
}
|
||||
|
||||
void VisItDataCollection::RegisterQField(const std::string& name,
|
||||
QuadratureFunction *qf)
|
||||
{
|
||||
int LOD = -1;
|
||||
Mesh *mesh = qf->GetSpace()->GetMesh();
|
||||
for (int e=0; e<qf->GetSpace()->GetNE(); e++)
|
||||
{
|
||||
int locLOD = GlobGeometryRefiner.GetRefinementLevelFromElems(
|
||||
mesh->GetElementBaseGeometry(e),
|
||||
qf->GetElementIntRule(e).GetNPoints());
|
||||
|
||||
LOD = std::max(LOD,locLOD);
|
||||
}
|
||||
|
||||
DataCollection::RegisterQField(name, qf);
|
||||
field_info_map[name] = VisItFieldInfo("elements", 1, LOD);
|
||||
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
|
||||
}
|
||||
|
||||
@@ -598,14 +616,28 @@ void VisItDataCollection::LoadFields()
|
||||
// TODO: 1) load parallel GridFunction on one processor
|
||||
if (serial)
|
||||
{
|
||||
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
|
||||
if ((it->second).association == "nodes")
|
||||
{
|
||||
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
|
||||
}
|
||||
else if ((it->second).association == "elements")
|
||||
{
|
||||
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
field_map.Register(
|
||||
it->first,
|
||||
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
|
||||
if ((it->second).association == "nodes")
|
||||
{
|
||||
field_map.Register(
|
||||
it->first,
|
||||
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
|
||||
}
|
||||
else if ((it->second).association == "elements")
|
||||
{
|
||||
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
|
||||
}
|
||||
#else
|
||||
error = READ_ERROR;
|
||||
MFEM_WARNING("Reading parallel format in serial is not supported");
|
||||
@@ -640,7 +672,7 @@ std::string VisItDataCollection::GetVisItRootString()
|
||||
{
|
||||
ftags["assoc"] = picojson::value((it->second).association);
|
||||
ftags["comps"] = picojson::value(to_string((it->second).num_components));
|
||||
ftags["lod"] = picojson::value(to_string(visit_levels_of_detail));
|
||||
ftags["lod"] = picojson::value(to_string((it->second).lod));
|
||||
field["path"] = picojson::value(path_str + it->first + file_ext_format);
|
||||
field["tags"] = picojson::value(ftags);
|
||||
fields[it->first] = picojson::value(field);
|
||||
|
||||
+10
-3
@@ -391,9 +391,10 @@ class VisItFieldInfo
|
||||
public:
|
||||
std::string association;
|
||||
int num_components;
|
||||
VisItFieldInfo() { association = ""; num_components = 0; }
|
||||
VisItFieldInfo(std::string _association, int _num_components)
|
||||
{ association = _association; num_components = _num_components; }
|
||||
int lod;
|
||||
VisItFieldInfo() { association = ""; num_components = 0; lod = 1;}
|
||||
VisItFieldInfo(std::string _association, int _num_components, int _lod = 1)
|
||||
{ association = _association; num_components = _num_components; lod =_lod;}
|
||||
};
|
||||
|
||||
/// Data collection with VisIt I/O routines
|
||||
@@ -445,6 +446,12 @@ public:
|
||||
/// Add a grid function to the collection and update the root file
|
||||
virtual void RegisterField(const std::string& field_name, GridFunction *gf);
|
||||
|
||||
/// Add a quadrature function to the collection and update the root file.
|
||||
/** Visualization of quadrature function is not supported in VisIt(3.12).
|
||||
A patch has been sent to VisIt developers in June 2020. */
|
||||
virtual void RegisterQField(const std::string& q_field_name,
|
||||
QuadratureFunction *qf);
|
||||
|
||||
/// Set VisIt parameter: default levels of detail for the MultiresControl
|
||||
void SetLevelsOfDetail(int levels_of_detail);
|
||||
|
||||
|
||||
+4
-1
@@ -37,7 +37,8 @@ public:
|
||||
ClosedUniform = 4, ///< Nodes: x_i = i/(n-1), i=0,...,n-1
|
||||
OpenHalfUniform = 5, ///< Nodes: x_i = (i+1/2)/n, i=0,...,n-1
|
||||
Serendipity = 6, ///< Serendipity basis (squares / cubes)
|
||||
NumBasisTypes = 7 /**< Keep track of maximum types to prevent
|
||||
ClosedGL = 7, ///< Closed GaussLegendre
|
||||
NumBasisTypes = 8 /**< Keep track of maximum types to prevent
|
||||
hard-coding */
|
||||
};
|
||||
/** @brief If the input does not represents a valid BasisType, abort with an
|
||||
@@ -69,6 +70,7 @@ public:
|
||||
case ClosedUniform: return Quadrature1D::ClosedUniform;
|
||||
case OpenHalfUniform: return Quadrature1D::OpenHalfUniform;
|
||||
case Serendipity: return Quadrature1D::GaussLobatto;
|
||||
case ClosedGL: return Quadrature1D::ClosedGL;
|
||||
}
|
||||
return Quadrature1D::Invalid;
|
||||
}
|
||||
@@ -82,6 +84,7 @@ public:
|
||||
case Quadrature1D::OpenUniform: return OpenUniform;
|
||||
case Quadrature1D::ClosedUniform: return ClosedUniform;
|
||||
case Quadrature1D::OpenHalfUniform: return OpenHalfUniform;
|
||||
case Quadrature1D::ClosedGL: return ClosedGL;
|
||||
}
|
||||
return Invalid;
|
||||
}
|
||||
|
||||
+6
-6
@@ -944,7 +944,7 @@ const Operator *FiniteElementSpace::GetFaceRestriction(
|
||||
}
|
||||
|
||||
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
const IntegrationRule &ir) const
|
||||
const IntegrationRule &ir, const DofToQuad::Mode mode) const
|
||||
{
|
||||
for (int i = 0; i < E2Q_array.Size(); i++)
|
||||
{
|
||||
@@ -952,13 +952,13 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
if (qi->IntRule == &ir) { return qi; }
|
||||
}
|
||||
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir);
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir, mode);
|
||||
E2Q_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
|
||||
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
const QuadratureSpace &qs) const
|
||||
const QuadratureSpace &qs, const DofToQuad::Mode mode) const
|
||||
{
|
||||
for (int i = 0; i < E2Q_array.Size(); i++)
|
||||
{
|
||||
@@ -966,7 +966,7 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
if (qi->qspace == &qs) { return qi; }
|
||||
}
|
||||
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs);
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs, mode);
|
||||
E2Q_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
@@ -983,8 +983,8 @@ const FaceQuadratureInterpolator
|
||||
if (qi->IntRule == &ir) { return qi; }
|
||||
}
|
||||
|
||||
FaceQuadratureInterpolator *qi = new FaceQuadratureInterpolator(*this, ir,
|
||||
type);
|
||||
FaceQuadratureInterpolator *qi =
|
||||
new FaceQuadratureInterpolator(*this, ir, type);
|
||||
E2IFQ_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
|
||||
+8
-2
@@ -367,7 +367,7 @@ public:
|
||||
All elements will use the same IntegrationRule, @a ir as the target
|
||||
quadrature points. */
|
||||
const QuadratureInterpolator *GetQuadratureInterpolator(
|
||||
const IntegrationRule &ir) const;
|
||||
const IntegrationRule &ir, const DofToQuad::Mode = DofToQuad::FULL) const;
|
||||
|
||||
/** @brief Return a QuadratureInterpolator that interpolates E-vectors to
|
||||
quadrature point values and/or derivatives (Q-vectors). */
|
||||
@@ -378,7 +378,7 @@ public:
|
||||
The target quadrature points in the elements are described by the given
|
||||
QuadratureSpace, @a qs. */
|
||||
const QuadratureInterpolator *GetQuadratureInterpolator(
|
||||
const QuadratureSpace &qs) const;
|
||||
const QuadratureSpace &qs, const DofToQuad::Mode = DofToQuad::FULL) const;
|
||||
|
||||
/** @brief Return a FaceQuadratureInterpolator that interpolates E-vectors to
|
||||
quadrature point values and/or derivatives (Q-vectors). */
|
||||
@@ -756,6 +756,12 @@ public:
|
||||
/// Return the total number of quadrature points.
|
||||
int GetSize() const { return size; }
|
||||
|
||||
/// Returns the mesh
|
||||
inline Mesh *GetMesh() const { return mesh; }
|
||||
|
||||
/// Returns number of elements in the mesh.
|
||||
inline int GetNE() const { return mesh->GetNE(); }
|
||||
|
||||
/// Get the IntegrationRule associated with mesh element @a idx.
|
||||
const IntegrationRule &GetElementIntRule(int idx) const
|
||||
{ return *int_rule[mesh->GetElementBaseGeometry(idx)]; }
|
||||
|
||||
+141
-33
@@ -1344,15 +1344,14 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, Times-1);
|
||||
if (ir == NULL)
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule(Times-1);
|
||||
for (int i = 1; i < Times; i++)
|
||||
{
|
||||
ir = new IntegrationRule(Times-1);
|
||||
for (int i = 1; i < Times; i++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(i-1);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = ip.z = 0.0;
|
||||
}
|
||||
IntegrationPoint &ip = ir->IntPoint(i-1);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -1364,18 +1363,17 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, ((Times-1)*(Times-2))/2);
|
||||
if (ir == NULL)
|
||||
{
|
||||
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
|
||||
for (int k = 0, j = 1; j < Times-1; j++)
|
||||
for (int i = 1; i < Times-j; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
|
||||
for (int k = 0, j = 1; j < Times-1; j++)
|
||||
for (int i = 1; i < Times-j; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -1386,18 +1384,17 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, (Times-1)*(Times-1));
|
||||
if (ir == NULL)
|
||||
{
|
||||
ir = new IntegrationRule((Times-1)*(Times-1));
|
||||
for (int k = 0, j = 1; j < Times; j++)
|
||||
for (int i = 1; i < Times; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule((Times-1)*(Times-1));
|
||||
for (int k = 0, j = 1; j < Times; j++)
|
||||
for (int i = 1; i < Times; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -1405,10 +1402,121 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
mfem_error("GeometryRefiner::RefineInterior(...)");
|
||||
}
|
||||
|
||||
if (ir) { IntPts[Geom].Append(ir); }
|
||||
MFEM_ASSERT(ir != NULL, "Failed to construct the refined IntegrationRule.");
|
||||
IntPts[Geom].Append(ir);
|
||||
|
||||
return ir;
|
||||
}
|
||||
|
||||
|
||||
int GeometryRefiner::GetRefinementLevelFromPoints(Geometry::Type geom, int Npts)
|
||||
{
|
||||
switch (geom)
|
||||
{
|
||||
case Geometry::POINT:
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SEGMENT:
|
||||
{
|
||||
return Npts -1;
|
||||
}
|
||||
case Geometry::TRIANGLE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+2)/2;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SQUARE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1);
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::CUBE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1)*(n+1);
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::TETRAHEDRON:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+3)*(n+2)*(n+1)/6;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::PRISM:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1)*(n+2)/2;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
default:
|
||||
{
|
||||
mfem_error("Non existing Geometry.");
|
||||
}
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
int GeometryRefiner::GetRefinementLevelFromElems(Geometry::Type geom, int Nels)
|
||||
{
|
||||
switch (geom)
|
||||
{
|
||||
case Geometry::POINT:
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SEGMENT:
|
||||
{
|
||||
return Nels;
|
||||
}
|
||||
case Geometry::TRIANGLE:
|
||||
case Geometry::SQUARE:
|
||||
{
|
||||
for (int n = 0; (n < 15) && (n*n < Nels+1) ; n++)
|
||||
{
|
||||
if (n*n == Nels) { return n-1; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::CUBE:
|
||||
case Geometry::TETRAHEDRON:
|
||||
case Geometry::PRISM:
|
||||
{
|
||||
for (int n = 0; (n < 15) && (n*n*n < Nels+1) ; n++)
|
||||
{
|
||||
if (n*n*n == Nels) { return n-1; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
default:
|
||||
{
|
||||
mfem_error("Non existing Geometry.");
|
||||
}
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
GeometryRefiner GlobGeometryRefiner;
|
||||
|
||||
}
|
||||
|
||||
@@ -273,6 +273,12 @@ public:
|
||||
/// @note This method always uses Quadrature1D::OpenUniform points.
|
||||
const IntegrationRule *RefineInterior(Geometry::Type Geom, int Times);
|
||||
|
||||
/// Get the Refinement level based on number of points
|
||||
virtual int GetRefinementLevelFromPoints(Geometry::Type Geom, int Npts);
|
||||
|
||||
/// Get the Refinement level based on number of elements
|
||||
virtual int GetRefinementLevelFromElems(Geometry::Type geom, int Npts);
|
||||
|
||||
~GeometryRefiner();
|
||||
};
|
||||
|
||||
|
||||
+2
-2
@@ -2984,7 +2984,7 @@ double GridFunction::ComputeLpError(const double p, Coefficient &exsol,
|
||||
}
|
||||
else
|
||||
{
|
||||
int intorder = 2*fe->GetOrder() + 1; // <----------
|
||||
int intorder = 2*fe->GetOrder() + 3; // <----------
|
||||
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
|
||||
}
|
||||
GetValues(i, *ir, vals);
|
||||
@@ -3117,7 +3117,7 @@ double GridFunction::ComputeLpError(const double p, VectorCoefficient &exsol,
|
||||
}
|
||||
else
|
||||
{
|
||||
int intorder = 2*fe->GetOrder() + 1; // <----------
|
||||
int intorder = 2*fe->GetOrder() + 3; // <----------
|
||||
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
|
||||
}
|
||||
T = fes->GetElementTransformation(i);
|
||||
|
||||
+2
-2
@@ -105,7 +105,7 @@ public:
|
||||
have the same size.
|
||||
|
||||
@note Defining this method overwrites the implicitly defined copy
|
||||
assignemnt operator. */
|
||||
assignment operator. */
|
||||
GridFunction &operator=(const GridFunction &rhs)
|
||||
{ return operator=((const Vector &)rhs); }
|
||||
|
||||
@@ -714,7 +714,7 @@ public:
|
||||
the same size.
|
||||
|
||||
@note Defining this method overwrites the implicitly defined copy
|
||||
assignemnt operator. */
|
||||
assignment operator. */
|
||||
QuadratureFunction &operator=(const QuadratureFunction &v);
|
||||
|
||||
/// Get the IntegrationRule associated with mesh element @a idx.
|
||||
|
||||
@@ -618,6 +618,26 @@ void QuadratureFunctions1D::OpenHalfUniform(const int np, IntegrationRule* ir)
|
||||
CalculateUniformWeights(ir, Quadrature1D::OpenHalfUniform);
|
||||
}
|
||||
|
||||
void QuadratureFunctions1D::ClosedGL(const int np, IntegrationRule* ir)
|
||||
{
|
||||
ir->SetSize(np);
|
||||
ir->IntPoint(0).x = 0.0;
|
||||
ir->IntPoint(np-1).x = 1.0;
|
||||
|
||||
if ( np > 2 )
|
||||
{
|
||||
IntegrationRule gl_ir;
|
||||
GaussLegendre(np-1, &gl_ir);
|
||||
|
||||
for (int i = 1; i < np-1; ++i)
|
||||
{
|
||||
ir->IntPoint(i).x = (gl_ir.IntPoint(i-1).x + gl_ir.IntPoint(i).x)/2;
|
||||
}
|
||||
}
|
||||
|
||||
CalculateUniformWeights(ir, Quadrature1D::ClosedGL);
|
||||
}
|
||||
|
||||
void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
|
||||
const int type)
|
||||
{
|
||||
@@ -650,6 +670,11 @@ void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
|
||||
OpenHalfUniform(np, &ir);
|
||||
break;
|
||||
}
|
||||
case Quadrature1D::ClosedGL:
|
||||
{
|
||||
ClosedGL(np, &ir);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
{
|
||||
MFEM_ABORT("Asking for an unknown type of 1D Quadrature points, "
|
||||
|
||||
+3
-1
@@ -272,6 +272,7 @@ public:
|
||||
void OpenUniform(const int np, IntegrationRule *ir);
|
||||
void ClosedUniform(const int np, IntegrationRule *ir);
|
||||
void OpenHalfUniform(const int np, IntegrationRule *ir);
|
||||
void ClosedGL(const int np, IntegrationRule *ir);
|
||||
///@}
|
||||
|
||||
/// A helper function that will play nice with Poly_1D::OpenPoints and
|
||||
@@ -293,7 +294,8 @@ public:
|
||||
GaussLobatto = 1,
|
||||
OpenUniform = 2, ///< aka open Newton-Cotes
|
||||
ClosedUniform = 3, ///< aka closed Newton-Cotes
|
||||
OpenHalfUniform = 4 ///< aka "open half" Newton-Cotes
|
||||
OpenHalfUniform = 4, ///< aka "open half" Newton-Cotes
|
||||
ClosedGL = 5 ///< aka closed Gauss Legendre
|
||||
};
|
||||
/** @brief If the Quadrature1D type is not closed return Invalid; otherwise
|
||||
return type. */
|
||||
|
||||
+1465
File diff suppressed because it is too large
Load Diff
@@ -415,6 +415,59 @@ void CeedPAAssemble(const CeedPAOperator& op,
|
||||
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
|
||||
}
|
||||
|
||||
void CeedAddMultPA(const CeedData *ceedDataPtr,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
{
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorTakeArray(ceedDataPtr->u, mem, const_cast<CeedScalar**>(&x_ptr));
|
||||
CeedVectorTakeArray(ceedDataPtr->v, mem, &y_ptr);
|
||||
}
|
||||
|
||||
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
|
||||
Vector &diag)
|
||||
{
|
||||
CeedScalar *d_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
d_ptr = diag.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
d_ptr = diag.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, d_ptr);
|
||||
|
||||
CeedOperatorLinearAssembleAddDiagonal(ceedDataPtr->oper, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorTakeArray(ceedDataPtr->v, mem, &d_ptr);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_CEED
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
#ifdef MFEM_USE_CEED
|
||||
#include "../../general/device.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include <ceed.h>
|
||||
|
||||
namespace mfem
|
||||
@@ -144,6 +145,15 @@ const std::string &GetCeedPath();
|
||||
void CeedPAAssemble(const CeedPAOperator& op,
|
||||
CeedData& ceedData);
|
||||
|
||||
/** @brief Function that applies a libCEED PA operator. */
|
||||
void CeedAddMultPA(const CeedData *ceedDataPtr,
|
||||
const Vector &x,
|
||||
Vector &y);
|
||||
|
||||
/** @brief Function that assembles a libCEED PA operator diagonal. */
|
||||
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
|
||||
Vector &diag);
|
||||
|
||||
/** @brief Function that determines if a CEED kernel should be used, based on
|
||||
the current mfem::Device configuration. */
|
||||
inline bool DeviceCanUseCeed()
|
||||
|
||||
+2
-1
@@ -199,7 +199,8 @@ void LinearForm::Assemble()
|
||||
void LinearForm::Update(FiniteElementSpace *f, Vector &v, int v_offset)
|
||||
{
|
||||
fes = f;
|
||||
NewDataAndSize((double *)v + v_offset, fes->GetVSize());
|
||||
NewMemoryAndSize(Memory<double>(v.GetMemory(), v_offset, f->GetVSize()),
|
||||
f->GetVSize(), false);
|
||||
ResetDeltaLocations();
|
||||
}
|
||||
|
||||
|
||||
+11
-4
@@ -135,11 +135,18 @@ void Multigrid::SetOperator(const Operator& op)
|
||||
MFEM_ABORT("SetOperator not supported in Multigrid");
|
||||
}
|
||||
|
||||
void Multigrid::SmoothingStep(int level) const
|
||||
void Multigrid::SmoothingStep(int level, bool transpose) const
|
||||
{
|
||||
GetOperatorAtLevel(level)->Mult(*Y[level], *R[level]); // r = A x
|
||||
subtract(*X[level], *R[level], *R[level]); // r = b - A x
|
||||
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
|
||||
if (transpose)
|
||||
{
|
||||
GetSmootherAtLevel(level)->MultTranspose(*R[level], *Z[level]); // z = S r
|
||||
}
|
||||
else
|
||||
{
|
||||
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
|
||||
}
|
||||
add(*Y[level], 1.0, *Z[level], *Y[level]); // x = x + S (b - A x)
|
||||
}
|
||||
|
||||
@@ -153,7 +160,7 @@ void Multigrid::Cycle(int level) const
|
||||
|
||||
for (int i = 0; i < preSmoothingSteps; i++)
|
||||
{
|
||||
SmoothingStep(level);
|
||||
SmoothingStep(level, false);
|
||||
}
|
||||
|
||||
// Compute residual
|
||||
@@ -187,7 +194,7 @@ void Multigrid::Cycle(int level) const
|
||||
// Post-smooth
|
||||
for (int i = 0; i < postSmoothingSteps; i++)
|
||||
{
|
||||
SmoothingStep(level);
|
||||
SmoothingStep(level, true);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -108,7 +108,7 @@ public:
|
||||
|
||||
private:
|
||||
/// Application of a smoothing step at particular level
|
||||
void SmoothingStep(int level) const;
|
||||
void SmoothingStep(int level, bool transpose) const;
|
||||
|
||||
/// Application of a cycle at particular level
|
||||
void Cycle(int level) const;
|
||||
|
||||
+63
-14
@@ -10,6 +10,7 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -27,7 +28,7 @@ void NonlinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
// This is the default behavior.
|
||||
break;
|
||||
case AssemblyLevel::PARTIAL:
|
||||
ext = new PANonlinearFormExtension(this);
|
||||
ext = new PANonlinearForm(this);
|
||||
break;
|
||||
default:
|
||||
mfem_error("Unknown assembly level for this form.");
|
||||
@@ -80,6 +81,13 @@ void NonlinearForm::SetEssentialVDofs(const Array<int> &ess_vdofs_list)
|
||||
|
||||
double NonlinearForm::GetGridFunctionEnergy(const Vector &x) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
MFEM_VERIFY(!fnfi.Size(), "Interior faces terms not yet implemented!");
|
||||
MFEM_VERIFY(!bfnfi.Size(), "Boundary face terms not yet implemented!");
|
||||
return ext->GetGridFunctionEnergy(x);
|
||||
}
|
||||
|
||||
Array<int> vdofs;
|
||||
Vector el_x;
|
||||
const FiniteElement *fe;
|
||||
@@ -138,6 +146,14 @@ void NonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
if (ext)
|
||||
{
|
||||
ext->Mult(px, py);
|
||||
if (Serial())
|
||||
{
|
||||
if (cP) { cP->MultTranspose(py, y); }
|
||||
const int N = ess_tdof_list.Size();
|
||||
const auto tdof = ess_tdof_list.Read();
|
||||
auto Y = y.ReadWrite();
|
||||
MFEM_FORALL(i, N, Y[tdof[i]] = 0.0; );
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -264,7 +280,16 @@ Operator &NonlinearForm::GetGradient(const Vector &x) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
MFEM_ABORT("Not yet implemented!");
|
||||
Operator &grad = ext->GetGradient(Prolongate(x));
|
||||
hGrad.Reset(&grad, false);
|
||||
if (Serial())
|
||||
{
|
||||
Operator *Gop;
|
||||
if (cP) { hGrad.Reset(new RAPOperator(*cP, grad, *cP)); }
|
||||
hGrad.Ptr()->Operator::FormSystemOperator(ess_tdof_list, Gop);
|
||||
hGrad.Reset(Gop);
|
||||
}
|
||||
return *hGrad.Ptr();
|
||||
}
|
||||
|
||||
const int skip_zeros = 0;
|
||||
@@ -426,7 +451,31 @@ void NonlinearForm::Update()
|
||||
|
||||
void NonlinearForm::Setup()
|
||||
{
|
||||
if (ext) { return ext->AssemblePA(); }
|
||||
if (ext) { return ext->Setup(); }
|
||||
}
|
||||
|
||||
void NonlinearForm::AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
|
||||
"Vector for holding diagonal has wrong size!");
|
||||
const Operator *P = fes->GetProlongationMatrix();
|
||||
if (!IsIdentityProlongation(P))
|
||||
{
|
||||
Vector local_diag(P->Height());
|
||||
ext->AssembleGradientDiagonal(local_diag);
|
||||
P->MultTranspose(local_diag, diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
ext->AssembleGradientDiagonal(diag);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Not implemented. Can be obtained through GetGradient().");
|
||||
}
|
||||
}
|
||||
|
||||
NonlinearForm::~NonlinearForm()
|
||||
@@ -933,6 +982,17 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
|
||||
}
|
||||
}
|
||||
|
||||
if (!Grads(0,0)->Finalized())
|
||||
{
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
{
|
||||
Grads(i,j)->Finalize(skip_zeros);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int s=0; s<fes.Size(); ++s)
|
||||
{
|
||||
for (int i = 0; i < ess_vdofs[s]->Size(); ++i)
|
||||
@@ -952,17 +1012,6 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
|
||||
}
|
||||
}
|
||||
|
||||
if (!Grads(0,0)->Finalized())
|
||||
{
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
{
|
||||
Grads(i,j)->Finalize(skip_zeros);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
|
||||
@@ -45,6 +45,7 @@ protected:
|
||||
Array<Array<int>*> bfnfi_marker; // not owned
|
||||
|
||||
mutable SparseMatrix *Grad, *cGrad; // owned
|
||||
mutable OperatorHandle hGrad;
|
||||
|
||||
/// A list of all essential true dofs
|
||||
Array<int> ess_tdof_list;
|
||||
@@ -165,6 +166,15 @@ public:
|
||||
/// Setup the NonlinearForm
|
||||
virtual void Setup();
|
||||
|
||||
/** @brief Assemble the diagonal of the gradient into diag
|
||||
|
||||
For adaptively refined meshes, this returns P^T d_e, where d_e is the
|
||||
locally assembled diagonal on each element and P^T is the transpose of
|
||||
the conforming prolongation. In general this is not the correct diagonal
|
||||
for an AMR mesh. */
|
||||
void AssembleGradientDiagonal(Vector &diag) const;
|
||||
|
||||
|
||||
/// Get the finite element space prolongation matrix
|
||||
virtual const Operator *GetProlongation() const { return P; }
|
||||
/// Get the finite element space restriction matrix
|
||||
|
||||
+79
-40
@@ -13,62 +13,101 @@
|
||||
// PABilinearFormExtension and MFBilinearFormExtension.
|
||||
|
||||
#include "nonlinearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
NonlinearFormExtension::NonlinearFormExtension(NonlinearForm *form)
|
||||
: Operator(form->FESpace()->GetTrueVSize()), n(form)
|
||||
NonlinearFormExtension::NonlinearFormExtension(const NonlinearForm *nlf)
|
||||
: Operator(nlf->FESpace()->GetTrueVSize()), nlf(nlf) { }
|
||||
|
||||
PANonlinearForm::PANonlinearForm(NonlinearForm *nlf):
|
||||
NonlinearFormExtension(nlf),
|
||||
x_grad(NULL),
|
||||
fes(*nlf->FESpace()),
|
||||
dnfi(*nlf->GetDNFI()),
|
||||
R(fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC))
|
||||
{
|
||||
// empty
|
||||
MFEM_VERIFY(R, "Not yet implemented!");
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
ye.SetSize(R->Height(), Device::GetMemoryType());
|
||||
ye.UseDevice(true);
|
||||
}
|
||||
|
||||
PANonlinearFormExtension::PANonlinearFormExtension(NonlinearForm *form):
|
||||
NonlinearFormExtension(form), fes(*form->FESpace())
|
||||
double PANonlinearForm::GetGridFunctionEnergy(const Vector &x) const
|
||||
{
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
elem_restrict_lex = fes.GetElementRestriction(ordering);
|
||||
if (elem_restrict_lex)
|
||||
double energy = 0.0;
|
||||
|
||||
R->Mult(x, xe);
|
||||
for (int i = 0; i < dnfi.Size(); i++)
|
||||
{
|
||||
localX.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
|
||||
localY.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
|
||||
localY.UseDevice(true); // ensure 'localY = 0.0' is done on device
|
||||
energy += dnfi[i]->GetGridFunctionEnergyPA(xe);
|
||||
}
|
||||
return energy;
|
||||
}
|
||||
|
||||
void PANonlinearFormExtension::AssemblePA()
|
||||
void PANonlinearForm::Setup()
|
||||
{
|
||||
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
|
||||
const int Ni = integrators.Size();
|
||||
for (int i = 0; i < Ni; ++i)
|
||||
{
|
||||
integrators[i]->AssemblePA(*n->FESpace());
|
||||
}
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(fes); }
|
||||
}
|
||||
|
||||
void PANonlinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
void PANonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
|
||||
const int iSz = integrators.Size();
|
||||
if (elem_restrict_lex)
|
||||
{
|
||||
elem_restrict_lex->Mult(x, localX);
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AddMultPA(localX, localY);
|
||||
}
|
||||
elem_restrict_lex->MultTranspose(localY, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
y.UseDevice(true); // typically this is a large vector, so store on device
|
||||
y = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AddMultPA(x, y);
|
||||
}
|
||||
}
|
||||
ye = 0.0;
|
||||
R->Mult(x, xe);
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultPA(xe, ye); }
|
||||
R->MultTranspose(ye, y);
|
||||
}
|
||||
|
||||
void PANonlinearForm::AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
MFEM_VERIFY(x_grad, "GetGradient() has not been called");
|
||||
R->Mult(*x_grad, xe);
|
||||
|
||||
ye = 0.0;
|
||||
for (int i = 0; i < dnfi.Size(); ++i)
|
||||
{
|
||||
dnfi[i]->AssembleGradientDiagonalPA(xe, ye);
|
||||
}
|
||||
R->MultTranspose(ye, diag);
|
||||
}
|
||||
|
||||
Operator &PANonlinearForm::GetGradient(const Vector &x) const
|
||||
{
|
||||
// Store the last x that was used to compute the gradient.
|
||||
x_grad = &x;
|
||||
|
||||
Grad.Reset(new PANonlinearForm::Gradient(x, *this));
|
||||
return *Grad.Ptr();
|
||||
}
|
||||
|
||||
PANonlinearForm::Gradient::Gradient(const Vector &x, const PANonlinearForm &e):
|
||||
Operator(e.fes.GetVSize()), R(e.R), dnfi(e.dnfi)
|
||||
{
|
||||
ge.UseDevice(true);
|
||||
ge.SetSize(R->Height(), Device::GetMemoryType());
|
||||
R->Mult(x, ge);
|
||||
|
||||
xe.UseDevice(true);
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
ye.UseDevice(true);
|
||||
ye.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
ze.UseDevice(true);
|
||||
ze.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
// Do we still need to do this?
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(e.fes); }
|
||||
}
|
||||
|
||||
void PANonlinearForm::Gradient::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
ze = x;
|
||||
ye = 0.0;
|
||||
R->Mult(ze, xe);
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultGradPA(ge, xe, ye); }
|
||||
R->MultTranspose(ye, y);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
+42
-10
@@ -17,28 +17,60 @@
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
class NonlinearForm;
|
||||
|
||||
class NonlinearForm;
|
||||
class NonlinearFormIntegrator;
|
||||
|
||||
/** @brief Class extending the NonlinearForm class to support the different
|
||||
AssemblyLevel%s. */
|
||||
class NonlinearFormExtension : public Operator
|
||||
{
|
||||
protected:
|
||||
NonlinearForm *n; ///< Not owned
|
||||
const NonlinearForm *nlf;
|
||||
public:
|
||||
NonlinearFormExtension(NonlinearForm *form);
|
||||
virtual void AssemblePA() = 0;
|
||||
NonlinearFormExtension(const NonlinearForm*);
|
||||
virtual void Setup() = 0;
|
||||
virtual Operator &GetGradient(const Vector&) const = 0;
|
||||
virtual double GetGridFunctionEnergy(const Vector &x) const = 0;
|
||||
virtual void AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
MFEM_ABORT("Not implemented for this assembly level!");
|
||||
}
|
||||
};
|
||||
|
||||
class PANonlinearForm;
|
||||
|
||||
|
||||
/// Data and methods for partially-assembled nonlinear forms
|
||||
class PANonlinearFormExtension : public NonlinearFormExtension
|
||||
class PANonlinearForm : public NonlinearFormExtension
|
||||
{
|
||||
private:
|
||||
class Gradient : public Operator
|
||||
{
|
||||
protected:
|
||||
const Operator *R;
|
||||
mutable Vector ge, xe, ye, ze;
|
||||
const Array<NonlinearFormIntegrator*> &dnfi;
|
||||
public:
|
||||
Gradient(const Vector &x, const PANonlinearForm &ext);
|
||||
virtual void Mult(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
protected:
|
||||
const FiniteElementSpace &fes; // Not owned
|
||||
mutable Vector localX, localY;
|
||||
const Operator *elem_restrict_lex; // Not owned
|
||||
mutable Vector xe, ye;
|
||||
mutable const Vector *x_grad;
|
||||
mutable OperatorHandle Grad;
|
||||
const FiniteElementSpace &fes;
|
||||
const Array<NonlinearFormIntegrator*> &dnfi;
|
||||
const Operator *R;
|
||||
|
||||
public:
|
||||
PANonlinearFormExtension(NonlinearForm*);
|
||||
void AssemblePA();
|
||||
PANonlinearForm(NonlinearForm *nlf);
|
||||
void Setup();
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
Operator &GetGradient(const Vector &x) const;
|
||||
double GetGridFunctionEnergy(const Vector &x) const;
|
||||
void AssembleGradientDiagonal(Vector &diag) const;
|
||||
};
|
||||
}
|
||||
#endif // NONLINEARFORM_EXT_HPP
|
||||
|
||||
@@ -15,6 +15,13 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
double NonlinearFormIntegrator::GetGridFunctionEnergyPA(const Vector &x) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::GetGridFunctionEnergyPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
|
||||
@@ -34,6 +41,20 @@ void NonlinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AddMultGradPA(const Vector&,
|
||||
const Vector&, Vector&) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AddMultGradPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssembleGradientDiagonalPA(const mfem::Vector &x,
|
||||
mfem::Vector &diag) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AssembleDiagonalPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssembleElementVector(
|
||||
const FiniteElement &el, ElementTransformation &Tr,
|
||||
const Vector &elfun, Vector &elvect)
|
||||
|
||||
@@ -68,6 +68,9 @@ public:
|
||||
ElementTransformation &Tr,
|
||||
const Vector &elfun);
|
||||
|
||||
/// Compute the local energy with partial assembly.
|
||||
virtual double GetGridFunctionEnergyPA(const Vector &x) const;
|
||||
|
||||
/// Method defining partial assembly.
|
||||
/** The result of the partial assembly is stored internally so that it can be
|
||||
used later in the methods AddMultPA(). */
|
||||
@@ -88,6 +91,12 @@ public:
|
||||
called. */
|
||||
virtual void AddMultPA(const Vector &x, Vector &y) const;
|
||||
|
||||
/// Method for partially assembled gradient action.
|
||||
virtual void AddMultGradPA(const Vector &g,
|
||||
const Vector &x, Vector &y) const;
|
||||
|
||||
virtual void AssembleGradientDiagonalPA(const Vector &x, Vector &diag) const;
|
||||
|
||||
virtual ~NonlinearFormIntegrator() { }
|
||||
};
|
||||
|
||||
|
||||
@@ -241,7 +241,7 @@ void ParBilinearForm::Assemble(int skip_zeros)
|
||||
|
||||
BilinearForm::Assemble(skip_zeros);
|
||||
|
||||
if (fbfi.Size() > 0)
|
||||
if (!ext && fbfi.Size() > 0)
|
||||
{
|
||||
AssembleSharedFaces(skip_zeros);
|
||||
}
|
||||
|
||||
+2
-2
@@ -3147,7 +3147,7 @@ static void SetSubVector(const int N,
|
||||
const Array<int> &indices,
|
||||
const Vector &in, Vector &out)
|
||||
{
|
||||
auto y = out.Write();
|
||||
auto y = out.ReadWrite();
|
||||
const auto x = in.Read();
|
||||
const auto I = indices.Read();
|
||||
MFEM_FORALL(i, N, y[I[i]] = x[i];);
|
||||
@@ -3234,7 +3234,7 @@ static void AddSubVector(const int num_unique_dst_indices,
|
||||
const Vector &src,
|
||||
Vector &dst)
|
||||
{
|
||||
auto y = dst.Write();
|
||||
auto y = dst.ReadWrite();
|
||||
const auto x = src.Read();
|
||||
const auto DST_I = unique_dst_indices.Read();
|
||||
const auto SRC_O = unique_to_src_offsets.Read();
|
||||
|
||||
@@ -711,7 +711,9 @@ void ParGridFunction::SaveAsOne(std::ostream &out)
|
||||
int *nfdofs = new int[NRanks];
|
||||
int *nrdofs = new int[NRanks];
|
||||
|
||||
HostReadWrite();
|
||||
values[0] = data;
|
||||
|
||||
nv[0] = pfes -> GetVSize();
|
||||
nvdofs[0] = pfes -> GetNVDofs();
|
||||
nedofs[0] = pfes -> GetNEDofs();
|
||||
|
||||
+16
-9
@@ -14,6 +14,7 @@
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -49,6 +50,7 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
|
||||
if (fnfi.Size())
|
||||
{
|
||||
MFEM_VERIFY(!NonlinearForm::ext,"");
|
||||
// Terms over shared interior faces in parallel.
|
||||
ParFiniteElementSpace *pfes = ParFESpace();
|
||||
ParMesh *pmesh = pfes->GetParMesh();
|
||||
@@ -86,15 +88,16 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
|
||||
P->MultTranspose(aux2, y);
|
||||
|
||||
y.HostReadWrite();
|
||||
for (int i = 0; i < ess_tdof_list.Size(); i++)
|
||||
{
|
||||
y(ess_tdof_list[i]) = 0.0;
|
||||
}
|
||||
const int N = ess_tdof_list.Size();
|
||||
const auto idx = ess_tdof_list.Read();
|
||||
auto Y = y.ReadWrite();
|
||||
MFEM_FORALL(i, N, Y[idx[i]] = 0.0; );
|
||||
}
|
||||
|
||||
const SparseMatrix &ParNonlinearForm::GetLocalGradient(const Vector &x) const
|
||||
{
|
||||
if (NonlinearForm::ext) { MFEM_ABORT("Not yet implemented!"); }
|
||||
|
||||
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
|
||||
return *Grad;
|
||||
@@ -104,16 +107,20 @@ Operator &ParNonlinearForm::GetGradient(const Vector &x) const
|
||||
{
|
||||
ParFiniteElementSpace *pfes = ParFESpace();
|
||||
|
||||
pGrad.Clear();
|
||||
Operator &grad = NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
|
||||
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
pGrad.Clear();
|
||||
|
||||
OperatorHandle dA(pGrad.Type()), Ph(pGrad.Type());
|
||||
|
||||
if (fnfi.Size() == 0)
|
||||
{
|
||||
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
|
||||
pfes->GetDofOffsets(), Grad);
|
||||
if (NonlinearForm::ext) { dA.Reset(&grad, false); }
|
||||
else
|
||||
{
|
||||
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
|
||||
pfes->GetDofOffsets(), Grad);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
+136
-57
@@ -27,35 +27,24 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
ElementDofOrdering e_ordering,
|
||||
FaceType type,
|
||||
L2FaceValues m)
|
||||
: fes(fes),
|
||||
nf(fes.GetNFbyType(type)),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndofs(fes.GetNDofs()),
|
||||
dof(nf>0 ?
|
||||
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
|
||||
: 0),
|
||||
m(m),
|
||||
nfdofs(nf*dof),
|
||||
scatter_indices1(nf*dof),
|
||||
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
|
||||
offsets(ndofs+1),
|
||||
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
|
||||
: L2FaceRestriction(fes, type, m)
|
||||
{
|
||||
if (nf==0) { return; }
|
||||
// If fespace == L2
|
||||
const FiniteElement *fe = fes.GetFE(0);
|
||||
const ParFiniteElementSpace &pfes =
|
||||
static_cast<const ParFiniteElementSpace&>(this->fes);
|
||||
const FiniteElement *fe = pfes.GetFE(0);
|
||||
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement*>(fe);
|
||||
MFEM_VERIFY(tfe != NULL &&
|
||||
(tfe->GetBasisType()==BasisType::GaussLobatto ||
|
||||
tfe->GetBasisType()==BasisType::Positive),
|
||||
"Only Gauss-Lobatto and Bernstein basis are supported in "
|
||||
"ParL2FaceRestriction.");
|
||||
MFEM_VERIFY(fes.GetMesh()->Conforming(),
|
||||
MFEM_VERIFY(pfes.GetMesh()->Conforming(),
|
||||
"Non-conforming meshes not yet supported with partial assembly.");
|
||||
// Assuming all finite elements are using Gauss-Lobatto dofs
|
||||
height = (m==L2FaceValues::DoubleValued? 2 : 1)*vdim*nf*dof;
|
||||
width = fes.GetVSize();
|
||||
width = pfes.GetVSize();
|
||||
const bool dof_reorder = (e_ordering == ElementDofOrdering::LEXICOGRAPHIC);
|
||||
if (!dof_reorder)
|
||||
{
|
||||
@@ -63,32 +52,32 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
}
|
||||
if (dof_reorder && nf > 0)
|
||||
{
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
{
|
||||
const FiniteElement *fe =
|
||||
fes.GetTraceElement(f, fes.GetMesh()->GetFaceBaseGeometry(f));
|
||||
pfes.GetTraceElement(f, pfes.GetMesh()->GetFaceBaseGeometry(f));
|
||||
const TensorBasisElement* el =
|
||||
dynamic_cast<const TensorBasisElement*>(fe);
|
||||
if (el) { continue; }
|
||||
mfem_error("Finite element not suitable for lexicographic ordering");
|
||||
}
|
||||
}
|
||||
const Table& e2dTable = fes.GetElementToDofTable();
|
||||
const Table& e2dTable = pfes.GetElementToDofTable();
|
||||
const int* elementMap = e2dTable.GetJ();
|
||||
Array<int> faceMap1(dof), faceMap2(dof);
|
||||
int e1, e2;
|
||||
int inf1, inf2;
|
||||
int face_id1, face_id2;
|
||||
int orientation;
|
||||
const int dof1d = fes.GetFE(0)->GetOrder()+1;
|
||||
const int elem_dofs = fes.GetFE(0)->GetDof();
|
||||
const int dim = fes.GetMesh()->SpaceDimension();
|
||||
const int dof1d = pfes.GetFE(0)->GetOrder()+1;
|
||||
const int elem_dofs = pfes.GetFE(0)->GetDof();
|
||||
const int dim = pfes.GetMesh()->SpaceDimension();
|
||||
// Computation of scatter indices
|
||||
int f_ind=0;
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
{
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if (dof_reorder)
|
||||
{
|
||||
orientation = inf1 % 64;
|
||||
@@ -136,7 +125,7 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
{
|
||||
const int se2 = -1 - e2;
|
||||
Array<int> sharedDofs;
|
||||
fes.GetFaceNbrElementVDofs(se2, sharedDofs);
|
||||
pfes.GetFaceNbrElementVDofs(se2, sharedDofs);
|
||||
for (int d = 0; d < dof; ++d)
|
||||
{
|
||||
const int pd = PermuteFaceL2(dim, face_id1, face_id2,
|
||||
@@ -180,10 +169,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
offsets[i] = 0;
|
||||
}
|
||||
f_ind = 0;
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
{
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
|
||||
(type==FaceType::Boundary && e2<0 && inf2<0) )
|
||||
{
|
||||
@@ -222,10 +211,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
f_ind = 0;
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
{
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
|
||||
(type==FaceType::Boundary && e2<0 && inf2<0) )
|
||||
{
|
||||
@@ -272,8 +261,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
|
||||
void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
|
||||
{
|
||||
const ParFiniteElementSpace &pfes =
|
||||
static_cast<const ParFiniteElementSpace&>(this->fes);
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&fes),
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&pfes),
|
||||
const_cast<Vector&>(x), 0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
|
||||
@@ -337,34 +328,122 @@ void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
void ParL2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
|
||||
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
|
||||
{
|
||||
// Assumes all elements have the same number of dofs
|
||||
const int nd = dof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
const int dofs = nfdofs;
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = gather_indices.Read();
|
||||
auto d_x = Reshape(x.Read(), nd, vd, 2, nf);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
int val = AtomicAdd(I[iE],dofs);
|
||||
return val;
|
||||
}
|
||||
|
||||
void ParL2FaceRestriction::FillI(SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int Ndofs = ndofs;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
auto I_face = face_mat.ReadWriteI();
|
||||
MFEM_FORALL(i, ne*elemDofs*vdim+1,
|
||||
{
|
||||
const int offset = d_offsets[i];
|
||||
const int nextOffset = d_offsets[i + 1];
|
||||
for (int c = 0; c < vd; ++c)
|
||||
I_face[i] = 0;
|
||||
});
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
if (iE1 < Ndofs)
|
||||
{
|
||||
double dofValue = 0;
|
||||
for (int j = offset; j < nextOffset; ++j)
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
int idx_j = d_indices[j];
|
||||
bool isE1 = idx_j < dofs;
|
||||
idx_j = isE1 ? idx_j : idx_j - dofs;
|
||||
dofValue += isE1 ?
|
||||
d_x(idx_j % nd, c, 0, idx_j / nd)
|
||||
:d_x(idx_j % nd, c, 1, idx_j / nd);
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
if (jE2 < Ndofs)
|
||||
{
|
||||
AddNnz(iE1,I,1);
|
||||
}
|
||||
else
|
||||
{
|
||||
AddNnz(iE1,I_face,1);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
if (iE2 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
if (jE1 < Ndofs)
|
||||
{
|
||||
AddNnz(iE2,I,1);
|
||||
}
|
||||
else
|
||||
{
|
||||
AddNnz(iE2,I_face,1);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void ParL2FaceRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int Ndofs = ndofs;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto I = mat.ReadWriteI();
|
||||
auto I_face = face_mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto J_face = face_mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto Data_face = face_mat.WriteData();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
if (iE1 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
if (jE2 < Ndofs)
|
||||
{
|
||||
const int offset = AddNnz(iE1,I,1);
|
||||
J[offset] = jE2;
|
||||
Data[offset] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
else
|
||||
{
|
||||
const int offset = AddNnz(iE1,I_face,1);
|
||||
J_face[offset] = jE2-Ndofs;
|
||||
Data_face[offset] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
if (iE2 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
if (jE1 < Ndofs)
|
||||
{
|
||||
const int offset = AddNnz(iE2,I,1);
|
||||
J[offset] = jE1;
|
||||
Data[offset] = mat_fea(jF,iF,0,f);
|
||||
}
|
||||
else
|
||||
{
|
||||
const int offset = AddNnz(iE2,I_face,1);
|
||||
J_face[offset] = jE1-Ndofs;
|
||||
Data_face[offset] = mat_fea(jF,iF,0,f);
|
||||
}
|
||||
}
|
||||
d_y(t?c:i,t?i:c) += dofValue;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
+9
-16
@@ -26,28 +26,21 @@ class ParFiniteElementSpace;
|
||||
/// Operator that extracts Face degrees of freedom in parallel.
|
||||
/** Objects of this type are typically created and owned by FiniteElementSpace
|
||||
objects, see FiniteElementSpace::GetFaceRestriction(). */
|
||||
class ParL2FaceRestriction : public Operator
|
||||
class ParL2FaceRestriction : public L2FaceRestriction
|
||||
{
|
||||
protected:
|
||||
const ParFiniteElementSpace &fes;
|
||||
const int nf;
|
||||
const int vdim;
|
||||
const bool byvdim;
|
||||
const int ndofs;
|
||||
const int dof;
|
||||
const L2FaceValues m;
|
||||
const int nfdofs;
|
||||
Array<int> scatter_indices1;
|
||||
Array<int> scatter_indices2;
|
||||
Array<int> offsets;
|
||||
Array<int> gather_indices;
|
||||
|
||||
public:
|
||||
ParL2FaceRestriction(const ParFiniteElementSpace&, ElementDofOrdering,
|
||||
FaceType type,
|
||||
L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ParL2FaceRestriction. */
|
||||
void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this ParL2FaceRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const;
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
+398
-1123
File diff suppressed because it is too large
Load Diff
+33
-15
@@ -41,10 +41,11 @@ protected:
|
||||
const FiniteElementSpace *fespace; ///< Not owned
|
||||
const QuadratureSpace *qspace; ///< Not owned
|
||||
const IntegrationRule *IntRule; ///< Not owned
|
||||
|
||||
mutable QVectorLayout q_layout; ///< Output Q-vector layout
|
||||
mutable bool use_tensor_products; ///< Tensor product evaluation mmode
|
||||
|
||||
mutable bool use_tensor_products;
|
||||
|
||||
public:
|
||||
static const int MAX_NQ2D = 100;
|
||||
static const int MAX_ND2D = 100;
|
||||
static const int MAX_VDIM2D = 3;
|
||||
@@ -53,7 +54,6 @@ protected:
|
||||
static const int MAX_ND3D = 1000;
|
||||
static const int MAX_VDIM3D = 3;
|
||||
|
||||
public:
|
||||
enum EvalFlags
|
||||
{
|
||||
VALUES = 1 << 0, ///< Evaluate the values at quadrature points
|
||||
@@ -61,21 +61,28 @@ public:
|
||||
/** @brief Assuming the derivative at quadrature points form a matrix,
|
||||
this flag can be used to compute and store their determinants. This
|
||||
flag can only be used in Mult(). */
|
||||
DETERMINANTS = 1 << 2
|
||||
DETERMINANTS = 1 << 2,
|
||||
PHYSICAL_DERIVATIVES = 1 << 3 ///< Evaluate the physical derivatives
|
||||
};
|
||||
|
||||
QuadratureInterpolator(const FiniteElementSpace &fes,
|
||||
const IntegrationRule &ir);
|
||||
const IntegrationRule &ir,
|
||||
const bool use_tensor_products = false);
|
||||
|
||||
QuadratureInterpolator(const FiniteElementSpace &fes,
|
||||
const QuadratureSpace &qs);
|
||||
const QuadratureSpace &qs,
|
||||
const bool use_tensor_products = false);
|
||||
|
||||
/** @brief Disable the use of tensor product evaluations, for tensor-product
|
||||
elements, e.g. quads and hexes. */
|
||||
/** Currently, tensor product evaluations are not implemented and this method
|
||||
has no effect. */
|
||||
void DisableTensorProducts(bool disable = true) const
|
||||
{ use_tensor_products = !disable; }
|
||||
void DisableTensorProducts() const { use_tensor_products = false; }
|
||||
|
||||
/** @brief Enable the use of tensor product evaluations, for tensor-product
|
||||
elements, e.g. quads and hexes. */
|
||||
void EnableTensorProducts() const { use_tensor_products = true; }
|
||||
|
||||
/** @brief Query the current evaluation mode. */
|
||||
bool UseTensorProducts() const { return use_tensor_products; }
|
||||
|
||||
/** @brief Query the current output Q-vector layout. The default value is
|
||||
QVectorLayout::byNODES. */
|
||||
@@ -83,8 +90,7 @@ public:
|
||||
|
||||
/** @brief Set the desired output Q-vector layout. The default value is
|
||||
QVectorLayout::byNODES. */
|
||||
void SetOutputLayout(QVectorLayout out_layout) const
|
||||
{ q_layout = out_layout; }
|
||||
void SetOutputLayout(QVectorLayout layout) const { q_layout = layout; }
|
||||
|
||||
/// Interpolate the E-vector @a e_vec to quadrature points.
|
||||
/** The @a eval_flags are a bitwise mask of constants from the EvalFlags
|
||||
@@ -99,26 +105,36 @@ public:
|
||||
Vector &q_val, Vector &q_der, Vector &q_det) const;
|
||||
|
||||
/// Interpolate the values of the E-vector @a e_vec at quadrature points.
|
||||
template <QVectorLayout>
|
||||
void Values(const Vector &e_vec, Vector &q_val) const;
|
||||
void Values(const Vector &e_vec, Vector &q_val) const;
|
||||
|
||||
/** @brief Interpolate the derivatives of the E-vector @a e_vec at quadrature
|
||||
points. */
|
||||
template <QVectorLayout>
|
||||
void Derivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
void Derivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
|
||||
/** @brief Interpolate the derivatives in physical space of the E-vector
|
||||
@a e_vec at quadrature points. */
|
||||
template <QVectorLayout>
|
||||
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
|
||||
/// Compute the determinant of the E-vector @a e_vec at quadrature points.
|
||||
void Determinants(const Vector &e_vec, Vector &q_det) const;
|
||||
|
||||
/// Perform the transpose operation of Mult(). (TODO)
|
||||
void MultTranspose(unsigned eval_flags, const Vector &q_val,
|
||||
const Vector &q_der, Vector &e_vec) const;
|
||||
|
||||
// Compute kernels follow (cannot be private or protected with nvcc)
|
||||
|
||||
/// Template compute kernel for 2D.
|
||||
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
|
||||
static void Eval2D(const int NE,
|
||||
static void Mult2D(const int NE,
|
||||
const int vdim,
|
||||
const QVectorLayout q_layout,
|
||||
const GeometricFactors *geom,
|
||||
const DofToQuad &maps,
|
||||
const Vector &e_vec,
|
||||
Vector &q_val,
|
||||
@@ -128,8 +144,10 @@ public:
|
||||
|
||||
/// Template compute kernel for 3D.
|
||||
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
|
||||
static void Eval3D(const int NE,
|
||||
static void Mult3D(const int NE,
|
||||
const int vdim,
|
||||
const QVectorLayout q_layout,
|
||||
const GeometricFactors *geom,
|
||||
const DofToQuad &maps,
|
||||
const Vector &e_vec,
|
||||
Vector &q_val,
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop_pa.hpp"
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../fem/kernels.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Det2D(const int NE,
|
||||
const double *b,
|
||||
const double *g,
|
||||
const double *x,
|
||||
double *y,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b, Q1D, D1D);
|
||||
const auto G = Reshape(g, Q1D, D1D);
|
||||
const auto X = Reshape(x, D1D, D1D, DIM, NE);
|
||||
auto Y = Reshape(y, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double J[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,J);
|
||||
Y(qx,qy,e) = kernels::Det<2>(J);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Det3D(const int NE,
|
||||
const double *b,
|
||||
const double *g,
|
||||
const double *x,
|
||||
double *y,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b, Q1D, D1D);
|
||||
const auto G = Reshape(g, Q1D, D1D);
|
||||
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
|
||||
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double sm0[9][MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[9][MDQ*MDQ*MDQ];
|
||||
|
||||
double (*DDD)[MD1*MD1*MD1] = (double (*)[MD1*MD1*MD1]) (sm0);
|
||||
double (*DDQ)[MD1*MD1*MQ1] = (double (*)[MD1*MD1*MQ1]) (sm1);
|
||||
double (*DQQ)[MD1*MQ1*MQ1] = (double (*)[MD1*MQ1*MQ1]) (sm0);
|
||||
double (*QQQ)[MQ1*MQ1*MQ1] = (double (*)[MQ1*MQ1*MQ1]) (sm1);
|
||||
|
||||
kernels::LoadX<MD1>(e,D1D,X,DDD);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
|
||||
kernels::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
|
||||
kernels::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double J[9];
|
||||
kernels::PullGrad<MQ1>(qx,qy,qz, QQQ, J);
|
||||
Y(qx,qy,qz,e) = kernels::Det<3>(J);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void QuadratureInterpolator::Determinants(const Vector &e_vec,
|
||||
Vector &q_det) const
|
||||
{
|
||||
if (use_tensor_products)
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_det.Write();
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2222: return Det2D<2,2>(NE,B,G,X,Y);
|
||||
case 0x2223: return Det2D<2,3>(NE,B,G,X,Y);
|
||||
case 0x2224: return Det2D<2,4>(NE,B,G,X,Y);
|
||||
case 0x2226: return Det2D<2,6>(NE,B,G,X,Y);
|
||||
case 0x2234: return Det2D<3,4>(NE,B,G,X,Y);
|
||||
case 0x2236: return Det2D<3,6>(NE,B,G,X,Y);
|
||||
case 0x2244: return Det2D<4,4>(NE,B,G,X,Y);
|
||||
case 0x2246: return Det2D<4,6>(NE,B,G,X,Y);
|
||||
case 0x2256: return Det2D<5,6>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3324: return Det3D<2,4>(NE,B,G,X,Y);
|
||||
case 0x3333: return Det3D<3,3>(NE,B,G,X,Y);
|
||||
case 0x3335: return Det3D<3,5>(NE,B,G,X,Y);
|
||||
case 0x3336: return Det3D<3,6>(NE,B,G,X,Y);
|
||||
//case 0x3348: return Det3D<4,8>(NE,B,G,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
return Det2D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
constexpr int MD1 = 6;
|
||||
constexpr int MQ1 = 6;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
return Det3D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
MFEM_ABORT("Kernel " << std::hex << id << std::dec << " not supported yet");
|
||||
}
|
||||
else
|
||||
{
|
||||
Vector empty;
|
||||
Mult(e_vec, DETERMINANTS, empty, empty, q_det);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,233 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Eval2D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, NE):
|
||||
Reshape(y_, VDIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_DD[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> DD((double*)(s_DD+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double s_DQ[NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ((double*)(s_DQ+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
DD(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
u += B(qx,dx) * DD(dx,dy);
|
||||
}
|
||||
DQ(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ(dy,qx) * B(qy,dy);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,e) = u; }
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,c,e) = u; }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Eval3D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, NE):
|
||||
Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[MDQ*MDQ*MDQ];
|
||||
DeviceTensor<3,double> DDD(sm0, MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ(sm1, MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ(sm0, MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
DDD(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
u += B(qx,dx) * DDD(dx,dy,dz);
|
||||
}
|
||||
DDQ(dz,dy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ(dz,qy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ(dz,qy,qx) * B(qz,dz);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,qz,e) = u; }
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,qz,c,e) = u; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,110 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_eval.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Values<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_val.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2133: return Eval2D<L,1,3,3>(NE,B,X,Y);
|
||||
case 0x2124: return Eval2D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x2132: return Eval2D<L,1,3,2>(NE,B,X,Y);
|
||||
case 0x2134: return Eval2D<L,1,3,4>(NE,B,X,Y);
|
||||
case 0x2143: return Eval2D<L,1,4,3>(NE,B,X,Y);
|
||||
case 0x2144: return Eval2D<L,1,4,4>(NE,B,X,Y);
|
||||
|
||||
case 0x2222: return Eval2D<L,2,2,2>(NE,B,X,Y);
|
||||
case 0x2223: return Eval2D<L,2,2,3>(NE,B,X,Y);
|
||||
case 0x2224: return Eval2D<L,2,2,4>(NE,B,X,Y);
|
||||
case 0x2225: return Eval2D<L,2,2,5>(NE,B,X,Y);
|
||||
case 0x2226: return Eval2D<L,2,2,6>(NE,B,X,Y);
|
||||
case 0x2233: return Eval2D<L,2,3,3>(NE,B,X,Y);
|
||||
case 0x2234: return Eval2D<L,2,3,4>(NE,B,X,Y);
|
||||
case 0x2236: return Eval2D<L,2,3,6>(NE,B,X,Y);
|
||||
case 0x2243: return Eval2D<L,2,4,3>(NE,B,X,Y);
|
||||
case 0x2244: return Eval2D<L,2,4,4>(NE,B,X,Y);
|
||||
case 0x2245: return Eval2D<L,2,4,5>(NE,B,X,Y);
|
||||
case 0x2246: return Eval2D<L,2,4,6>(NE,B,X,Y);
|
||||
case 0x2247: return Eval2D<L,2,4,7>(NE,B,X,Y);
|
||||
case 0x2256: return Eval2D<L,2,5,6>(NE,B,X,Y);
|
||||
|
||||
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x3133: return Eval3D<L,1,3,3>(NE,B,X,Y);
|
||||
case 0x3134: return Eval3D<L,1,3,4>(NE,B,X,Y);
|
||||
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
|
||||
case 0x3143: return Eval3D<L,1,4,3>(NE,B,X,Y);
|
||||
case 0x3144: return Eval3D<L,1,4,4>(NE,B,X,Y);
|
||||
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
|
||||
|
||||
case 0x3222: return Eval3D<L,2,2,2>(NE,B,X,Y);
|
||||
case 0x3223: return Eval3D<L,2,2,3>(NE,B,X,Y);
|
||||
case 0x3234: return Eval3D<L,2,3,4>(NE,B,X,Y);
|
||||
|
||||
case 0x3323: return Eval3D<L,3,2,3>(NE,B,X,Y);
|
||||
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
|
||||
case 0x3325: return Eval3D<L,3,2,5>(NE,B,X,Y);
|
||||
case 0x3326: return Eval3D<L,3,2,6>(NE,B,X,Y);
|
||||
case 0x3333: return Eval3D<L,3,3,3>(NE,B,X,Y);
|
||||
case 0x3334: return Eval3D<L,3,3,4>(NE,B,X,Y);
|
||||
case 0x3335: return Eval3D<L,3,3,5>(NE,B,X,Y);
|
||||
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
|
||||
case 0x3343: return Eval3D<L,3,4,3>(NE,B,X,Y);
|
||||
case 0x3344: return Eval3D<L,3,4,4>(NE,B,X,Y);
|
||||
case 0x3346: return Eval3D<L,3,4,6>(NE,B,X,Y);
|
||||
case 0x3347: return Eval3D<L,3,4,7>(NE,B,X,Y);
|
||||
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,79 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_eval.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Values<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_val.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2124: return Eval2D<L,1,2,4,8>(NE,B,X,Y);
|
||||
case 0x2136: return Eval2D<L,1,3,6,4>(NE,B,X,Y);
|
||||
case 0x2148: return Eval2D<L,1,4,8,2>(NE,B,X,Y);
|
||||
|
||||
case 0x2224: return Eval2D<L,2,2,4,8>(NE,B,X,Y);
|
||||
case 0x2234: return Eval2D<L,2,3,4,8>(NE,B,X,Y);
|
||||
case 0x2236: return Eval2D<L,2,3,6,4>(NE,B,X,Y);
|
||||
case 0x2248: return Eval2D<L,2,4,8,2>(NE,B,X,Y);
|
||||
|
||||
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
|
||||
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
|
||||
|
||||
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
|
||||
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
|
||||
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -495,8 +495,9 @@ void FaceQuadratureInterpolator::Mult(
|
||||
}
|
||||
}
|
||||
|
||||
void FaceQuadratureInterpolator::Values(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
|
||||
void FaceQuadratureInterpolator::Values(const Vector &e_vec,
|
||||
Vector &q_val) const
|
||||
{
|
||||
Vector q_der, q_det, q_nor;
|
||||
Mult(e_vec, VALUES, q_val, q_der, q_det, q_nor);
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Grad2D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
|
||||
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
MFEM_SHARED double s_G[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
DeviceTensor<2,double> G(s_G, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_X[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double s_DQ[2][NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ0((double*)(s_DQ[0]+tidz), MD1, MQ1);
|
||||
DeviceTensor<2,double> DQ1((double*)(s_DQ[1]+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
X(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DQ0(dy,qx) = u;
|
||||
DQ1(dy,qx) = v;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ1(dy,qx) * B(qy,dy);
|
||||
v += DQ0(dy,qx) * G(qy,dy);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,c,0,e) = u;
|
||||
y(qx,qy,c,1,e) = v;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,e) = u;
|
||||
y(c,1,qx,qy,e) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Grad3D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
|
||||
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
MFEM_SHARED double s_G[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
DeviceTensor<2,double> G(s_G, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
|
||||
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
X(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy,dz);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DDQ0(dz,dy,qx) = u;
|
||||
DDQ1(dz,dy,qx) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ1(dz,dy,qx) * B(qy,dy);
|
||||
v += DDQ0(dz,dy,qx) * G(qy,dy);
|
||||
w += DDQ0(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ0(dz,qy,qx) = u;
|
||||
DQQ1(dz,qy,qx) = v;
|
||||
DQQ2(dz,qy,qx) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ0(dz,qy,qx) * B(qz,dz);
|
||||
v += DQQ1(dz,qy,qx) * B(qz,dz);
|
||||
w += DQQ2(dz,qy,qx) * G(qz,dz);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,qz,c,0,e) = u;
|
||||
y(qx,qy,qz,c,1,e) = v;
|
||||
y(qx,qy,qz,c,2,e) = w;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,qz,e) = u;
|
||||
y(c,1,qx,qy,qz,e) = v;
|
||||
y(c,2,qx,qy,qz,e) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,109 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Derivatives<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2133: return Grad2D<L,1,3,3,16>(NE,B,G,X,Y);
|
||||
case 0x2134: return Grad2D<L,1,3,4,16>(NE,B,G,X,Y);
|
||||
case 0x2143: return Grad2D<L,1,4,3,16>(NE,B,G,X,Y);
|
||||
case 0x2144: return Grad2D<L,1,4,4,16>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2222: return Grad2D<L,2,2,2,16>(NE,B,G,X,Y);
|
||||
case 0x2223: return Grad2D<L,2,2,3,8>(NE,B,G,X,Y);
|
||||
case 0x2224: return Grad2D<L,2,2,4,4>(NE,B,G,X,Y);
|
||||
case 0x2225: return Grad2D<L,2,2,5,4>(NE,B,G,X,Y);
|
||||
case 0x2226: return Grad2D<L,2,2,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2233: return Grad2D<L,2,3,3,2>(NE,B,G,X,Y);
|
||||
case 0x2234: return Grad2D<L,2,3,4,4>(NE,B,G,X,Y);
|
||||
case 0x2243: return Grad2D<L,2,4,3,4>(NE,B,G,X,Y);
|
||||
case 0x2236: return Grad2D<L,2,3,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2244: return Grad2D<L,2,4,4,2>(NE,B,G,X,Y);
|
||||
case 0x2245: return Grad2D<L,2,4,5,2>(NE,B,G,X,Y);
|
||||
case 0x2246: return Grad2D<L,2,4,6,2>(NE,B,G,X,Y);
|
||||
case 0x2247: return Grad2D<L,2,4,7,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2256: return Grad2D<L,2,5,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3124: return Grad3D<L,1,2,4>(NE,B,G,X,Y);
|
||||
case 0x3133: return Grad3D<L,1,3,3>(NE,B,G,X,Y);
|
||||
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
|
||||
case 0x3136: return Grad3D<L,1,3,6>(NE,B,G,X,Y);
|
||||
case 0x3144: return Grad3D<L,1,4,4>(NE,B,G,X,Y);
|
||||
case 0x3148: return Grad3D<L,1,4,8>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3323: return Grad3D<L,3,2,3>(NE,B,G,X,Y);
|
||||
case 0x3324: return Grad3D<L,3,2,4>(NE,B,G,X,Y);
|
||||
case 0x3325: return Grad3D<L,3,2,5>(NE,B,G,X,Y);
|
||||
case 0x3326: return Grad3D<L,3,2,6>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3333: return Grad3D<L,3,3,3>(NE,B,G,X,Y);
|
||||
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
|
||||
case 0x3335: return Grad3D<L,3,3,5>(NE,B,G,X,Y);
|
||||
case 0x3336: return Grad3D<L,3,3,6>(NE,B,G,X,Y);
|
||||
case 0x3344: return Grad3D<L,3,4,4>(NE,B,G,X,Y);
|
||||
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
|
||||
case 0x3347: return Grad3D<L,3,4,7>(NE,B,G,X,Y);
|
||||
case 0x3348: return Grad3D<L,3,4,8>(NE,B,G,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2)
|
||||
{
|
||||
return Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
return Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,76 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Derivatives<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2134: return Grad2D<L,1,3,4,8>(NE,B,G,X,Y);
|
||||
case 0x2146: return Grad2D<L,1,4,6,4>(NE,B,G,X,Y);
|
||||
case 0x2158: return Grad2D<L,1,5,8,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2234: return Grad2D<L,2,3,4,8>(NE,B,G,X,Y);
|
||||
case 0x2246: return Grad2D<L,2,4,6,4>(NE,B,G,X,Y);
|
||||
case 0x2258: return Grad2D<L,2,5,8,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
|
||||
case 0x3146: return Grad3D<L,1,4,6>(NE,B,G,X,Y);
|
||||
case 0x3158: return Grad3D<L,1,5,8>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
|
||||
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
|
||||
case 0x3358: return Grad3D<L,3,5,8>(NE,B,G,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,303 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void PhysGrad2D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *j_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, 2, 2, NE);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
|
||||
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1][MD1];
|
||||
MFEM_SHARED double s_G[MQ1][MD1];
|
||||
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
|
||||
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_X[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double sm[2][NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ0((double*)(sm[0]+tidz), MD1, MQ1);
|
||||
DeviceTensor<2,double> DQ1((double*)(sm[1]+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
X(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DQ0(dy,qx) = u;
|
||||
DQ1(dy,qx) = v;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ1(dy,qx) * B(qy,dy);
|
||||
v += DQ0(dy,qx) * G(qy,dy);
|
||||
}
|
||||
double Jloc[4], Jinv[4];
|
||||
Jloc[0] = j(qx,qy,0,0,e);
|
||||
Jloc[1] = j(qx,qy,1,0,e);
|
||||
Jloc[2] = j(qx,qy,0,1,e);
|
||||
Jloc[3] = j(qx,qy,1,1,e);
|
||||
kernels::CalcInverse<2>(Jloc, Jinv);
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,e) = Jinv[0]*u + Jinv[1]*v;
|
||||
y(c,1,qx,qy,e) = Jinv[2]*u + Jinv[3]*v;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,c,0,e) = Jinv[0]*u + Jinv[1]*v;
|
||||
y(qx,qy,c,1,e) = Jinv[2]*u + Jinv[3]*v;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D = 0, int MAX_Q = 0>
|
||||
static void PhysGrad3D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *j_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, Q1D, 3, 3, NE);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
|
||||
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1][MD1];
|
||||
MFEM_SHARED double s_G[MQ1][MD1];
|
||||
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
|
||||
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
|
||||
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
X(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double coords = X(dx,dy,dz);
|
||||
u += coords * B(qx,dx);
|
||||
v += coords * G(qx,dx);
|
||||
}
|
||||
DDQ0(dz,dy,qx) = u;
|
||||
DDQ1(dz,dy,qx) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ1(dz,dy,qx) * B(qy,dy);
|
||||
v += DDQ0(dz,dy,qx) * G(qy,dy);
|
||||
w += DDQ0(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ0(dz,qy,qx) = u;
|
||||
DQQ1(dz,qy,qx) = v;
|
||||
DQQ2(dz,qy,qx) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ0(dz,qy,qx) * B(qz,dz);
|
||||
v += DQQ1(dz,qy,qx) * B(qz,dz);
|
||||
w += DQQ2(dz,qy,qx) * G(qz,dz);
|
||||
}
|
||||
double Jloc[9], Jinv[9];
|
||||
for (int col = 0; col < 3; col++)
|
||||
{
|
||||
for (int row = 0; row < 3; row++)
|
||||
{
|
||||
Jloc[row+3*col] = j(qx,qy,qz,row,col,e);
|
||||
}
|
||||
}
|
||||
kernels::CalcInverse<3>(Jloc, Jinv);
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,qz,c,0,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
|
||||
y(qx,qy,qz,c,1,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
|
||||
y(qx,qy,qz,c,2,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,qz,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
|
||||
y(c,1,qx,qy,qz,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
|
||||
y(c,2,qx,qy,qz,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,110 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad_phys.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
Mesh *mesh = fespace->GetMesh();
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
|
||||
const GeometricFactors *geom =
|
||||
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *J = geom->J.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x133: return PhysGrad2D<L,1,3,3,8>(NE,B,G,J,X,Y);
|
||||
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE,B,G,J,X,Y);
|
||||
case 0x143: return PhysGrad2D<L,1,4,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x144: return PhysGrad2D<L,1,4,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE,B,G,J,X,Y);
|
||||
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE,B,G,J,X,Y);
|
||||
|
||||
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE,B,G,J,X,Y);
|
||||
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE,B,G,J,X,Y);
|
||||
case 0x243: return PhysGrad2D<L,2,4,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x244: return PhysGrad2D<L,2,4,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE,B,G,J,X,Y);
|
||||
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE,B,G,J,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = MAX_D1D;
|
||||
constexpr int MQ = MAX_Q1D;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x133: return PhysGrad3D<L,1,3,3>(NE,B,G,J,X,Y);
|
||||
case 0x134: return PhysGrad3D<L,1,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x144: return PhysGrad3D<L,1,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x146: return PhysGrad3D<L,1,4,6>(NE,B,G,J,X,Y);
|
||||
case 0x158: return PhysGrad3D<L,1,5,8>(NE,B,G,J,X,Y);
|
||||
|
||||
case 0x333: return PhysGrad3D<L,3,3,3>(NE,B,G,J,X,Y);
|
||||
case 0x334: return PhysGrad3D<L,3,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x344: return PhysGrad3D<L,3,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x346: return PhysGrad3D<L,3,4,6>(NE,B,G,J,X,Y);
|
||||
case 0x358: return PhysGrad3D<L,3,5,8>(NE,B,G,J,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = 8;
|
||||
constexpr int MQ = 8;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,101 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad_phys.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
Mesh *mesh = fespace->GetMesh();
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
|
||||
const GeometricFactors *geom =
|
||||
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *J = geom->J.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE, B, G, J, X, Y);
|
||||
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE, B, G, J, X, Y);
|
||||
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE, B, G, J, X, Y);
|
||||
|
||||
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE, B, G, J, X, Y);
|
||||
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE, B, G, J, X, Y);
|
||||
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE, B, G, J, X, Y);
|
||||
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE, B, G, J, X, Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = MAX_D1D;
|
||||
constexpr int MQ = MAX_Q1D;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x134: return PhysGrad3D<L,1,3,4>(NE, B, G, J, X, Y);
|
||||
case 0x146: return PhysGrad3D<L,1,4,6>(NE, B, G, J, X, Y);
|
||||
case 0x158: return PhysGrad3D<L,1,5,8>(NE, B, G, J, X, Y);
|
||||
|
||||
case 0x334: return PhysGrad3D<L,3,3,4>(NE, B, G, J, X, Y);
|
||||
case 0x346: return PhysGrad3D<L,3,4,6>(NE, B, G, J, X, Y);
|
||||
case 0x358: return PhysGrad3D<L,3,5,8>(NE, B, G, J, X, Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = 8;
|
||||
constexpr int MQ = 8;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
+421
-52
@@ -17,55 +17,6 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
|
||||
: ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
|
||||
ndofs(fes.GetNDofs())
|
||||
{
|
||||
height = vdim*ne*ndof;
|
||||
width = vdim*ne*ndof;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_y = Reshape(y.Write(), nd, vd, ne);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), nd, vd, ne);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
|
||||
ElementDofOrdering e_ordering)
|
||||
: fes(f),
|
||||
@@ -281,7 +232,309 @@ void ElementRestriction::BooleanMask(Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
/// Return the face degrees of freedom returned in Lexicographic order.
|
||||
void ElementRestriction::FillSparseMatrix(const Vector &mat_ea,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
mat.GetMemoryI().New(mat.Height()+1, mat.GetMemoryI().GetMemoryType());
|
||||
const int nnz = FillI(mat);
|
||||
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
|
||||
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
|
||||
FillJAndData(mat_ea, mat);
|
||||
}
|
||||
|
||||
template <int MaxNbNbr>
|
||||
static MFEM_HOST_DEVICE int GetMinElt(const int *my_elts, const int nbElts,
|
||||
const int *nbr_elts, const int nbrNbElts)
|
||||
{
|
||||
// Building the intersection
|
||||
int inter[MaxNbNbr];
|
||||
int cpt = 0;
|
||||
for (int i = 0; i < nbElts; i++)
|
||||
{
|
||||
const int e_i = my_elts[i];
|
||||
for (int j = 0; j < nbrNbElts; j++)
|
||||
{
|
||||
if (e_i==nbr_elts[j])
|
||||
{
|
||||
inter[cpt] = e_i;
|
||||
cpt++;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finding the minimum
|
||||
int min = inter[0];
|
||||
for (int i = 1; i < cpt; i++)
|
||||
{
|
||||
if (inter[i] < min)
|
||||
{
|
||||
min = inter[i];
|
||||
}
|
||||
}
|
||||
return min;
|
||||
}
|
||||
|
||||
/** Returns the index where a non-zero entry should be added and increment the
|
||||
number of non-zeros for the row i_L. */
|
||||
static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
|
||||
{
|
||||
int ind = AtomicAdd(I[i_L],1);
|
||||
return ind;
|
||||
}
|
||||
|
||||
int ElementRestriction::FillI(SparseMatrix &mat) const
|
||||
{
|
||||
static constexpr int Max = MaxNbNbr;
|
||||
const int all_dofs = ndofs;
|
||||
const int vd = vdim;
|
||||
const int elt_dofs = dof;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = indices.Read();
|
||||
auto d_gatherMap = gatherMap.Read();
|
||||
MFEM_FORALL(i_L, vd*all_dofs+1,
|
||||
{
|
||||
I[i_L] = 0;
|
||||
});
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < elt_dofs; i++)
|
||||
{
|
||||
int i_elts[Max];
|
||||
const int i_E = e*elt_dofs + i;
|
||||
const int i_L = d_gatherMap[i_E];
|
||||
const int i_offset = d_offsets[i_L];
|
||||
const int i_nextOffset = d_offsets[i_L+1];
|
||||
const int i_nbElts = i_nextOffset - i_offset;
|
||||
for (int e_i = 0; e_i < i_nbElts; ++e_i)
|
||||
{
|
||||
const int i_E = d_indices[i_offset+e_i];
|
||||
i_elts[e_i] = i_E/elt_dofs;
|
||||
}
|
||||
for (int j = 0; j < elt_dofs; j++)
|
||||
{
|
||||
const int j_E = e*elt_dofs + j;
|
||||
const int j_L = d_gatherMap[j_E];
|
||||
const int j_offset = d_offsets[j_L];
|
||||
const int j_nextOffset = d_offsets[j_L+1];
|
||||
const int j_nbElts = j_nextOffset - j_offset;
|
||||
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
|
||||
{
|
||||
GetAndIncrementNnzIndex(i_L, I);
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
for (int e_j = 0; e_j < j_nbElts; ++e_j)
|
||||
{
|
||||
const int j_E = d_indices[j_offset+e_j];
|
||||
const int elt = j_E/elt_dofs;
|
||||
j_elts[e_j] = elt;
|
||||
}
|
||||
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
|
||||
if (e == min_e) // add the nnz only once
|
||||
{
|
||||
GetAndIncrementNnzIndex(i_L, I);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
// We need to sum the entries of I, we do it on CPU as it is very sequential.
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
const int nTdofs = vd*all_dofs;
|
||||
int sum = 0;
|
||||
for (int i = 0; i < nTdofs; i++)
|
||||
{
|
||||
const int nnz = h_I[i];
|
||||
h_I[i] = sum;
|
||||
sum+=nnz;
|
||||
}
|
||||
h_I[nTdofs] = sum;
|
||||
// We return the number of nnz
|
||||
return h_I[nTdofs];
|
||||
}
|
||||
|
||||
void ElementRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
static constexpr int Max = MaxNbNbr;
|
||||
const int all_dofs = ndofs;
|
||||
const int vd = vdim;
|
||||
const int elt_dofs = dof;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = indices.Read();
|
||||
auto d_gatherMap = gatherMap.Read();
|
||||
auto mat_ea = Reshape(ea_data.Read(), elt_dofs, elt_dofs, ne);
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < elt_dofs; i++)
|
||||
{
|
||||
int i_elts[Max];
|
||||
int i_B[Max];
|
||||
const int i_E = e*elt_dofs + i;
|
||||
const int i_L = d_gatherMap[i_E];
|
||||
const int i_offset = d_offsets[i_L];
|
||||
const int i_nextOffset = d_offsets[i_L+1];
|
||||
const int i_nbElts = i_nextOffset - i_offset;
|
||||
for (int e_i = 0; e_i < i_nbElts; ++e_i)
|
||||
{
|
||||
const int i_E = d_indices[i_offset+e_i];
|
||||
i_elts[e_i] = i_E/elt_dofs;
|
||||
i_B[e_i] = i_E%elt_dofs;
|
||||
}
|
||||
for (int j = 0; j < elt_dofs; j++)
|
||||
{
|
||||
const int j_E = e*elt_dofs + j;
|
||||
const int j_L = d_gatherMap[j_E];
|
||||
const int j_offset = d_offsets[j_L];
|
||||
const int j_nextOffset = d_offsets[j_L+1];
|
||||
const int j_nbElts = j_nextOffset - j_offset;
|
||||
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
|
||||
{
|
||||
const int nnz = GetAndIncrementNnzIndex(i_L, I);
|
||||
J[nnz] = j_L;
|
||||
Data[nnz] = mat_ea(j,i,e);
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
int j_B[Max];
|
||||
for (int e_j = 0; e_j < j_nbElts; ++e_j)
|
||||
{
|
||||
const int j_E = d_indices[j_offset+e_j];
|
||||
const int elt = j_E/elt_dofs;
|
||||
j_elts[e_j] = elt;
|
||||
j_B[e_j] = j_E%elt_dofs;
|
||||
}
|
||||
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
|
||||
if (e == min_e) // add the nnz only once
|
||||
{
|
||||
double val = 0.0;
|
||||
for (int i = 0; i < i_nbElts; i++)
|
||||
{
|
||||
const int e_i = i_elts[i];
|
||||
const int i_Bloc = i_B[i];
|
||||
for (int j = 0; j < j_nbElts; j++)
|
||||
{
|
||||
const int e_j = j_elts[j];
|
||||
const int j_Bloc = j_B[j];
|
||||
if (e_i == e_j)
|
||||
{
|
||||
val += mat_ea(j_Bloc, i_Bloc, e_i);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int nnz = GetAndIncrementNnzIndex(i_L, I);
|
||||
J[nnz] = j_L;
|
||||
Data[nnz] = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
// We need to shift again the entries of I, we do it on CPU as it is very
|
||||
// sequential.
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
const int size = vd*all_dofs;
|
||||
for (int i = 0; i < size; i++)
|
||||
{
|
||||
h_I[size-i] = h_I[size-(i+1)];
|
||||
}
|
||||
h_I[0] = 0;
|
||||
}
|
||||
|
||||
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
|
||||
: ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
|
||||
ndofs(fes.GetNDofs())
|
||||
{
|
||||
height = vdim*ne*ndof;
|
||||
width = vdim*ne*ndof;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_y = Reshape(y.Write(), nd, vd, ne);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), nd, vd, ne);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::FillI(SparseMatrix &mat) const
|
||||
{
|
||||
const int elem_dofs = ndof;
|
||||
const int vd = vdim;
|
||||
auto I = mat.WriteI();
|
||||
MFEM_FORALL(dof, ne*elem_dofs*vd,
|
||||
{
|
||||
I[dof] = elem_dofs;
|
||||
});
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
|
||||
{
|
||||
int val = AtomicAdd(I[iE],dofs);
|
||||
return val;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
const int elem_dofs = ndof;
|
||||
const int vd = vdim;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto mat_ea = Reshape(ea_data.Read(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(iE, ne*elem_dofs*vd,
|
||||
{
|
||||
const int offset = AddNnz(iE,I,elem_dofs);
|
||||
const int e = iE/elem_dofs;
|
||||
const int i = iE%elem_dofs;
|
||||
for (int j = 0; j < elem_dofs; j++)
|
||||
{
|
||||
J[offset+j] = e*elem_dofs+j;
|
||||
Data[offset+j] = mat_ea(j,i,e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Return the face degrees of freedom returned in Lexicographic order.
|
||||
void GetFaceDofs(const int dim, const int face_id,
|
||||
const int dof1d, Array<int> &faceMap)
|
||||
{
|
||||
@@ -700,7 +953,7 @@ static int PermuteFace3D(const int face_id1, const int face_id2,
|
||||
return ToLexOrdering3D(face_id2, size1d, new_i, new_j);
|
||||
}
|
||||
|
||||
/// Permute dofs or quads on a face for e2 to match with the ordering of e1
|
||||
// Permute dofs or quads on a face for e2 to match with the ordering of e1
|
||||
int PermuteFaceL2(const int dim, const int face_id1,
|
||||
const int face_id2, const int orientation,
|
||||
const int size1d, const int index)
|
||||
@@ -720,23 +973,32 @@ int PermuteFaceL2(const int dim, const int face_id1,
|
||||
}
|
||||
|
||||
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
|
||||
const ElementDofOrdering e_ordering,
|
||||
const FaceType type,
|
||||
const L2FaceValues m)
|
||||
: fes(fes),
|
||||
nf(fes.GetNFbyType(type)),
|
||||
ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndofs(fes.GetNDofs()),
|
||||
dof(nf > 0 ?
|
||||
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
|
||||
: 0),
|
||||
elemDofs(fes.GetFE(0)->GetDof()),
|
||||
m(m),
|
||||
nfdofs(nf*dof),
|
||||
scatter_indices1(nf*dof),
|
||||
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
|
||||
offsets(ndofs+1),
|
||||
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
|
||||
{
|
||||
}
|
||||
|
||||
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
|
||||
const ElementDofOrdering e_ordering,
|
||||
const FaceType type,
|
||||
const L2FaceValues m)
|
||||
: L2FaceRestriction(fes, type, m)
|
||||
{
|
||||
// If fespace == L2
|
||||
const FiniteElement *fe = fes.GetFE(0);
|
||||
@@ -1034,6 +1296,113 @@ void L2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
void L2FaceRestriction::FillI(SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int iE1 = d_indices1[fdof];
|
||||
const int iE2 = d_indices2[fdof];
|
||||
AddNnz(iE1,I,face_dofs);
|
||||
AddNnz(iE2,I,face_dofs);
|
||||
});
|
||||
}
|
||||
|
||||
void L2FaceRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
const int offset1 = AddNnz(iE1,I,face_dofs);
|
||||
const int offset2 = AddNnz(iE2,I,face_dofs);
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
J[offset2+jF] = jE1;
|
||||
J[offset1+jF] = jE2;
|
||||
Data[offset2+jF] = mat_fea(jF,iF,0,f);
|
||||
Data[offset1+jF] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2FaceRestriction::AddFaceMatricesToElementMatrices(Vector &fea_data,
|
||||
Vector &ea_data) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int elem_dofs = elemDofs;
|
||||
const int NE = ne;
|
||||
if (m==L2FaceValues::DoubleValued)
|
||||
{
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
const int e1 = d_indices1[f*face_dofs]/elem_dofs;
|
||||
const int e2 = d_indices2[f*face_dofs]/elem_dofs;
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jB1 = d_indices1[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iB1 = d_indices1[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iB1,jB1,e1), mat_fea(i,j,0,f));
|
||||
}
|
||||
}
|
||||
if (e2 < NE)
|
||||
{
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jB2 = d_indices2[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iB2 = d_indices2[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iB2,jB2,e2), mat_fea(i,j,1,f));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
auto d_indices = scatter_indices1.Read();
|
||||
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, nf);
|
||||
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
const int e = d_indices[f*face_dofs]/elem_dofs;
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jE = d_indices[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iE = d_indices[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iE,jE,e), mat_fea(i,j,f));
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
int ToLexOrdering(const int dim, const int face_id, const int size1d,
|
||||
const int index)
|
||||
{
|
||||
|
||||
+39
-1
@@ -30,6 +30,11 @@ enum class L2FaceValues : bool {SingleValued, DoubleValued};
|
||||
objects, see FiniteElementSpace::GetElementRestriction(). */
|
||||
class ElementRestriction : public Operator
|
||||
{
|
||||
private:
|
||||
/** This number defines the maximum number of elements any dof can belong to
|
||||
for the FillSparseMatrix method. */
|
||||
static const int MaxNbNbr = 16;
|
||||
|
||||
protected:
|
||||
const FiniteElementSpace &fes;
|
||||
const int ne;
|
||||
@@ -59,6 +64,16 @@ public:
|
||||
emulate SetSubVector and its transpose on GPUs. This method is running on
|
||||
the host, since the `processed` array requires a large shared memory. */
|
||||
void BooleanMask(Vector& y) const;
|
||||
|
||||
/// Fill a Sparse Matrix with Element Matrices.
|
||||
void FillSparseMatrix(const Vector &mat_ea, SparseMatrix &mat) const;
|
||||
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ElementRestriction. */
|
||||
int FillI(SparseMatrix &mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this ElementRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
|
||||
};
|
||||
|
||||
/// Operator that converts L2 FiniteElementSpace L-vectors to E-vectors.
|
||||
@@ -77,6 +92,12 @@ public:
|
||||
L2ElementRestriction(const FiniteElementSpace&);
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ElementRestriction. */
|
||||
void FillI(SparseMatrix &mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this L2FaceRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
|
||||
};
|
||||
|
||||
/// Operator that extracts Face degrees of freedom.
|
||||
@@ -111,10 +132,12 @@ class L2FaceRestriction : public Operator
|
||||
protected:
|
||||
const FiniteElementSpace &fes;
|
||||
const int nf;
|
||||
const int ne;
|
||||
const int vdim;
|
||||
const bool byvdim;
|
||||
const int ndofs;
|
||||
const int dof;
|
||||
const int elemDofs;
|
||||
const L2FaceValues m;
|
||||
const int nfdofs;
|
||||
Array<int> scatter_indices1;
|
||||
@@ -122,12 +145,27 @@ protected:
|
||||
Array<int> offsets;
|
||||
Array<int> gather_indices;
|
||||
|
||||
L2FaceRestriction(const FiniteElementSpace&,
|
||||
const FaceType,
|
||||
const L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
|
||||
public:
|
||||
L2FaceRestriction(const FiniteElementSpace&, const ElementDofOrdering,
|
||||
const FaceType,
|
||||
const L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
virtual void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this L2FaceRestriction. */
|
||||
virtual void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this L2FaceRestriction, and the values of ea_data. */
|
||||
virtual void FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const;
|
||||
/// This methods adds the DG face matrices to the element matrices.
|
||||
void AddFaceMatricesToElementMatrices(Vector &fea_data,
|
||||
Vector &ea_data) const;
|
||||
};
|
||||
|
||||
// Return the face degrees of freedom returned in Lexicographic order.
|
||||
|
||||
+573
-75
@@ -13,6 +13,7 @@
|
||||
#include "linearform.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
#include "tmop_tools.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -441,8 +442,8 @@ void TMOP_Metric_058::AssembleH(const DenseMatrix &Jpt,
|
||||
double TMOP_Metric_077::EvalW(const DenseMatrix &Jpt) const
|
||||
{
|
||||
ie.SetJacobian(Jpt.GetData());
|
||||
const double I2 = ie.Get_I2b();
|
||||
return 0.5*(I2*I2 + 1./(I2*I2) - 2.);
|
||||
const double I2b = ie.Get_I2b();
|
||||
return 0.5*(I2b*I2b + 1./(I2b*I2b) - 2.);
|
||||
}
|
||||
|
||||
void TMOP_Metric_077::EvalP(const DenseMatrix &Jpt, DenseMatrix &P) const
|
||||
@@ -927,9 +928,20 @@ void TargetConstructor::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void TargetConstructor::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
|
||||
|
||||
// TODO: Compute derivative for targets with GIVEN_SHAPE or/and GIVEN_SIZE
|
||||
for (int i = 0; i < Tpr.GetFE()->GetDim()*ir.GetNPoints(); i++) { dJtr(i) = 0.; }
|
||||
}
|
||||
|
||||
void AnalyticAdaptTC::SetAnalyticTargetSpec(Coefficient *sspec,
|
||||
VectorCoefficient *vspec,
|
||||
MatrixCoefficient *mspec)
|
||||
TMOPMatrixCoefficient *mspec)
|
||||
{
|
||||
scalar_tspec = sspec;
|
||||
vector_tspec = vspec;
|
||||
@@ -970,6 +982,39 @@ void AnalyticAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void AnalyticAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
const FiniteElement *fe = Tpr.GetFE();
|
||||
DenseMatrix point_mat;
|
||||
point_mat.UseExternalData(elfun.GetData(), fe->GetDof(), fe->GetDim());
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
case GIVEN_FULL:
|
||||
{
|
||||
MFEM_VERIFY(matrix_tspec != NULL,
|
||||
"Target type GIVEN_FULL requires a TMOPMatrixCoefficient.");
|
||||
|
||||
for (int d = 0; d < fe->GetDim(); d++)
|
||||
{
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
Tpr.SetIntPoint(&ip);
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
matrix_tspec->EvalGrad(dJtr_i, Tpr, ip, d);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MFEM_ABORT("Incompatible target type for analytic adaptation!");
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
void DiscreteAdaptTC::FinalizeParDiscreteTargetSpec(const ParGridFunction
|
||||
&tspec_)
|
||||
@@ -994,11 +1039,10 @@ void DiscreteAdaptTC::SetTspecAtIndex(int idx, const ParGridFunction &tspec_)
|
||||
{
|
||||
const int vdim = tspec_.FESpace()->GetVDim(),
|
||||
dof_cnt = tspec_.Size()/vdim;
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+idx*dof_cnt) = tspec_(i);
|
||||
}
|
||||
|
||||
const auto tspec__d = tspec_.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
const int offset = idx*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
FinalizeParDiscreteTargetSpec(tspec_);
|
||||
}
|
||||
|
||||
@@ -1058,34 +1102,33 @@ void DiscreteAdaptTC::SetDiscreteTargetBase(const GridFunction &tspec_)
|
||||
// make a copy of tspec->tspec_temp, increase its size, and
|
||||
// copy data from tspec_temp -> tspec, then add new entries
|
||||
Vector tspec_temp = tspec;
|
||||
tspec.UseDevice(true);
|
||||
tspec_sav.UseDevice(true);
|
||||
tspec.SetSize(ncomp*dof_cnt);
|
||||
|
||||
for (int i = 0; i < tspec_temp.Size(); i++)
|
||||
{
|
||||
tspec(i) = tspec_temp(i);
|
||||
}
|
||||
const auto tspec_temp_d = tspec_temp.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
MFEM_FORALL(i, tspec_temp.Size(), tspec_d[i] = tspec_temp_d[i];);
|
||||
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+(ncomp-vdim)*dof_cnt) = tspec_(i);
|
||||
}
|
||||
const auto tspec__d = tspec_.Read();
|
||||
const int offset = (ncomp-vdim)*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::SetTspecAtIndex(int idx, const GridFunction &tspec_)
|
||||
{
|
||||
const int vdim = tspec_.FESpace()->GetVDim(),
|
||||
dof_cnt = tspec_.Size()/vdim;
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+idx*dof_cnt) = tspec_(i);
|
||||
}
|
||||
|
||||
const auto tspec__d = tspec_.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
const int offset = idx*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
FinalizeSerialDiscreteTargetSpec();
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::SetSerialDiscreteTargetSize(const GridFunction &tspec_)
|
||||
{
|
||||
|
||||
if (sizeidx > -1) { SetTspecAtIndex(sizeidx, tspec_); return; }
|
||||
sizeidx = ncomp;
|
||||
SetDiscreteTargetBase(tspec_);
|
||||
@@ -1168,7 +1211,7 @@ void DiscreteAdaptTC::UpdateTargetSpecificationAtNode(const FiniteElement &el,
|
||||
|
||||
Array<int> dofs;
|
||||
tspec_fes->GetElementDofs(T.ElementNo, dofs);
|
||||
const int cnt = tspec.Size()/ncomp; //dofs per scalar-field
|
||||
const int cnt = tspec.Size()/ncomp; // dofs per scalar-field
|
||||
|
||||
for (int i = 0; i < ncomp; i++)
|
||||
{
|
||||
@@ -1196,6 +1239,9 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
DenseTensor &Jtr) const
|
||||
{
|
||||
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
|
||||
const int dim = fe.GetDim(),
|
||||
nqp = ir.GetNPoints();
|
||||
Jtrcomp.SetSize(dim, dim, 4*nqp);
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
@@ -1205,36 +1251,44 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
const DenseMatrix &Wideal =
|
||||
Geometries.GetGeomToPerfGeomJac(fe.GetGeomType());
|
||||
const int dim = Wideal.Height(),
|
||||
ndofs = tspec_fes->GetFE(0)->GetDof(),
|
||||
ndofs = tspec_fes->GetFE(e_id)->GetDof(),
|
||||
ntspec_dofs = ndofs*ncomp;
|
||||
|
||||
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
|
||||
par_vals_c1, par_vals_c2, par_vals_c3;
|
||||
|
||||
Array<int> dofs;
|
||||
DenseMatrix D_rho(dim), Q_phi(dim), R_theta(dim);
|
||||
tspec_fesv->GetElementVDofs(e_id, dofs);
|
||||
tspec.UseDevice(true);
|
||||
tspec.GetSubVector(dofs, tspec_vals);
|
||||
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
const IntegrationPoint &ip = ir.IntPoint(q);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
Jtr(i) = Wideal; //Initialize to identity
|
||||
|
||||
if (sizeidx != -1) //Set size
|
||||
Jtr(q) = Wideal; // Initialize to identity
|
||||
for (int d = 0; d < 4; d++)
|
||||
{
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(d + 4*q), dim, dim);
|
||||
Jtrcomp_q = Wideal; // Initialize to identity
|
||||
}
|
||||
|
||||
if (sizeidx != -1) // Set size
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
|
||||
const double min_size = par_vals.Min();
|
||||
MFEM_VERIFY(min_size > 0.0,
|
||||
"Non-positive size propagated in the target definition.");
|
||||
const double size = std::max(shape * par_vals, min_size);
|
||||
Jtr(i).Set(std::pow(size, 1.0/dim), Jtr(i));
|
||||
} //Done size
|
||||
Jtr(q).Set(std::pow(size, 1.0/dim), Jtr(q));
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(0 + 4*q), dim, dim);
|
||||
Jtrcomp_q = Jtr(q);
|
||||
} // Done size
|
||||
|
||||
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
|
||||
|
||||
if (aspectratioidx != -1) //Set aspect ratio
|
||||
if (aspectratioidx != -1) // Set aspect ratio
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1262,12 +1316,13 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
D_rho(1,1) = pow(rho2,2./3.);
|
||||
D_rho(2,2) = pow(rho3,2./3.);
|
||||
}
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(1 + 4*q), dim, dim);
|
||||
Jtrcomp_q = D_rho;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(D_rho, Temp, Jtr(q));
|
||||
} // Done aspect ratio
|
||||
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(D_rho, Temp, Jtr(i));
|
||||
} //Done aspect ratio
|
||||
|
||||
if (skewidx != -1) //Set skew
|
||||
if (skewidx != -1) // Set skew
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1303,12 +1358,13 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
|
||||
Q_phi(2,2) = sin(phi13)*sin(chi);
|
||||
}
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*q), dim, dim);
|
||||
Jtrcomp_q = Q_phi;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(Q_phi, Temp, Jtr(q));
|
||||
} // Done skew
|
||||
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(Q_phi, Temp, Jtr(i));
|
||||
} // done skew
|
||||
|
||||
if (orientationidx != -1) //Set orientation
|
||||
if (orientationidx != -1) // Set orientation
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1333,33 +1389,28 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
const double psi = shape * par_vals_c2;
|
||||
const double beta = shape * par_vals_c3;
|
||||
|
||||
DenseMatrix R_tp(dim), R_beta(dim), R_theta(dim);
|
||||
double ct = cos(theta), st = sin(theta),
|
||||
cp = cos(psi), sp = sin(psi);
|
||||
R_tp(0,0) = ct*sp;
|
||||
R_tp(1,0) = st*sp;
|
||||
R_tp(2,0) = cp;
|
||||
cp = cos(psi), sp = sin(psi),
|
||||
cb = cos(beta), sb = sin(beta);
|
||||
|
||||
R_tp(0,1) = -(ct*st*sp*sp)/(1+cp);
|
||||
R_tp(1,1) = cp+(pow(ct,2.)*pow(sp,2.))/(1+cp);
|
||||
R_tp(2,1) = -st*sp;
|
||||
R_theta = 0.;
|
||||
R_theta(0,0) = ct*sp;
|
||||
R_theta(1,0) = st*sp;
|
||||
R_theta(2,0) = cp;
|
||||
|
||||
R_tp(0,2) = -cp-(pow(st,2.)*pow(sp,2.))/(1+cp);
|
||||
R_tp(1,2) = -R_tp(0,1);
|
||||
R_tp(2,2) = ct*sp;
|
||||
R_theta(0,1) = -st*cb + ct*cp*sb;
|
||||
R_theta(1,1) = ct*cb + st*cp*sb;
|
||||
R_theta(2,1) = -sp*sb;
|
||||
|
||||
R_beta = 0.;
|
||||
R_beta(0,0) = 1.;
|
||||
R_beta(1,1) = cos(beta);
|
||||
R_beta(1,2) = -sin(beta);
|
||||
R_beta(2,1) = sin(beta);
|
||||
R_beta(2,2) = cos(beta);
|
||||
|
||||
Mult(R_tp, R_beta, R_theta);
|
||||
R_theta(0,0) = -st*sb - ct*cp*cb;
|
||||
R_theta(1,0) = ct*sb - st*cp*cb;
|
||||
R_theta(2,0) = sp*cb;
|
||||
}
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(R_theta, Temp, Jtr(i));
|
||||
} // done orientation
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(3 + 4*q), dim, dim);
|
||||
Jtrcomp_q = R_theta;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(R_theta, Temp, Jtr(q));
|
||||
} // Done orientation
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1368,6 +1419,353 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
|
||||
|
||||
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
|
||||
|
||||
dJtr = 0.;
|
||||
const int e_id = Tpr.ElementNo;
|
||||
const FiniteElement *fe = Tpr.GetFE();
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
case IDEAL_SHAPE_GIVEN_SIZE:
|
||||
case GIVEN_SHAPE_AND_SIZE:
|
||||
{
|
||||
const DenseMatrix &Wideal =
|
||||
Geometries.GetGeomToPerfGeomJac(fe->GetGeomType());
|
||||
const int dim = Wideal.Height(),
|
||||
ndofs = fe->GetDof(),
|
||||
ntspec_dofs = ndofs*ncomp;
|
||||
|
||||
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
|
||||
par_vals_c1(ndofs), par_vals_c2(ndofs), par_vals_c3(ndofs);
|
||||
|
||||
Array<int> dofs;
|
||||
DenseMatrix dD_rho(dim), dQ_phi(dim), dR_theta(dim);
|
||||
DenseMatrix dQ_phi13(dim), dQ_phichi(dim); // dQ_phi is used for dQ/dphi12 in 3D
|
||||
DenseMatrix dR_psi(dim), dR_beta(dim);
|
||||
tspec_fesv->GetElementVDofs(e_id, dofs);
|
||||
tspec.GetSubVector(dofs, tspec_vals);
|
||||
|
||||
DenseMatrix grad_e_c1(ndofs, dim),
|
||||
grad_e_c2(ndofs, dim),
|
||||
grad_e_c3(ndofs, dim);
|
||||
Vector grad_ptr_c1(grad_e_c1.GetData(), ndofs*dim),
|
||||
grad_ptr_c2(grad_e_c2.GetData(), ndofs*dim),
|
||||
grad_ptr_c3(grad_e_c3.GetData(), ndofs*dim);
|
||||
|
||||
DenseMatrix grad_phys; // This will be (dof x dim, dof).
|
||||
fe->ProjectGrad(*fe, Tpr, grad_phys);
|
||||
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
DenseMatrix Jtrcomp_s(Jtrcomp.GetData(0 + 4*i), dim, dim); // size
|
||||
DenseMatrix Jtrcomp_d(Jtrcomp.GetData(1 + 4*i), dim, dim); // aspect-ratio
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*i), dim, dim); // skew
|
||||
DenseMatrix Jtrcomp_r(Jtrcomp.GetData(3 + 4*i), dim, dim); // orientation
|
||||
DenseMatrix work1(dim), work2(dim), work3(dim);
|
||||
|
||||
if (sizeidx != -1) // Set size
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double min_size = par_vals.Min();
|
||||
MFEM_VERIFY(min_size > 0.0,
|
||||
"Non-positive size propagated in the target definition.");
|
||||
const double size = std::max(shape * par_vals, min_size);
|
||||
double dz_dsize = (1./dim)*pow(size, 1./dim - 1.);
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_r, work1, work2); // R*Q*D
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = Wideal;
|
||||
work1.Set(dz_dsize, work1); // dz/dsize
|
||||
work1 *= grad_q(d); // dz/dsize*dsize/dx
|
||||
AddMult(work1, work2, dJtr_i); // dz/dx*R*Q*D
|
||||
}
|
||||
} // Done size
|
||||
|
||||
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
|
||||
|
||||
if (aspectratioidx != -1) // Set aspect ratio
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
aspectratioidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double aspectratio = shape * par_vals;
|
||||
dD_rho = 0.;
|
||||
dD_rho(0,0) = -0.5*pow(aspectratio,-1.5);
|
||||
dD_rho(1,1) = 0.5*pow(aspectratio,-0.5);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
|
||||
Mult(work1, Jtrcomp_q, work2); // z*R*Q
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dD_rho;
|
||||
work1 *= grad_q(d); // work1 = dD/drho*drho/dx
|
||||
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
|
||||
}
|
||||
}
|
||||
else // 3D
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
aspectratioidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double rho1 = shape * par_vals_c1;
|
||||
const double rho2 = shape * par_vals_c2;
|
||||
const double rho3 = shape * par_vals_c3;
|
||||
dD_rho = 0.;
|
||||
dD_rho(0,0) = (2./3.)*pow(rho1,-1./3.);
|
||||
dD_rho(1,1) = (2./3.)*pow(rho2,-1./3.);
|
||||
dD_rho(2,2) = (2./3.)*pow(rho3,-1./3.);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
|
||||
Mult(work1, Jtrcomp_q, work2); // z*R*Q
|
||||
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dD_rho;
|
||||
work1(0,0) *= grad_q1(d);
|
||||
work1(1,2) *= grad_q2(d);
|
||||
work1(2,2) *= grad_q3(d);
|
||||
// work1 = dD/dx = dD/drho1*drho1/dx + dD/drho2*drho2/dx
|
||||
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
|
||||
}
|
||||
}
|
||||
} // Done aspect ratio
|
||||
|
||||
if (skewidx != -1) // Set skew
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
skewidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double skew = shape * par_vals;
|
||||
|
||||
dQ_phi = 0.;
|
||||
dQ_phi(0,0) = 1.;
|
||||
dQ_phi(0,1) = -sin(skew);
|
||||
dQ_phi(1,1) = cos(skew);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dQ_phi;
|
||||
work1 *= grad_q(d); // work1 = dQ/dphi*dphi/dx
|
||||
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
|
||||
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
skewidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double phi12 = shape * par_vals_c1;
|
||||
const double phi13 = shape * par_vals_c2;
|
||||
const double chi = shape * par_vals_c3;
|
||||
|
||||
dQ_phi = 0.;
|
||||
dQ_phi(0,0) = 1.;
|
||||
dQ_phi(0,1) = -sin(phi12);
|
||||
dQ_phi(1,1) = cos(phi12);
|
||||
|
||||
dQ_phi13 = 0.;
|
||||
dQ_phi13(0,2) = -sin(phi13);
|
||||
dQ_phi13(1,2) = cos(phi13)*cos(chi);
|
||||
dQ_phi13(2,2) = cos(phi13)*sin(chi);
|
||||
|
||||
dQ_phichi = 0.;
|
||||
dQ_phichi(1,2) = -sin(phi13)*sin(chi);
|
||||
dQ_phichi(2,2) = sin(phi13)*cos(chi);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dQ_phi;
|
||||
work1 *= grad_q1(d); // work1 = dQ/dphi12*dphi12/dx
|
||||
work1.Add(grad_q2(d), dQ_phi13); // + dQ/dphi13*dphi13/dx
|
||||
work1.Add(grad_q3(d), dQ_phichi); // + dQ/dchi*dchi/dx
|
||||
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
|
||||
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
|
||||
}
|
||||
}
|
||||
} // Done skew
|
||||
|
||||
if (orientationidx != -1) // Set orientation
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
orientationidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double theta = shape * par_vals;
|
||||
dR_theta(0,0) = -sin(theta);
|
||||
dR_theta(0,1) = -cos(theta);
|
||||
dR_theta(1,0) = cos(theta);
|
||||
dR_theta(1,1) = -sin(theta);
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_s, work1, work2); // z*Q*D
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dR_theta;
|
||||
work1 *= grad_q(d); // work1 = dR/dtheta*dtheta/dx
|
||||
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
orientationidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double theta = shape * par_vals_c1;
|
||||
const double psi = shape * par_vals_c2;
|
||||
const double beta = shape * par_vals_c3;
|
||||
|
||||
const double ct = cos(theta), st = sin(theta),
|
||||
cp = cos(psi), sp = sin(psi),
|
||||
cb = cos(beta), sb = sin(beta);
|
||||
|
||||
dR_theta = 0.;
|
||||
dR_theta(0,0) = -st*sp;
|
||||
dR_theta(1,0) = ct*sp;
|
||||
dR_theta(2,0) = 0;
|
||||
|
||||
dR_theta(0,1) = -ct*cb - st*cp*sb;
|
||||
dR_theta(1,1) = -st*cb + ct*cp*sb;
|
||||
dR_theta(2,1) = 0.;
|
||||
|
||||
dR_theta(0,0) = -ct*sb + st*cp*cb;
|
||||
dR_theta(1,0) = -st*sb - ct*cp*cb;
|
||||
dR_theta(2,0) = 0.;
|
||||
|
||||
dR_beta = 0.;
|
||||
dR_beta(0,0) = 0.;
|
||||
dR_beta(1,0) = 0.;
|
||||
dR_beta(2,0) = 0.;
|
||||
|
||||
dR_beta(0,1) = st*sb + ct*cp*cb;
|
||||
dR_beta(1,1) = -ct*sb + st*cp*cb;
|
||||
dR_beta(2,1) = -sp*cb;
|
||||
|
||||
dR_beta(0,0) = -st*cb + ct*cp*sb;
|
||||
dR_beta(1,0) = ct*cb + st*cp*sb;
|
||||
dR_beta(2,0) = 0.;
|
||||
|
||||
dR_psi = 0.;
|
||||
dR_psi(0,0) = ct*cp;
|
||||
dR_psi(1,0) = st*cp;
|
||||
dR_psi(2,0) = -sp;
|
||||
|
||||
dR_psi(0,1) = 0. - ct*sp*sb;
|
||||
dR_psi(1,1) = 0. + st*sp*sb;
|
||||
dR_psi(2,1) = -cp*sb;
|
||||
|
||||
dR_psi(0,0) = 0. + ct*sp*cb;
|
||||
dR_psi(1,0) = 0. + st*sp*cb;
|
||||
dR_psi(2,0) = cp*cb;
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_s, work1, work2); // z*Q*D
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dR_theta;
|
||||
work1 *= grad_q1(d); // work1 = dR/dtheta*dtheta/dx
|
||||
work1.Add(grad_q2(d), dR_psi); // +dR/dpsi*dpsi/dx
|
||||
work1.Add(grad_q3(d), dR_beta); // +dR/dbeta*dbeta/dx
|
||||
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
|
||||
}
|
||||
}
|
||||
} // Done orientation
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MFEM_ABORT("Incompatible target type for discrete adaptation!");
|
||||
}
|
||||
Jtrcomp.Clear();
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::UpdateGradientTargetSpecification(const Vector &x,
|
||||
const double dx,
|
||||
bool use_flag)
|
||||
@@ -1474,6 +1872,17 @@ void AdaptivityEvaluator::SetParMetaInfo(const ParMesh &m,
|
||||
}
|
||||
#endif
|
||||
|
||||
void AdaptivityEvaluator::ClearGeometricFactors()
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pmesh) pmesh->DeleteGeometricFactors();
|
||||
if (pfes) pfes->GetParMesh()->DeleteGeometricFactors();
|
||||
#else
|
||||
if (mesh) mesh->DeleteGeometricFactors();
|
||||
if (fes) fes->GetMesh()->DeleteGeometricFactors();
|
||||
#endif
|
||||
}
|
||||
|
||||
AdaptivityEvaluator::~AdaptivityEvaluator()
|
||||
{
|
||||
delete fes;
|
||||
@@ -1501,6 +1910,7 @@ void TMOP_Integrator::EnableLimiting(const GridFunction &n0,
|
||||
{
|
||||
EnableLimiting(n0, w0, lfunc);
|
||||
lim_dist = &dist;
|
||||
if (PA.enabled) { EnableLimitingPA(n0); }
|
||||
}
|
||||
void TMOP_Integrator::EnableLimiting(const GridFunction &n0, Coefficient &w0,
|
||||
TMOP_LimiterFunction *lfunc)
|
||||
@@ -1646,7 +2056,8 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
|
||||
PMatI.MultTranspose(shape, p);
|
||||
pos0.MultTranspose(shape, p0);
|
||||
val += lim_normal *
|
||||
lim_func->Eval(p, p0, d_vals(i)) * coeff0->Eval(*Tpr, ip);
|
||||
lim_func->Eval(p, p0, d_vals(i)) *
|
||||
coeff0->Eval(*Tpr, ip);
|
||||
}
|
||||
|
||||
if (adaptive_limiting)
|
||||
@@ -1697,6 +2108,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
{
|
||||
const int dof = el.GetDof(), dim = el.GetDim();
|
||||
|
||||
DenseMatrix Amat(dim), work1(dim), work2(dim);
|
||||
DSh.SetSize(dof, dim);
|
||||
DS.SetSize(dof, dim);
|
||||
Jrt.SetSize(dim);
|
||||
@@ -1712,14 +2124,15 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
elvect = 0.0;
|
||||
Vector weights(nqp);
|
||||
DenseTensor Jtr(dim, dim, nqp);
|
||||
DenseTensor dJtr(dim, dim, dim*nqp);
|
||||
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
|
||||
|
||||
// Limited case.
|
||||
DenseMatrix pos0;
|
||||
Vector shape, p, p0, d_vals, grad;
|
||||
shape.SetSize(dof);
|
||||
if (coeff0)
|
||||
{
|
||||
shape.SetSize(dof);
|
||||
p.SetSize(dim);
|
||||
p0.SetSize(dim);
|
||||
pos0.SetSize(dof, dim);
|
||||
@@ -1739,7 +2152,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
|
||||
// Define ref->physical transformation, when a Coefficient is specified.
|
||||
IsoparametricTransformation *Tpr = NULL;
|
||||
if (coeff1 || coeff0 || zeta)
|
||||
if (coeff1 || coeff0 || zeta || exact_action)
|
||||
{
|
||||
Tpr = new IsoparametricTransformation;
|
||||
Tpr->SetFE(&el);
|
||||
@@ -1747,8 +2160,16 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
Tpr->ElementType = ElementTransformation::ELEMENT;
|
||||
Tpr->Attribute = T.Attribute;
|
||||
Tpr->GetPointMat().Transpose(PMatI); // PointMat = PMatI^T
|
||||
if (exact_action)
|
||||
{
|
||||
targetC->ComputeElementTargetsGradient(*ir, elfun, *Tpr, dJtr);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Vector d_detW_dx(dim);
|
||||
Vector d_Winv_dx(dim);
|
||||
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(q);
|
||||
@@ -1767,13 +2188,44 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
if (coeff1) { weight_m *= coeff1->Eval(*Tpr, ip); }
|
||||
|
||||
P *= weight_m;
|
||||
AddMultABt(DS, P, PMatO);
|
||||
AddMultABt(DS, P, PMatO); // w_q det(W) dmu/dx : dA/dx Winv
|
||||
|
||||
// TODO: derivatives of adaptivity-based targets.
|
||||
if (exact_action)
|
||||
{
|
||||
el.CalcShape(ip, shape);
|
||||
// Derivatives of adaptivity-based targets.
|
||||
// First term: w_q d*(Det W)/dx * mu(T)
|
||||
// d(Det W)/dx = det(W)*Tr[Winv*dW/dx]
|
||||
DenseMatrix dwdx(dim);
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
const DenseMatrix &dJtr_q = dJtr(q + d*ir->GetNPoints());
|
||||
Mult(Jrt, dJtr_q, dwdx );
|
||||
d_detW_dx(d) = dwdx.Trace();
|
||||
}
|
||||
d_detW_dx *= weight_m*metric->EvalW(Jpt); // *[w_q*det(W)]*mu(T)
|
||||
|
||||
// Second term: w_q det(W) dmu/dx : AdWinv/dx
|
||||
// dWinv/dx = -Winv*dW/dx*Winv
|
||||
MultAtB(PMatI, DSh, Amat);
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
const DenseMatrix &dJtr_q = dJtr(q + d*nqp);
|
||||
Mult(Jrt, dJtr_q, work1); // Winv*dw/dx
|
||||
Mult(work1, Jrt, work2); // Winv*dw/dx*Winv
|
||||
Mult(Amat, work2, work1); // A*Winv*dw/dx*Winv
|
||||
MultAtB(P, work1, work2); // dmu/dT^T*A*Winv*dw/dx*Winv
|
||||
d_Winv_dx(d) = work2.Trace(); // Tr[dmu/dT : AWinv*dw/dx*Winv]
|
||||
}
|
||||
d_Winv_dx *= -weight_m; // Include (-) factor as well
|
||||
|
||||
d_detW_dx += d_Winv_dx;
|
||||
AddMultVWt(shape, d_detW_dx, PMatO);
|
||||
}
|
||||
|
||||
if (coeff0)
|
||||
{
|
||||
el.CalcShape(ip, shape);
|
||||
if (!exact_action) { el.CalcShape(ip, shape); }
|
||||
PMatI.MultTranspose(shape, p);
|
||||
pos0.MultTranspose(shape, p0);
|
||||
lim_func->Eval_d1(p, p0, d_vals(q), grad);
|
||||
@@ -2214,7 +2666,8 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
|
||||
Jpt.SetSize(dim);
|
||||
|
||||
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
|
||||
DenseTensor Jtr(dim, dim, ir->GetNPoints());
|
||||
const int nqp = ir->GetNPoints();
|
||||
DenseTensor Jtr(dim, dim, nqp);
|
||||
|
||||
metric_energy = 0.0;
|
||||
lim_energy = 0.0;
|
||||
@@ -2227,12 +2680,12 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
|
||||
|
||||
targetC->ComputeElementTargets(i, *fe, *ir, x_vals, Jtr);
|
||||
|
||||
for (int i = 0; i < ir->GetNPoints(); i++)
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(i);
|
||||
metric->SetTargetJacobian(Jtr(i));
|
||||
CalcInverse(Jtr(i), Jrt);
|
||||
const double weight = ip.weight * Jtr(i).Det();
|
||||
const IntegrationPoint &ip = ir->IntPoint(q);
|
||||
metric->SetTargetJacobian(Jtr(q));
|
||||
CalcInverse(Jtr(q), Jrt);
|
||||
const double weight = ip.weight * Jtr(q).Det();
|
||||
|
||||
fe->CalcDShape(ip, DSh);
|
||||
MultAtB(PMatI, DSh, Jpr);
|
||||
@@ -2284,6 +2737,8 @@ void TMOP_Integrator::ComputeMinJac(const Vector &x,
|
||||
|
||||
void TMOP_Integrator::UpdateAfterMeshChange(const Vector &new_x)
|
||||
{
|
||||
PA.setup_Jtr = false;
|
||||
PA.setup_Grad = false;
|
||||
// Update zeta if adaptive limiting is enabled.
|
||||
if (zeta) { adapt_eval->ComputeAtNewPosition(new_x, *zeta); }
|
||||
}
|
||||
@@ -2439,6 +2894,49 @@ void TMOPComboIntegrator::ParEnableNormalization(const ParGridFunction &x)
|
||||
}
|
||||
#endif
|
||||
|
||||
void TMOPComboIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AssemblePA(fes);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AssembleGradientDiagonalPA(const Vector &xe,
|
||||
Vector &de) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AssembleGradientDiagonalPA(xe, de);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AddMultPA(const Vector &xe, Vector &ye) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AddMultPA(xe, ye);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AddMultGradPA(const Vector &xe, const Vector &re,
|
||||
Vector &ce) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AddMultGradPA(xe, re, ce);
|
||||
}
|
||||
}
|
||||
|
||||
double TMOPComboIntegrator::GetGridFunctionEnergyPA(const Vector &xe) const
|
||||
{
|
||||
double energy = 0.0;
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
energy += tmopi[i]->GetGridFunctionEnergyPA(xe);
|
||||
}
|
||||
return energy;
|
||||
}
|
||||
|
||||
void InterpolateTMOP_QualityMetric(TMOP_QualityMetric &metric,
|
||||
const TargetConstructor &tc,
|
||||
|
||||
+179
-12
@@ -68,6 +68,10 @@ public:
|
||||
*/
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const = 0;
|
||||
|
||||
/** @brief Return the metric ID.
|
||||
*/
|
||||
virtual int Id() const { return 0; }
|
||||
};
|
||||
|
||||
|
||||
@@ -85,6 +89,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 1; }
|
||||
};
|
||||
|
||||
/// Skew metric, 2D.
|
||||
@@ -176,6 +182,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 2; }
|
||||
};
|
||||
|
||||
/// Shape & area, ideal barrier metric, 2D
|
||||
@@ -192,6 +200,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 7; }
|
||||
};
|
||||
|
||||
/// Shape & area metric, 2D
|
||||
@@ -278,7 +288,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
};
|
||||
|
||||
/// Shape, ideal barrier metric, 2D
|
||||
@@ -296,7 +305,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
};
|
||||
|
||||
/// Area, ideal barrier metric, 2D
|
||||
@@ -314,6 +322,7 @@ public:
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 77; }
|
||||
};
|
||||
|
||||
/// Shape & orientation metric, 2D.
|
||||
@@ -400,6 +409,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 302; }
|
||||
};
|
||||
|
||||
/// Shape, ideal barrier metric, 3D
|
||||
@@ -416,6 +427,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 303; }
|
||||
};
|
||||
|
||||
/// Volume metric, 3D
|
||||
@@ -432,6 +445,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 315; }
|
||||
};
|
||||
|
||||
/// Volume, ideal barrier metric, 3D
|
||||
@@ -466,6 +481,8 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 321; }
|
||||
};
|
||||
|
||||
/// Shifted barrier form of 3D metric 16 (volume, ideal barrier metric), 3D
|
||||
@@ -589,6 +606,8 @@ public:
|
||||
|
||||
virtual void ComputeAtNewPosition(const Vector &new_nodes,
|
||||
Vector &new_field) = 0;
|
||||
|
||||
void ClearGeometricFactors();
|
||||
};
|
||||
|
||||
/** @brief Base class representing target-matrix construction algorithms for
|
||||
@@ -664,9 +683,14 @@ public:
|
||||
nodes are used by all target types except IDEAL_SHAPE_UNIT_SIZE. */
|
||||
void SetNodes(const GridFunction &n) { nodes = &n; avg_volume = 0.0; }
|
||||
|
||||
/** @brief Get the nodes to be used in the target-matrix construction. */
|
||||
const GridFunction *GetNodes() const { return nodes; }
|
||||
|
||||
/// Used by target type IDEAL_SHAPE_EQUAL_SIZE. The default volume scale is 1.
|
||||
void SetVolumeScale(double vol_scale) { volume_scale = vol_scale; }
|
||||
|
||||
const TargetType &Type() const { return target_type; }
|
||||
|
||||
/// Checks if the target matrices contain non-trivial size specification.
|
||||
virtual bool ContainsVolumeInfo() const;
|
||||
|
||||
@@ -677,6 +701,35 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
template<int DIM>
|
||||
bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
class TMOPMatrixCoefficient : public MatrixCoefficient
|
||||
{
|
||||
public:
|
||||
explicit TMOPMatrixCoefficient(int dim) : MatrixCoefficient(dim, dim) { }
|
||||
|
||||
/** @brief Evaluate the derivative of the matrix coefficient with respect to
|
||||
@a comp in the element described by @a T at the point @a ip, storing the
|
||||
result in @a K. */
|
||||
virtual void EvalGrad(DenseMatrix &K, ElementTransformation &T,
|
||||
const IntegrationPoint &ip, int comp) = 0;
|
||||
|
||||
virtual ~TMOPMatrixCoefficient() { }
|
||||
};
|
||||
|
||||
class AnalyticAdaptTC : public TargetConstructor
|
||||
@@ -685,7 +738,7 @@ protected:
|
||||
// Analytic target specification.
|
||||
Coefficient *scalar_tspec;
|
||||
VectorCoefficient *vector_tspec;
|
||||
MatrixCoefficient *matrix_tspec;
|
||||
TMOPMatrixCoefficient *matrix_tspec;
|
||||
|
||||
public:
|
||||
AnalyticAdaptTC(TargetType ttype)
|
||||
@@ -694,7 +747,7 @@ public:
|
||||
|
||||
virtual void SetAnalyticTargetSpec(Coefficient *sspec,
|
||||
VectorCoefficient *vspec,
|
||||
MatrixCoefficient *mspec);
|
||||
TMOPMatrixCoefficient *mspec);
|
||||
|
||||
/** @brief Given an element and quadrature rule, computes ref->target
|
||||
transformation Jacobians for each quadrature point in the element.
|
||||
@@ -703,6 +756,16 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
@@ -724,27 +787,38 @@ protected:
|
||||
// eta1(x+h,y), eta2(x+h,y) ... etan(x+h,y), eta1(x,y+h), eta2(x,y+h) ...
|
||||
// same for tspec_pert2h and tspec_pertmix.
|
||||
|
||||
// Components of Target Jacobian at each quadrature point of an element. This
|
||||
// is required for computation of the derivative using chain rule.
|
||||
mutable DenseTensor Jtrcomp;
|
||||
|
||||
// Note: do not use the Nodes of this space as they may not be on the
|
||||
// positions corresponding to the values of tspec.
|
||||
const FiniteElementSpace *tspec_fes;
|
||||
const FiniteElementSpace *tspec_fesv;
|
||||
|
||||
// These flags can be used by outside functions to avoid recomputing
|
||||
// the tspec and tspec_perth fields again on the same mesh.
|
||||
// These flags can be used by outside functions to avoid recomputing the
|
||||
// tspec and tspec_perth fields again on the same mesh.
|
||||
bool good_tspec, good_tspec_grad, good_tspec_hess;
|
||||
|
||||
// Evaluation of the discrete target specification on different meshes.
|
||||
// Owned.
|
||||
AdaptivityEvaluator *adapt_eval;
|
||||
|
||||
void SetDiscreteTargetBase(const GridFunction &tspec_);
|
||||
void SetTspecAtIndex(int idx, const GridFunction &tspec_);
|
||||
// PA extension
|
||||
struct { mutable Vector tspec_e; } PA;
|
||||
|
||||
void FinalizeSerialDiscreteTargetSpec();
|
||||
#ifdef MFEM_USE_MPI
|
||||
void SetTspecAtIndex(int idx, const ParGridFunction &tspec_);
|
||||
void FinalizeParDiscreteTargetSpec(const ParGridFunction &tspec_);
|
||||
#endif
|
||||
|
||||
public: // MFEM_FORALL nvcc restriction that it must be public
|
||||
void SetDiscreteTargetBase(const GridFunction &tspec_);
|
||||
void SetTspecAtIndex(int idx, const GridFunction &tspec_);
|
||||
#ifdef MFEM_USE_MPI
|
||||
void SetTspecAtIndex(int idx, const ParGridFunction &tspec_);
|
||||
#endif
|
||||
|
||||
public:
|
||||
DiscreteAdaptTC(TargetType ttype)
|
||||
: TargetConstructor(ttype),
|
||||
@@ -827,6 +901,7 @@ public:
|
||||
const Vector &GetTspecPert1H() { return tspec_pert1h; }
|
||||
const Vector &GetTspecPert2H() { return tspec_pert2h; }
|
||||
const Vector &GetTspecPertMixH() { return tspec_pertmix; }
|
||||
const FiniteElementSpace *GetTspecFesv() const { return tspec_fesv; }
|
||||
|
||||
/** @brief Given an element and quadrature rule, computes ref->target
|
||||
transformation Jacobians for each quadrature point in the element.
|
||||
@@ -837,6 +912,16 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
class TMOPNewtonSolver;
|
||||
@@ -889,6 +974,9 @@ protected:
|
||||
// Specifies that ComputeElementTargets is being called by a FD function.
|
||||
// It's used to skip terms that have exact derivative calculations.
|
||||
bool fd_call_flag;
|
||||
// Compute the exact action of the Integrator (includes derivative of the
|
||||
// target with respect to spatial position)
|
||||
bool exact_action;
|
||||
|
||||
Array <Vector *> ElemDer; //f'(x)
|
||||
Array <Vector *> ElemPertEnergy; //f(x+h)
|
||||
@@ -905,10 +993,25 @@ protected:
|
||||
// output - the result of AssembleElementVector() (dof x dim).
|
||||
DenseMatrix DSh, DS, Jrt, Jpr, Jpt, P, PMatI, PMatO;
|
||||
|
||||
// PA extension
|
||||
struct
|
||||
{
|
||||
bool enabled;
|
||||
int dim, ne, nq;
|
||||
mutable DenseTensor Jtr;
|
||||
mutable bool setup_Grad, setup_Jtr;
|
||||
mutable Vector E, O, W, X0, H, C0, LD, H0;
|
||||
const DofToQuad *maps;
|
||||
const DofToQuad *maps_lim = nullptr;
|
||||
const GeometricFactors *geom;
|
||||
const FiniteElementSpace *fes;
|
||||
const Operator *R;
|
||||
const IntegrationRule *ir;
|
||||
} PA;
|
||||
|
||||
void ComputeNormalizationEnergies(const GridFunction &x,
|
||||
double &metric_energy, double &lim_energy);
|
||||
|
||||
|
||||
void AssembleElementVectorExact(const FiniteElement &el,
|
||||
ElementTransformation &T,
|
||||
const Vector &elfun, Vector &elvect);
|
||||
@@ -978,8 +1081,8 @@ public:
|
||||
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
|
||||
zeta_0(NULL), zeta(NULL), coeff_zeta(NULL), adapt_eval(NULL),
|
||||
discr_tc(dynamic_cast<DiscreteAdaptTC *>(tc)),
|
||||
fdflag(false), dxscale(1.0e3), fd_call_flag(false)
|
||||
{ }
|
||||
fdflag(false), dxscale(1.0e3), fd_call_flag(false), exact_action(false)
|
||||
{ PA.enabled = false; }
|
||||
|
||||
~TMOP_Integrator();
|
||||
|
||||
@@ -1047,6 +1150,45 @@ public:
|
||||
virtual void AssembleElementGrad(const FiniteElement &el,
|
||||
ElementTransformation &T,
|
||||
const Vector &elfun, DenseMatrix &elmat);
|
||||
/// PA extension
|
||||
void SetupGradPA(const Vector &xe) const;
|
||||
void EnableLimitingPA(const GridFunction &n0);
|
||||
void ComputeElementTargetsPA(const Vector &xe = Vector()) const;
|
||||
|
||||
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
|
||||
double GetGridFunctionEnergyPA_2D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_C0_2D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_3D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_C0_3D(const Vector&) const;
|
||||
virtual double GetGridFunctionEnergyPA(const Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace&);
|
||||
|
||||
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
|
||||
void AssembleDiagonalPA_2D(Vector&) const;
|
||||
void AssembleDiagonalPA_3D(Vector&) const;
|
||||
void AssembleDiagonalPA_C0_2D(Vector&) const;
|
||||
void AssembleDiagonalPA_C0_3D(Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AddMultPA;
|
||||
void AddMultPA_2D(const Vector&, Vector&) const;
|
||||
void AddMultPA_3D(const Vector&, Vector&) const;
|
||||
void AddMultPA_C0_2D(const Vector&, Vector&) const;
|
||||
void AddMultPA_C0_3D(const Vector&, Vector&) const;
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AddMultGradPA;
|
||||
void AddMultGradPA_2D(const Vector&, Vector&) const;
|
||||
void AddMultGradPA_3D(const Vector&, const Vector&, Vector&) const;
|
||||
void AddMultGradPA_C0_2D(const Vector&, const Vector&, Vector&) const;
|
||||
void AddMultGradPA_C0_3D(const Vector&, const Vector&, Vector&) const;
|
||||
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
|
||||
|
||||
void AssembleGradPA_2D(const Vector&) const;
|
||||
void AssembleGradPA_3D(const Vector&) const;
|
||||
void AssembleGradPA_C0_2D(const Vector&) const;
|
||||
void AssembleGradPA_C0_3D(const Vector&) const;
|
||||
|
||||
DiscreteAdaptTC *GetDiscreteAdaptTC() const { return discr_tc; }
|
||||
|
||||
@@ -1066,6 +1208,20 @@ public:
|
||||
void SetFDhScale(double _dxscale) { dxscale = _dxscale; }
|
||||
bool GetFDFlag() const { return fdflag; }
|
||||
double GetFDh() const { return dx; }
|
||||
|
||||
/** @brief Flag to control if exact action of Integration is effected. */
|
||||
void SetExactActionFlag(bool flag_) { exact_action = flag_; }
|
||||
|
||||
void ReleaseTemporaryMemory()
|
||||
{
|
||||
if (PA.enabled)
|
||||
{
|
||||
PA.H.GetMemory().DeleteDevice();
|
||||
PA.H0.GetMemory().DeleteDevice();
|
||||
//PA.Jtr.GetMemory().DeleteDevice();
|
||||
//PA.setup_Jtr = false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
class TMOPComboIntegrator : public NonlinearFormIntegrator
|
||||
@@ -1114,6 +1270,17 @@ public:
|
||||
#ifdef MFEM_USE_MPI
|
||||
void ParEnableNormalization(const ParGridFunction &x);
|
||||
#endif
|
||||
|
||||
/// PA extension
|
||||
using NonlinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace&);
|
||||
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::AddMultPA;
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::AddMultGradPA;
|
||||
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
|
||||
virtual double GetGridFunctionEnergyPA(const Vector&) const;
|
||||
};
|
||||
|
||||
/// Interpolates the @a metric's values at the nodes of @a metric_gf.
|
||||
|
||||
+363
@@ -0,0 +1,363 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
#include "tmop_tools.hpp"
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
|
||||
void TMOP_Integrator::SetupGradPA(const Vector &xe) const
|
||||
{
|
||||
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
|
||||
PA.setup_Grad = true;
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AssembleGradPA_2D(xe);
|
||||
if (coeff0) { AssembleGradPA_C0_2D(xe); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AssembleGradPA_3D(xe);
|
||||
if (coeff0) { AssembleGradPA_C0_3D(xe); }
|
||||
}
|
||||
}
|
||||
|
||||
// We might come here w/o knowing that PA will be used.
|
||||
// It is the case when EnableLimiting is called before the Setup => AssemblePA.
|
||||
void TMOP_Integrator::EnableLimitingPA(const GridFunction &n0)
|
||||
{
|
||||
MFEM_VERIFY(PA.enabled, "EnableLimitingPA but PA is not enabled!");
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
|
||||
// Nodes0
|
||||
const FiniteElementSpace *n0_fes = n0.FESpace();
|
||||
const Operator *n0_R = n0_fes->GetElementRestriction(ordering);
|
||||
PA.X0.SetSize(n0_R->Height(), Device::GetMemoryType());
|
||||
PA.X0.UseDevice(true);
|
||||
n0_R->Mult(n0, PA.X0);
|
||||
|
||||
// Get the 1D maps for the distance FE space.
|
||||
const IntegrationRule &ir = *EnergyIntegrationRule(*n0.FESpace()->GetFE(0));
|
||||
PA.maps_lim =
|
||||
&lim_dist->FESpace()->GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
|
||||
// lim_dist & lim_func checks
|
||||
MFEM_VERIFY(lim_dist, "No lim_dist!")
|
||||
const FiniteElementSpace *ld_fes = lim_dist->FESpace();
|
||||
const Operator *ld_R = ld_fes->GetElementRestriction(ordering);
|
||||
MFEM_VERIFY(ld_R, "No lim_dist restriction operator found!");
|
||||
PA.LD.SetSize(ld_R->Height(), Device::GetMemoryType());
|
||||
PA.LD.UseDevice(true);
|
||||
ld_R->Mult(*lim_dist, PA.LD);
|
||||
|
||||
// Only TMOP_QuadraticLimiter is supported
|
||||
MFEM_VERIFY(lim_func, "No lim_func!")
|
||||
MFEM_VERIFY(dynamic_cast<TMOP_QuadraticLimiter*>(lim_func),
|
||||
"Only TMOP_QuadraticLimiter is supported");
|
||||
}
|
||||
|
||||
bool TargetConstructor::ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe) const
|
||||
{
|
||||
MFEM_VERIFY(Jtr.SizeI() == Jtr.SizeJ() && Jtr.SizeI() > 1, "");
|
||||
const int dim = Jtr.SizeI();
|
||||
if (dim == 2) { return ComputeElementTargetsPA<2>(fes, ir, Jtr, xe); }
|
||||
if (dim == 3) { return ComputeElementTargetsPA<3>(fes, ir, Jtr, xe); }
|
||||
return false;
|
||||
}
|
||||
|
||||
bool AnalyticAdaptTC::ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe) const
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Code paths leading to ComputeElementTargets:
|
||||
// - GetElementEnergy(elfun) which is done through GetGridFunctionEnergyPA(x)
|
||||
// - AssembleElementVectorExact(elfun)
|
||||
// - AssembleElementGradExact(elfun)
|
||||
// - EnableNormalization(x) -> ComputeNormalizationEnergies(x)
|
||||
// - (AssembleElementVectorFD(elfun))
|
||||
// - (AssembleElementGradFD(elfun))
|
||||
// ============================================================================
|
||||
// - TargetConstructor():
|
||||
// - IDEAL_SHAPE_UNIT_SIZE: Wideal
|
||||
// - IDEAL_SHAPE_EQUAL_SIZE: α * Wideal
|
||||
// - IDEAL_SHAPE_GIVEN_SIZE: β * Wideal
|
||||
// - GIVEN_SHAPE_AND_SIZE: β * Wideal
|
||||
// - AnalyticAdaptTC(elfun):
|
||||
// - GIVEN_FULL: matrix_tspec->Eval(Jtr(elfun))
|
||||
// - DiscreteAdaptTC():
|
||||
// - IDEAL_SHAPE_GIVEN_SIZE: size^{1.0/dim} * Jtr(i) (size)
|
||||
// - GIVEN_SHAPE_AND_SIZE: Jtr(i) *= D_rho (ratio)
|
||||
// Jtr(i) *= Q_phi (skew)
|
||||
// Jtr(i) *= R_theta (orientation)
|
||||
void TMOP_Integrator::ComputeElementTargetsPA(const Vector &xe) const
|
||||
{
|
||||
PA.setup_Jtr = false;
|
||||
const FiniteElementSpace *fes = PA.fes;
|
||||
const IntegrationRule *ir = EnergyIntegrationRule(*fes->GetFE(0));
|
||||
const TargetConstructor::TargetType &target_type = targetC->Type();
|
||||
const DiscreteAdaptTC *discr_tc = GetDiscreteAdaptTC();
|
||||
|
||||
// Skip when TargetConstructor needs the nodes but have not been set
|
||||
const bool use_nodes =
|
||||
target_type == TargetConstructor::IDEAL_SHAPE_EQUAL_SIZE ||
|
||||
target_type == TargetConstructor::IDEAL_SHAPE_GIVEN_SIZE ||
|
||||
target_type == TargetConstructor::GIVEN_SHAPE_AND_SIZE;
|
||||
if (targetC && !discr_tc && use_nodes && !targetC->GetNodes()) { return; }
|
||||
|
||||
// Try to use the TargetConstructor ComputeElementTargetsPA
|
||||
PA.setup_Jtr = targetC->ComputeElementTargetsPA(fes, ir, PA.Jtr);
|
||||
if (PA.setup_Jtr) { return; }
|
||||
|
||||
// Defaulting to host version
|
||||
PA.Jtr.HostWrite();
|
||||
|
||||
const int NE = PA.ne;
|
||||
const int NQ = PA.nq;
|
||||
const int dim = PA.dim;
|
||||
DenseTensor &Jtr = PA.Jtr;
|
||||
|
||||
Vector x;
|
||||
const bool useable_input_vector = xe.Size() > 0;
|
||||
const bool use_input_vector = target_type == TargetConstructor::GIVEN_FULL;
|
||||
|
||||
if (use_input_vector && !useable_input_vector) { return; }
|
||||
|
||||
if (discr_tc && !discr_tc->GetTspecFesv()) { return; }
|
||||
|
||||
if (use_input_vector)
|
||||
{
|
||||
x.SetSize(PA.R->Width(), Device::GetMemoryType());
|
||||
x.UseDevice(true);
|
||||
PA.R->MultTranspose(xe, x);
|
||||
// Scale by weights
|
||||
const int N = PA.W.Size();
|
||||
const auto W = Reshape(PA.W.Read(), N);
|
||||
auto X = Reshape(x.ReadWrite(), N);
|
||||
MFEM_FORALL(i, N, X(i) /= W(i););
|
||||
}
|
||||
|
||||
// Use TargetConstructor::ComputeElementTargets to fill the PA.Jtr
|
||||
Vector elfun;
|
||||
Array<int> vdofs;
|
||||
DenseTensor J;
|
||||
for (int e = 0; e < NE; e++)
|
||||
{
|
||||
const FiniteElement &fe = *fes->GetFE(e);
|
||||
if (use_input_vector)
|
||||
{
|
||||
fes->GetElementVDofs(e, vdofs);
|
||||
x.GetSubVector(vdofs, elfun);
|
||||
}
|
||||
J.UseExternalData(Jtr(e*NQ).Data(), dim, dim, NQ);
|
||||
targetC->ComputeElementTargets(e, fe, *ir, elfun, J);
|
||||
}
|
||||
PA.setup_Jtr = true;
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
PA.enabled = true;
|
||||
MFEM_ASSERT(fes.GetMesh()->GetNE() > 0, "");
|
||||
PA.ir = EnergyIntegrationRule(*fes.GetFE(0));
|
||||
const IntegrationRule *ir = PA.ir;
|
||||
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
|
||||
"PA Only supports Ordering::byNODES!");
|
||||
|
||||
PA.fes = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const int nq = PA.nq = ir->GetNPoints();
|
||||
const int ne = PA.ne = fes.GetMesh()->GetNE();
|
||||
const int dim = PA.dim = mesh->Dimension();
|
||||
MFEM_VERIFY(PA.dim == 2 || PA.dim == 3, "Not yet implemented!");
|
||||
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
PA.maps = &fes.GetFE(0)->GetDofToQuad(*ir, mode);
|
||||
PA.geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
// Energy vector
|
||||
PA.E.UseDevice(true);
|
||||
PA.E.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
|
||||
// Setup initialization
|
||||
PA.setup_Jtr = false;
|
||||
PA.setup_Grad = false;
|
||||
|
||||
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
|
||||
// H for Grad
|
||||
PA.H.SetSize(dim*dim * dim*dim * nq*ne, temp_type);
|
||||
// H0 for coeff0
|
||||
PA.H0.SetSize(dim * dim * nq*ne, temp_type);
|
||||
|
||||
// Restriction setup
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
PA.R = fes.GetElementRestriction(ordering);
|
||||
MFEM_VERIFY(PA.R, "Not yet implemented!");
|
||||
|
||||
// Weight of the R^t
|
||||
PA.W.SetSize(PA.R->Width(), Device::GetDeviceMemoryType());
|
||||
PA.W.UseDevice(true);
|
||||
PA.O.SetSize(dim*ne*nq, Device::GetDeviceMemoryType());
|
||||
PA.O.UseDevice(true);
|
||||
PA.O = 1.0;
|
||||
PA.R->MultTranspose(PA.O, PA.W);
|
||||
|
||||
// Scalar vector of '1'
|
||||
PA.O.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
PA.O = 1.0;
|
||||
|
||||
// TargetConstructor TargetType setup
|
||||
PA.Jtr.SetSize(dim, dim, PA.ne*PA.nq);//, temp_type);
|
||||
ComputeElementTargetsPA();
|
||||
|
||||
// Coeff0 PA.C0
|
||||
PA.C0.UseDevice(true);
|
||||
if (coeff0 == nullptr)
|
||||
{
|
||||
PA.C0.SetSize(1, Device::GetMemoryType());
|
||||
PA.C0.HostWrite();
|
||||
PA.C0(0) = 0.0;
|
||||
}
|
||||
else if (ConstantCoefficient* cQ =
|
||||
dynamic_cast<ConstantCoefficient*>(coeff0))
|
||||
{
|
||||
PA.C0.SetSize(1, Device::GetMemoryType());
|
||||
PA.C0.HostWrite();
|
||||
PA.C0(0) = cQ->constant;
|
||||
}
|
||||
else
|
||||
{
|
||||
PA.C0.SetSize(PA.nq * PA.ne, Device::GetMemoryType());
|
||||
auto C0 = Reshape(PA.C0.HostWrite(), PA.nq, PA.ne);
|
||||
for (int e = 0; e < ne; ++e)
|
||||
{
|
||||
ElementTransformation& T = *fes.GetElementTransformation(e);
|
||||
for (int q = 0; q < nq; ++q)
|
||||
{
|
||||
C0(q,e) = coeff0->Eval(T, ir->IntPoint(q));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (coeff0)
|
||||
{
|
||||
MFEM_VERIFY(nodes0, "nodes0 has not been set!");
|
||||
EnableLimitingPA(*nodes0);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradientDiagonalPA(const Vector &xe,
|
||||
Vector &de) const
|
||||
{
|
||||
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
|
||||
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
|
||||
|
||||
if (!PA.setup_Grad) { SetupGradPA(xe); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AssembleDiagonalPA_2D(de);
|
||||
if (coeff0) { AssembleDiagonalPA_C0_2D(de); }
|
||||
}
|
||||
else if (PA.dim == 3)
|
||||
{
|
||||
AssembleDiagonalPA_3D(de);
|
||||
if (coeff0) { AssembleDiagonalPA_C0_3D(de); }
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("3D diagonal computation is WIP.");
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultPA(const Vector &xe, Vector &ye) const
|
||||
{
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AddMultPA_2D(xe,ye);
|
||||
if (coeff0) { AddMultPA_C0_2D(xe,ye); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AddMultPA_3D(xe,ye);
|
||||
if (coeff0) { AddMultPA_C0_3D(xe,ye); }
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA(const Vector &xe,
|
||||
const Vector &re, Vector &ce) const
|
||||
{
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
|
||||
|
||||
if (!PA.setup_Grad) { SetupGradPA(xe); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AddMultGradPA_2D(re,ce);
|
||||
if (coeff0) { AddMultGradPA_C0_2D(xe,re,ce); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AddMultGradPA_3D(xe,re,ce);
|
||||
if (coeff0) { AddMultGradPA_C0_3D(xe,re,ce); }
|
||||
}
|
||||
}
|
||||
|
||||
double TMOP_Integrator::GetGridFunctionEnergyPA(const Vector &xe) const
|
||||
{
|
||||
double energy = 0.0;
|
||||
|
||||
ComputeElementTargetsPA(xe);
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
energy = GetGridFunctionEnergyPA_2D(xe);
|
||||
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_2D(xe); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
energy = GetGridFunctionEnergyPA_3D(xe);
|
||||
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_3D(xe); }
|
||||
}
|
||||
|
||||
return energy;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
+143
@@ -0,0 +1,143 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_TMOP_PA_HPP
|
||||
#define MFEM_TMOP_PA_HPP
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include "../fem/kernels.hpp"
|
||||
|
||||
#include <unordered_map>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace kernels
|
||||
{
|
||||
|
||||
/// Generic emplace
|
||||
template<typename K, const int N,
|
||||
typename Key_t = typename K::Key_t,
|
||||
typename Kernel_t = typename K::Kernel_t>
|
||||
void emplace(std::unordered_map<Key_t, Kernel_t> &map)
|
||||
{
|
||||
constexpr Key_t key = K::template GetKey<N>();
|
||||
constexpr Kernel_t value = K::template GetValue<key>();
|
||||
map.emplace(key, value);
|
||||
}
|
||||
|
||||
/// Instances
|
||||
template<class K, typename T, T... idx>
|
||||
struct instances
|
||||
{
|
||||
static void Fill(std::unordered_map<typename K::Key_t,
|
||||
typename K::Kernel_t> &map)
|
||||
{
|
||||
using unused = int[];
|
||||
(void) unused {0, (emplace<K,idx>(map), 0)... };
|
||||
}
|
||||
};
|
||||
|
||||
/// Cat instances
|
||||
template<class K, typename Offset, typename Lhs, typename Rhs> struct cat;
|
||||
template<class K, typename T, T Offset, T... Lhs, T... Rhs>
|
||||
struct cat<K, std::integral_constant<T, Offset>,
|
||||
instances<K, T, Lhs...>,
|
||||
instances<K, T, Rhs...> >
|
||||
{ using type = instances<K, T, Lhs..., (Offset + Rhs)...>; };
|
||||
|
||||
/// Sequence, empty and one element terminal cases
|
||||
template<class K, typename T, typename N>
|
||||
struct sequence
|
||||
{
|
||||
using Lhs = std::integral_constant<T, N::value/2>;
|
||||
using Rhs = std::integral_constant<T, N::value-Lhs::value>;
|
||||
using type = typename cat<K, Lhs,
|
||||
typename sequence<K, T, Lhs>::type,
|
||||
typename sequence<K, T, Rhs>::type>::type;
|
||||
};
|
||||
|
||||
template<class K, typename T>
|
||||
struct sequence<K, T, std::integral_constant<T,0> >
|
||||
{ using type = instances<K,T>; };
|
||||
|
||||
template<class K, typename T>
|
||||
struct sequence<K, T, std::integral_constant<T,1> >
|
||||
{ using type = instances<K,T,0>; };
|
||||
|
||||
/// Make_sequence
|
||||
template<class Instance, typename T = typename Instance::Key_t>
|
||||
using make_sequence =
|
||||
typename sequence<Instance, T, std::integral_constant<T,Instance::N> >::type;
|
||||
|
||||
/// Instantiator class
|
||||
template<class Instance,
|
||||
typename Key_t = typename Instance::Key_t,
|
||||
typename Return_t = typename Instance::Return_t,
|
||||
typename Kernel_t = typename Instance::Kernel_t>
|
||||
class Instantiator
|
||||
{
|
||||
private:
|
||||
using map_t = std::unordered_map<Key_t, Kernel_t>;
|
||||
map_t map;
|
||||
|
||||
public:
|
||||
Instantiator() { make_sequence<Instance>().Fill(map); }
|
||||
|
||||
bool Find(const Key_t id)
|
||||
{
|
||||
return (map.find(id) != map.end()) ? true : false;
|
||||
}
|
||||
|
||||
Kernel_t At(const Key_t id) { return map.at(id); }
|
||||
};
|
||||
|
||||
/// MFEM_REGISTER_TMOP_KERNELS macro:
|
||||
/// - forward declaration of the kernel
|
||||
/// - kernel pointer declaration
|
||||
/// - struct K##name##_T definition
|
||||
/// - Instantiator definition
|
||||
/// - re-use kernel return type and name before its body
|
||||
#define MFEM_REGISTER_TMOP_KERNELS(return_t, kernel, ...) \
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0> \
|
||||
return_t kernel(__VA_ARGS__);\
|
||||
typedef return_t (*kernel##_p)(__VA_ARGS__);\
|
||||
struct K##kernel##_T {\
|
||||
static const int N = 14;\
|
||||
using Key_t = std::size_t;\
|
||||
using Kernel_t = kernel##_p;\
|
||||
using Return_t = return_t;\
|
||||
template<Key_t I> static constexpr Key_t GetKey() noexcept { return \
|
||||
I==0 ? 0x22 : I==1 ? 0x23 : I==2 ? 0x24 : I==3 ? 0x25 : I==4 ? 0x26 :\
|
||||
I==5 ? 0x33 : I==6 ? 0x34 : I==7 ? 0x35 : I==8 ? 0x36 :\
|
||||
I==9 ? 0x44 : I==10 ? 0x45 : I==11 ? 0x46 :\
|
||||
I==12 ? 0x55 : I==13 ? 0x56 : 0; }\
|
||||
template<Key_t ID> static constexpr Kernel_t GetValue() noexcept\
|
||||
{ return &kernel<(ID>>4)&0xF, ID&0xF>; }\
|
||||
};\
|
||||
static kernels::Instantiator<K##kernel##_T> K##kernel;\
|
||||
template<int T_D1D, int T_Q1D, int T_MAX> return_t kernel(__VA_ARGS__)
|
||||
|
||||
/// MFEM_LAUNCH_TMOP_KERNEL macro
|
||||
#define MFEM_LAUNCH_TMOP_KERNEL(kernel, id, ...)\
|
||||
if (K##kernel.Find(id)) { return K##kernel.At(id)(__VA_ARGS__,0,0); }\
|
||||
else {\
|
||||
constexpr int T_MAX = 4;\
|
||||
MFEM_VERIFY(D1D <= MAX_D1D && Q1D <= MAX_Q1D, "Max size error!");\
|
||||
return kernel<0,0,T_MAX>(__VA_ARGS__,D1D,Q1D); }
|
||||
|
||||
} // namespace kernels
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_TMOP_PA_HPP
|
||||
@@ -0,0 +1,161 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/* // Original i-j assembly (old invariants code).
|
||||
for (int e = 0; e < NE; e++)
|
||||
{
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
el.CalcDShape(ip, DSh);
|
||||
Mult(DSh, Jrt, DS);
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
for (int r = 0; r < dim; r++)
|
||||
{
|
||||
for (int c = 0; c < dim; c++)
|
||||
{
|
||||
for (int rr = 0; rr < dim; rr++)
|
||||
{
|
||||
for (int cc = 0; cc < dim; cc++)
|
||||
{
|
||||
const double H = h(r, c, rr, cc);
|
||||
A(e, i + r*dof, j + rr*dof) +=
|
||||
weight_q * DS(i, c) * DS(j, cc) * H;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}*/
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_2D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const DenseTensor &j,
|
||||
const Vector &h,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qd[DIM*DIM*MQ1*MD1];
|
||||
DeviceTensor<4,double> QD(qd, DIM, DIM, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD(0,0,qx,dy) = 0.0;
|
||||
QD(0,1,qx,dy) = 0.0;
|
||||
QD(1,0,qx,dy) = 0.0;
|
||||
QD(1,1,qx,dy) = 0.0;
|
||||
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double j[4];
|
||||
ConstDeviceMatrix Jrt(j,2,2);
|
||||
kernels::CalcInverse<2>(Jtr, j);
|
||||
|
||||
const double gg = G(qy,dy) * G(qy,dy);
|
||||
const double gb = G(qy,dy) * B(qy,dy);
|
||||
const double bb = B(qy,dy) * B(qy,dy);
|
||||
const double bgb[4] = { bb, gb, gb, gg };
|
||||
ConstDeviceMatrix BG(bgb,2,2);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
const double Jij = Jrt(i,i) * Jrt(j,j);
|
||||
const double alpha = Jij * BG(i,j);
|
||||
QD(i,j,qx,dy) += alpha * H(v,i,v,j,qx,qy,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double gg = G(qx,dx) * G(qx,dx);
|
||||
const double gb = G(qx,dx) * B(qx,dx);
|
||||
const double bb = B(qx,dx) * B(qx,dx);
|
||||
d += gg * QD(0,0,qx,dy);
|
||||
d += gb * QD(0,1,qx,dy);
|
||||
d += gb * QD(1,0,qx,dy);
|
||||
d += bb * QD(1,1,qx,dy);
|
||||
}
|
||||
D(dx,dy,v,e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_2D,id,N,B,G,J,H,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,96 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_2D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &h0,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qd[MQ1*MD1];
|
||||
DeviceTensor<2,double> QD(qd, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD(qx,dy) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double bb = B(qy,dy) * B(qy,dy);
|
||||
QD(qx,dy) += bb * H0(v,v,qx,qy,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double bb = B(qx,dx) * B(qx,dx);
|
||||
d += bb * QD(qx,dy);
|
||||
}
|
||||
D(dx,dy,v,e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_2D,id,N,B,H0,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,128 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_2D,
|
||||
const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
const DenseTensor &j_,
|
||||
const Vector &h_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
|
||||
const auto H = Reshape(h_.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^T.DSh
|
||||
double Jpr[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,Jpr);
|
||||
|
||||
// Jpt = Jpr . Jrt
|
||||
double Jpt[4];
|
||||
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
|
||||
|
||||
// B = Jpt : H
|
||||
double B[4];
|
||||
DeviceMatrix M(B,2,2);
|
||||
ConstDeviceMatrix J(Jpt,2,2);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
M(i,j) = 0.0;
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
M(i,j) += H(r,c,i,j,qx,qy,e) * J(r,c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// C = Jrt . B
|
||||
double C[4];
|
||||
kernels::MultABt(2,2,2, Jrt, B, C);
|
||||
|
||||
// Overwrite QQ = Jrt . (Jpt : H)^t
|
||||
kernels::PushGrad<MQ1,NBZ>(qx,qy, C, QQ);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::LoadBGt<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
kernels::GradYt<MD1,MQ1,NBZ>(D1D,Q1D,BG,QQ,DQ);
|
||||
kernels::GradXt<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,Y,e);
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_2D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_2D,id,N,B,G,J,H,R,C);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,107 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_C0_2D,
|
||||
const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &h0_,
|
||||
const Vector &r_,
|
||||
Vector &c_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto H0 = Reshape(h0_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto R = Reshape(r_.Read(), D1D, D1D, DIM, NE);
|
||||
|
||||
auto Y = Reshape(c_.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[2][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[2][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,R,XY);
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,XY,DQ);
|
||||
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
// Xh = X^T . Sh
|
||||
double Xh[2];
|
||||
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,Xh);
|
||||
|
||||
double B[4];
|
||||
DeviceMatrix H(B,2,2);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
H(i,j) = H0(i,j,qx,qy,e);
|
||||
}
|
||||
}
|
||||
|
||||
// p2 = B . Xh
|
||||
double p2[2];
|
||||
kernels::Mult(2,2,B,Xh,p2);
|
||||
kernels::PushEval<MQ1,NBZ>(qx,qy,p2,QQ);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::LoadBt<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
kernels::EvalXt<MD1,MQ1,NBZ>(D1D,Q1D,B,QQ,DQ);
|
||||
kernels::EvalYt<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,Y,e);
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_C0_2D(const Vector &X, const Vector &R,
|
||||
Vector &C) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_C0_2D,id,N,B,H0,R,C);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,247 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
#include "../linalg/dinvariants.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
using Args = kernels::InvariantsEvaluator2D::Buffers;
|
||||
|
||||
// weight * ddI1
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_001(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const double h = ddi1(r,c);
|
||||
H(r,c,i,j,qx,qy,e) = weight * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 0.5 * weight * dI1b
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_002(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4], ddI1b[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.ddI1(ddI1)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2b(dI2b));
|
||||
const double w = 0.5 * weight;
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const double h = ddi1b(r,c);
|
||||
H(r,c,i,j,qx,qy,e) = w * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_007(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.ddI1(ddI1)
|
||||
.ddI2(ddI2)
|
||||
.dI1(dI1)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b));
|
||||
const double c1 = 1./ie.Get_I2();
|
||||
const double c2 = weight*c1*c1;
|
||||
const double c3 = ie.Get_I1()*c2;
|
||||
ConstDeviceMatrix di1(ie.Get_dI1(),DIM,DIM);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r,c,i,j,qx,qy,e) =
|
||||
weight * (1.0 + c1) * ddi1(r,c)
|
||||
- c3 * ddi2(r,c)
|
||||
- c2 * ( di1(i,j) * di2(r,c) + di2(i,j) * di1(r,c) )
|
||||
+ 2.0 * c1 * c3 * di2(r,c) * di2(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_077(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double dI2[4], dI2b[4], ddI2[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2));
|
||||
const double I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r,c,i,j,qx,qy,e) =
|
||||
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r,c)
|
||||
+ weight * (I2inv_sq / I2) * di2(r,c) * di2(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_2D,
|
||||
const Vector &x_,
|
||||
const double metric_normal,
|
||||
const int mid,
|
||||
const int NE,
|
||||
const Array<double> &w_,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
const DenseTensor &j_,
|
||||
Vector &h_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
MFEM_VERIFY(mid == 1 || mid == 2 || mid == 7 || mid == 77,
|
||||
"Metric not yet implemented!");
|
||||
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto W = Reshape(w_.Read(), Q1D, Q1D);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
|
||||
auto H = Reshape(h_.Write(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double s_BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double s_X[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double s_DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double s_QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,s_X);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D, Q1D, b, g, s_BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_X, s_DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_DQ, s_QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
const double detJtr = kernels::Det<2>(Jtr);
|
||||
const double weight = metric_normal * W(qx,qy) * detJtr;
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^t.DSh
|
||||
double Jpr[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,s_QQ,Jpr);
|
||||
|
||||
// Jpt = Jpr.Jrt
|
||||
double Jpt[4];
|
||||
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
|
||||
|
||||
// metric->AssembleH
|
||||
if (mid == 1) { EvalH_001(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 2) { EvalH_002(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 7) { EvalH_007(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 77) { EvalH_077(e,qx,qy,weight,Jpt,H); }
|
||||
} // qx
|
||||
} // qy
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_2D(const Vector &X) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int M = metric->Id();
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const double mn = metric_normal;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &W = PA.ir->GetWeights();
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_2D,id,X,mn,M,N,W,B,G,J,H);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,125 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_C0_2D,
|
||||
const double lim_normal,
|
||||
const Vector &lim_dist,
|
||||
const Vector &c0_,
|
||||
const int NE,
|
||||
const DenseTensor &j_,
|
||||
const Array<double> &w_,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bld_,
|
||||
Vector &h0_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const bool const_c0 = c0_.Size() == 1;
|
||||
const auto C0 = const_c0 ?
|
||||
Reshape(c0_.Read(), 1, 1, 1) :
|
||||
Reshape(c0_.Read(), Q1D, Q1D, NE);
|
||||
const auto LD = Reshape(lim_dist.Read(), D1D, D1D, NE);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto W = Reshape(w_.Read(), Q1D, Q1D);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto bld = Reshape(bld_.Read(), Q1D, D1D);
|
||||
|
||||
auto H0 = Reshape(h0_.Write(), DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
MFEM_SHARED double BLD[MQ1*MD1];
|
||||
|
||||
MFEM_SHARED double XY[NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,LD,XY);
|
||||
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,bld,BLD);
|
||||
|
||||
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,BLD,XY,DQ);
|
||||
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,BLD,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
const double detJtr = kernels::Det<2>(Jtr);
|
||||
const double weight = W(qx,qy) * detJtr;
|
||||
const double coeff0 = const_c0 ? C0(0,0,0) : C0(qx,qy,e);
|
||||
const double weight_m = weight * lim_normal * coeff0;
|
||||
|
||||
double D;
|
||||
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,D);
|
||||
const double dist = D; // GetValues, default comp set to 0
|
||||
|
||||
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
|
||||
// d2.Diag(1.0 / (dist * dist), x.Size());
|
||||
const double c = 1.0 / (dist * dist);
|
||||
double grad_grad[4];
|
||||
kernels::Diag<2>(c, grad_grad);
|
||||
ConstDeviceMatrix gg(grad_grad,DIM,DIM);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
H0(i,j,qx,qy,e) = weight_m * gg(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &X) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const double ln = lim_normal;
|
||||
const Vector &LD = PA.LD;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &W = PA.ir->GetWeights();
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &BLD = PA.maps_lim->B;
|
||||
const Vector &C0 = PA.C0;
|
||||
Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_C0_2D,id,ln,LD,C0,N,J,W,B,BLD,H0);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,150 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_3D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const DenseTensor &j,
|
||||
const Vector &h,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
|
||||
MFEM_SHARED double qdd[MQ1*MD1*MD1];
|
||||
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
|
||||
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; ++v)
|
||||
{
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD(qx,qy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,qz,e);
|
||||
double jrt[9];
|
||||
ConstDeviceMatrix Jrt(jrt,3,3);
|
||||
kernels::CalcInverse<3>(Jtr, jrt);
|
||||
const double Bz = B(qz,dz);
|
||||
const double Gz = G(qz,dz);
|
||||
const double L = i==2 ? Gz : Bz;
|
||||
const double R = j==2 ? Gz : Bz;
|
||||
const double Jij = Jrt(i,i) * Jrt(j,j);
|
||||
const double h = H(v,i,v,j,qx,qy,qz,e);
|
||||
QQD(qx,qy,dz) += L * Jij * h * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// second tensor contraction, along y direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD(qx,dy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
const double Gy = G(qy,dy);
|
||||
const double L = i==1 ? Gy : By;
|
||||
const double R = j==1 ? Gy : By;
|
||||
QDD(qx,dy,dz) += L * QQD(qx,qy,dz) * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// third tensor contraction, along x direction
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
const double Gx = G(qx,dx);
|
||||
const double L = i==0 ? Gx : Bx;
|
||||
const double R = j==0 ? Gx : Bx;
|
||||
d += L * QDD(qx,dy,dz) * R;
|
||||
}
|
||||
D(dx,dy,dz,v,e) += d;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_3D,id,N,B,G,J,H,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,123 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_3D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &h0,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
|
||||
MFEM_SHARED double qdd[MQ1*MD1*MD1];
|
||||
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
|
||||
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; ++v)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD(qx,qy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double Bz = B(qz,dz);
|
||||
QQD(qx,qy,dz) += Bz * H0(v,v,qx,qy,qz,e) * Bz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// second tensor contraction, along y direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD(qx,dy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
QDD(qx,dy,dz) += By * QQD(qx,qy,dz) * By;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// third tensor contraction, along x direction
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
d += Bx * QDD(qx,dy,dz) * Bx;
|
||||
}
|
||||
D(dx,dy,dz, v, e) += d;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_3D,id,N,B,H0,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user