Compare commits
185
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6497f4d829 | ||
|
|
3b533595b8 | ||
|
|
d21491f91e | ||
|
|
0a1f3a0b0a | ||
|
|
438a9968ce | ||
|
|
b1a343f514 | ||
|
|
a0a29e3303 | ||
|
|
a921f56177 | ||
|
|
2a6edf7d43 | ||
|
|
d6fa604631 | ||
|
|
3add636a18 | ||
|
|
21000a319e | ||
|
|
95703e3a6d | ||
|
|
39dfcac5b3 | ||
|
|
45b618fab2 | ||
|
|
22838a13af | ||
|
|
ee58c6828a | ||
|
|
b50b0c0ac1 | ||
|
|
6c40560766 | ||
|
|
cadc4516ef | ||
|
|
bb584f04f8 | ||
|
|
cec500bb11 | ||
|
|
a88917276a | ||
|
|
2227958c30 | ||
|
|
8c0d8c3708 | ||
|
|
d9c7e28444 | ||
|
|
5dd725f5f0 | ||
|
|
9405bc04b3 | ||
|
|
6ddaa46fc2 | ||
|
|
4400dc08ae | ||
|
|
e90095d854 | ||
|
|
c1154025e3 | ||
|
|
fcfe3c8353 | ||
|
|
143da8287a | ||
|
|
23d0f0b637 | ||
|
|
08af14793b | ||
|
|
6b6b871999 | ||
|
|
02d9c69c0d | ||
|
|
efd7aa8686 | ||
|
|
4abec105c3 | ||
|
|
723f891e37 | ||
|
|
889a7598f4 | ||
|
|
6a76844b4f | ||
|
|
d8550d7309 | ||
|
|
a86534bba6 | ||
|
|
a01a4ace7a | ||
|
|
070cdb3c6a | ||
|
|
13818643c1 | ||
|
|
61f60d2f88 | ||
|
|
3c1cf15c04 | ||
|
|
6624e2250c | ||
|
|
eff0a9c79f | ||
|
|
b814cf96d5 | ||
|
|
5ff5022af3 | ||
|
|
3623bd2994 | ||
|
|
cccbb11057 | ||
|
|
0de495cc3c | ||
|
|
c95db258bd | ||
|
|
2aa2759192 | ||
|
|
1fa4323498 | ||
|
|
fde74a329b | ||
|
|
319d4e7f05 | ||
|
|
ecb9d77a29 | ||
|
|
c7d5ee2654 | ||
|
|
a0e0c15913 | ||
|
|
4fda37f739 | ||
|
|
7b5e078731 | ||
|
|
7511d65aa9 | ||
|
|
18adf880d7 | ||
|
|
8df51cbc3f | ||
|
|
a8634a7fa0 | ||
|
|
0541c74f08 | ||
|
|
95f77d6c9a | ||
|
|
f469d54afa | ||
|
|
b4951bd02d | ||
|
|
f5137a5eed | ||
|
|
9ad28ce2c8 | ||
|
|
92aa5d65e2 | ||
|
|
d4c7b609b1 | ||
|
|
3c53e9a9c5 | ||
|
|
fb9726d4f8 | ||
|
|
84bdebeb8c | ||
|
|
8448b32bb1 | ||
|
|
5bf4712efd | ||
|
|
1d571c04a8 | ||
|
|
4aec552eeb | ||
|
|
a9c9808b49 | ||
|
|
0aff563810 | ||
|
|
a3ce5422d9 | ||
|
|
2f3010f2f0 | ||
|
|
b4dd6b5997 | ||
|
|
a739bcd2f8 | ||
|
|
ffbe7710b9 | ||
|
|
22a9fa0892 | ||
|
|
45ca17f09c | ||
|
|
3014937c08 | ||
|
|
c29c742a29 | ||
|
|
2e0d1d89bc | ||
|
|
c35ad8f18a | ||
|
|
2d289a1a0a | ||
|
|
53634954d6 | ||
|
|
b880107442 | ||
|
|
a25fbaa530 | ||
|
|
17c653afae | ||
|
|
3567ff10af | ||
|
|
f2b1c68fd3 | ||
|
|
a996e039a0 | ||
|
|
c2f300a13b | ||
|
|
a63c2adeac | ||
|
|
c14d057b33 | ||
|
|
268dabe5b3 | ||
|
|
89adf31fa9 | ||
|
|
1818159e6e | ||
|
|
39c4b2c6eb | ||
|
|
26f28aef94 | ||
|
|
14ff697e6d | ||
|
|
0024f03b36 | ||
|
|
7cf631f2ae | ||
|
|
bf6660abe7 | ||
|
|
aab7cbd0ba | ||
|
|
5e04c9e759 | ||
|
|
78deadd0b2 | ||
|
|
c88d99a660 | ||
|
|
ef5e133a06 | ||
|
|
97e2e04ef1 | ||
|
|
1c87d3aec9 | ||
|
|
5944bc1e98 | ||
|
|
a92b985363 | ||
|
|
a71d9f4aa2 | ||
|
|
438f5cd0f1 | ||
|
|
ccee9d48ba | ||
|
|
f7d2aad9dd | ||
|
|
420af6dacd | ||
|
|
0e05495bad | ||
|
|
d70fe9904e | ||
|
|
2591ff1f61 | ||
|
|
77407972c0 | ||
|
|
8e3dda40eb | ||
|
|
fcc0bcbd4c | ||
|
|
c60edb07f4 | ||
|
|
6cdf650c44 | ||
|
|
85651b4737 | ||
|
|
055d9f1052 | ||
|
|
7b92942e96 | ||
|
|
679739b1e3 | ||
|
|
7abd2a94e8 | ||
|
|
1f119edde1 | ||
|
|
ad86262437 | ||
|
|
07497cae07 | ||
|
|
c617b3afd4 | ||
|
|
bc27a94112 | ||
|
|
460492012e | ||
|
|
bc258e1d94 | ||
|
|
e0b0472bf2 | ||
|
|
71855ba554 | ||
|
|
229fe92b41 | ||
|
|
2b8cf51e09 | ||
|
|
bfffa918f7 | ||
|
|
a40bd2f790 | ||
|
|
b2382119da | ||
|
|
7995844fb9 | ||
|
|
5225e2dea3 | ||
|
|
0ba34e52e3 | ||
|
|
baaddcf782 | ||
|
|
ab2251197f | ||
|
|
3f9dc61322 | ||
|
|
421f43f6ec | ||
|
|
101cb10b18 | ||
|
|
265aa483ca | ||
|
|
d685787c50 | ||
|
|
0741c02a38 | ||
|
|
eb3c078e94 | ||
|
|
50c56fdc83 | ||
|
|
35331da9c9 | ||
|
|
3fc885ed73 | ||
|
|
d09d927b86 | ||
|
|
32b67c327c | ||
|
|
39827e802f | ||
|
|
cac649144f | ||
|
|
7903b11b9b | ||
|
|
8d17794793 | ||
|
|
175c47d8ff | ||
|
|
f3bdd37ecf | ||
|
|
5e943f7998 | ||
|
|
43555a801d |
+8
-10
@@ -15,10 +15,8 @@ install:
|
||||
- msmpisdk.msi /passive
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install METIS, use a mirror because the original source server is not always
|
||||
# up. Original url:
|
||||
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz
|
||||
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
|
||||
# Install METIS
|
||||
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
|
||||
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd metis-5.1.0
|
||||
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
|
||||
@@ -28,17 +26,17 @@ install:
|
||||
- cd ..
|
||||
|
||||
# Install hypre
|
||||
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/v2.19.0.tar.gz'
|
||||
- 7z x v2.19.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.19.0/src
|
||||
- cmake -H. -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/V2-10-0b.tar.gz'
|
||||
- 7z x V2-10-0b.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2-10-0b
|
||||
- cmake -H. -Bbuild -DHYPRE_USING_FEI=OFF -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- cmake --build build
|
||||
- cmake --build build --target install
|
||||
- cd ../..
|
||||
- cd ..
|
||||
|
||||
# MFEM
|
||||
before_build:
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_DIR=%cd%\hypre-2.19.0\src\hypre -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2-10-0b\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2-10-0b\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE
|
||||
|
||||
build_script:
|
||||
|
||||
+14
-12
@@ -163,6 +163,20 @@ miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_*
|
||||
|
||||
miniapps/hypsys/build
|
||||
miniapps/hypsys/errors.txt
|
||||
miniapps/hypsys/grid*
|
||||
miniapps/hypsys/hypsys
|
||||
miniapps/hypsys/initial*
|
||||
miniapps/hypsys/output
|
||||
miniapps/hypsys/phypsys
|
||||
miniapps/hypsys/pressure*
|
||||
miniapps/hypsys/results
|
||||
miniapps/hypsys/scripts/gridfunc-scatter
|
||||
miniapps/hypsys/ultimate*
|
||||
miniapps/hypsys/velocity*
|
||||
miniapps/hypsys/various
|
||||
|
||||
miniapps/meshing/mobius-strip
|
||||
miniapps/meshing/klein-bottle
|
||||
miniapps/meshing/toroid
|
||||
@@ -244,24 +258,12 @@ miniapps/navier/navier_3dfoc
|
||||
miniapps/navier/tgv_out*.txt
|
||||
miniapps/navier/*_output
|
||||
|
||||
miniapps/adjoint/cvsRoberts_ASAi_dns
|
||||
miniapps/adjoint/adjoint_advection_diffusion
|
||||
|
||||
# Unit test binary and outputs
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
tests/unit/punit_tests
|
||||
tests/unit/sedov_tests_*
|
||||
tests/unit/psedov_tests_*
|
||||
tests/unit/tmop_tests_*
|
||||
tests/unit/ptmop_tests_*
|
||||
tests/unit/cube.mesh
|
||||
tests/unit/star.mesh
|
||||
tests/unit/blade.mesh
|
||||
tests/unit/square01.mesh
|
||||
tests/unit/toroid-hex.mesh
|
||||
tests/unit/beam-hex-nurbs.mesh
|
||||
tests/unit/square-disc-nurbs.mesh
|
||||
|
||||
# Test script output
|
||||
tests/scripts/*.err
|
||||
|
||||
+33
-98
@@ -11,20 +11,11 @@
|
||||
|
||||
language: cpp
|
||||
|
||||
os: linux
|
||||
dist: bionic
|
||||
|
||||
stages:
|
||||
- checks
|
||||
- tests
|
||||
- optional
|
||||
|
||||
env:
|
||||
global:
|
||||
- HYPRE_ARCHIVE=v2.19.0.tar.gz
|
||||
HYPRE_URL=https://github.com/hypre-space/hypre/archive/$HYPRE_ARCHIVE
|
||||
HYPRE_TOP_DIR=hypre-2.19.0
|
||||
|
||||
jobs:
|
||||
include:
|
||||
|
||||
@@ -37,7 +28,6 @@ jobs:
|
||||
|
||||
- stage: checks
|
||||
os: linux
|
||||
dist: xenial
|
||||
name: "code-style"
|
||||
addons:
|
||||
apt:
|
||||
@@ -56,6 +46,9 @@ jobs:
|
||||
packages:
|
||||
- doxygen
|
||||
- graphviz
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
env: MPI=YES
|
||||
script:
|
||||
- cd ${TRAVIS_BUILD_DIR}
|
||||
- cd tests/scripts
|
||||
@@ -70,24 +63,13 @@ jobs:
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
env: MPI=YES
|
||||
before_script:
|
||||
script:
|
||||
- cd ${TRAVIS_BUILD_DIR}
|
||||
- mpicxx -v
|
||||
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
|
||||
- make all -j3
|
||||
- make test-noclean
|
||||
script:
|
||||
- cd tests/scripts
|
||||
- ./runtest gitignore
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a Lib ..; rm -rf * ; mv ../libmetis.a ../Lib .;
|
||||
rm -f Lib/*.{c,o}
|
||||
|
||||
# ========================
|
||||
# Optional Checks/Tests
|
||||
@@ -96,7 +78,6 @@ jobs:
|
||||
|
||||
- stage: optional
|
||||
name: "branch-history"
|
||||
if: branch != next
|
||||
# need full git history for the binary/big files check
|
||||
git:
|
||||
depth: false
|
||||
@@ -125,8 +106,6 @@ jobs:
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
@@ -135,8 +114,6 @@ jobs:
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
@@ -160,9 +137,9 @@ jobs:
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=2
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -191,9 +168,9 @@ jobs:
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=2
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -216,16 +193,16 @@ jobs:
|
||||
- cd ${TRAVIS_BUILD_DIR}/build
|
||||
- cmake ..
|
||||
-DMFEM_USE_MPI=ON
|
||||
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../$HYPRE_TOP_DIR/src/hypre
|
||||
-DHYPRE_DIR=${TRAVIS_BUILD_DIR}/../hypre-2.10.0b/src/hypre
|
||||
-DMFEM_MPI_NP=$NPROCS
|
||||
- make -j3 mfem examples
|
||||
- cd ${TRAVIS_BUILD_DIR}/build/tests/unit
|
||||
- make -j3
|
||||
- ctest --output-on-failure
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
@@ -241,43 +218,27 @@ jobs:
|
||||
# - parallel
|
||||
|
||||
- os: osx
|
||||
osx_image: xcode11.2
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
name: "Mac: Serial + Debug"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: osx
|
||||
osx_image: xcode11.2
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
name: "Mac: Serial"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
cache:
|
||||
ccache: true
|
||||
|
||||
- os: osx
|
||||
osx_image: xcode11.2
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
name: "Mac: Parallel + Debug"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
@@ -285,9 +246,9 @@ jobs:
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
@@ -296,13 +257,9 @@ jobs:
|
||||
rm -f Lib/*.{c,o}
|
||||
|
||||
- os: osx
|
||||
osx_image: xcode11.2
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
name: "Mac: Parallel"
|
||||
addons:
|
||||
homebrew:
|
||||
packages:
|
||||
- ccache
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
@@ -310,9 +267,9 @@ jobs:
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
ccache: true
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
@@ -326,19 +283,14 @@ before_install:
|
||||
# brew install open-mpi;
|
||||
# fi
|
||||
|
||||
# Disable ccache while building dependencies that are cached:
|
||||
- echo "before \$PATH = $PATH";
|
||||
export PATH=${PATH//\/usr\/lib\/ccache:/};
|
||||
echo "after \$PATH = $PATH"
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.6:
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.1:
|
||||
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
|
||||
mkdir -p $HOME/builds && cd $HOME/builds &&
|
||||
wget https://download.open-mpi.org/release/open-mpi/v2.1/openmpi-2.1.6.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.6.tar.bz2 &&
|
||||
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.1.tar.bz2 &&
|
||||
mkdir openmpi-build && cd openmpi-build &&
|
||||
../openmpi-2.1.6/configure --prefix=$HOME/local-cached &&
|
||||
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
|
||||
make -j3 all && make install;
|
||||
fi;
|
||||
PATH=$HOME/local-cached/bin:$PATH;
|
||||
@@ -383,28 +335,26 @@ install:
|
||||
|
||||
# hypre
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HYPRE_TOP_DIR/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget $HYPRE_URL;
|
||||
rm -rf $HYPRE_TOP_DIR;
|
||||
tar xvzf $HYPRE_ARCHIVE;
|
||||
cd $HYPRE_TOP_DIR/src;
|
||||
./configure --disable-fortran CC=mpicc CXX=mpic++;
|
||||
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
rm -rf hypre-2.10.0b;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j3;
|
||||
cd ../..;
|
||||
else
|
||||
echo "Reusing cached $HYPRE_TOP_DIR/";
|
||||
echo "Reusing cached hypre-2.10.0b/";
|
||||
fi;
|
||||
ln -s $HYPRE_TOP_DIR hypre;
|
||||
ln -s hypre-2.10.0b hypre;
|
||||
else
|
||||
echo "Serial build, not using hypre";
|
||||
fi
|
||||
|
||||
# METIS, use a mirror because the original source server is not always up.
|
||||
# Original url:
|
||||
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz
|
||||
# METIS
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e metis-4.0/libmetis.a ]; then
|
||||
wget https://mfem.github.io/tpls/metis-4.0.3.tar.gz;
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
|
||||
rm -rf metis-4.0;
|
||||
@@ -414,18 +364,6 @@ install:
|
||||
fi;
|
||||
fi
|
||||
|
||||
# Re-enable ccache on linux; enable ccache on mac os:
|
||||
- if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
export PATH="/usr/lib/ccache:$PATH";
|
||||
else
|
||||
if [ $TRAVIS_OS_NAME == "osx" ]; then
|
||||
export PATH="/usr/local/opt/ccache/libexec:$PATH";
|
||||
fi;
|
||||
fi
|
||||
|
||||
- printf "which \$CC = "; which $CC;
|
||||
printf "which \$CXX = "; which $CXX
|
||||
|
||||
script:
|
||||
# Compiler
|
||||
- if [ $MPI == "YES" ]; then
|
||||
@@ -446,9 +384,6 @@ script:
|
||||
if [ "$CODECOV" == "YES" ]; then
|
||||
CPPFLAGS="--coverage -g";
|
||||
fi;
|
||||
if [ "$TRAVIS_OS_NAME" != "linux" ] || [ "$DEBUG" == "YES" ]; then
|
||||
CPPFLAGS+=" -pedantic -Wall -Werror";
|
||||
fi
|
||||
|
||||
# Configure the library
|
||||
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG $MAKE_CXX_FLAG
|
||||
|
||||
@@ -33,9 +33,6 @@ Meshing improvements
|
||||
for example the periodic-annulus-sector and periodic-torus-sector files in
|
||||
the data directory.
|
||||
|
||||
- Added complete action of the TMOP Integrator to account for the spatial
|
||||
derivatives of discrete and analytic targets.
|
||||
|
||||
Performance improvements
|
||||
------------------------
|
||||
- Added support for explicit vectorization in the high-performance templated
|
||||
@@ -44,26 +41,12 @@ Performance improvements
|
||||
- x86 (SSE/AVX/AVX2/AVX512),
|
||||
- Power8 & Power9 (VSX),
|
||||
- BG/Q (QPX).
|
||||
These are disabled by default, and can be enabled with MFEM_USE_SIMD=YES.
|
||||
These are now enabled by default, and can be disabled with MFEM_USE_SIMD=NO.
|
||||
See the new file linalg/simd.hpp and the new directory linalg/simd.
|
||||
|
||||
Improved GPU capabilities
|
||||
-------------------------
|
||||
- Added support for Chebyshev accelerated polynomial smoother on GPU.
|
||||
- The TMOP mesh optimization algorithms were extended to GPU:
|
||||
- QualityMetric #1, #2 and #7 are available in 2D, #302, #303 and #321 in 3D
|
||||
- Both AnalyticAdaptTC and DiscreteAdaptTC TargetConstructor are available
|
||||
- Kernels for normalization and limiting have been added
|
||||
- The AdvectorCG now also support AssemblyLevel::PARTIAL
|
||||
|
||||
- Optimized AMD/HIP kernel support.
|
||||
|
||||
- Added a Full Assembly mode compatible with Device kernel execution. This
|
||||
assembly level builds on top of the current Element Assembly kernels to
|
||||
compute a global sparse matrix. All integrators supported by element assembly
|
||||
are also supported by full assembly. See the '-fa' option in Example 9.
|
||||
|
||||
- Added support for BlockOperator on GPU. See the updated Example 5.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
@@ -88,7 +71,7 @@ Discretization improvements
|
||||
and, in the continuous field case, arbitrary mesh edges and faces.
|
||||
|
||||
- Added new coefficient and vector coefficient classes for QuadratureFunctions.
|
||||
Additionally, new LinearForm integrators were also added which make use of
|
||||
Additionaly, new LinearForm integrators were also added which make use of
|
||||
these new QuadratureFunction coefficient classes.
|
||||
|
||||
- Added support face integrals on the boundaries of NURBS meshes.
|
||||
@@ -105,10 +88,6 @@ Linear and nonlinear solvers
|
||||
and solution during the solving process of an IterativeSolver after every
|
||||
iteration.
|
||||
|
||||
- Added support for the CVODES package in SUNDIALS which provides ODE
|
||||
solvers with sensitivity analysis capabilities. See the CVODESSolver
|
||||
class and the new adjoint miniapps below.
|
||||
|
||||
- Block arrays of parallel matrices can now be merged into a single parallel
|
||||
matrix with the function HypreParMatrixFromBlocks. This could be useful for
|
||||
solving block systems with parallel direct solvers such as STRUMPACK.
|
||||
@@ -136,14 +115,6 @@ New and updated examples and miniapps
|
||||
equations of incompressible fluid dynamics. See the miniapps/navier directory
|
||||
for more details.
|
||||
|
||||
- Added a new miniapps/adjoint directory with two miniapps demonstrating how to
|
||||
solve adjoint problems in MFEM using the CVODES package in SUNDIALS. Both of
|
||||
these miniapps require the MFEM_USE_SUNDIALS configuration option.
|
||||
* The cvsRoberts_ASAi_dns miniapp solves a backward adjoint problem for a
|
||||
system of ODEs, evaluating both forward and adjoint quadratures in serial.
|
||||
* The adjoint_advection_diffusion miniapp solves a backward adjoint problem
|
||||
for an advection diffusion PDE, evaluating adjoint quadratures in parallel.
|
||||
|
||||
- Ported Example 11p to SLEPc, to demonstrate solving the Laplace eigenvalue
|
||||
equation with the shift-and-invert spectral transformation method.
|
||||
|
||||
@@ -154,13 +125,11 @@ New and updated examples and miniapps
|
||||
- Added a new meshing miniapp, Minimal Surface, which solves Plateau's problem:
|
||||
the Dirichlet problem for the minimal surface equation.
|
||||
|
||||
- Added partial assembly support to Example 4/4p and Example 5/5p, with diagonal
|
||||
- Added partial assembly support to examples 4/4p and 5/5p, with diagonal
|
||||
preconditioning.
|
||||
|
||||
- Added full assembly support in Example 9/9p.
|
||||
|
||||
- Added a new test problem in Example 24/24p, demonstrating a mixed bilinear
|
||||
form for H1, H(curl), H(div) and L_2, with partial assembly support.
|
||||
- Added a new test problem in example 24/24p, demonstrating a mixed bilinear
|
||||
form for H(div) and L_2, with partial assembly support.
|
||||
|
||||
- Added weak Dirichlet boundary conditions (Nitsche) to the NURBS miniapp.
|
||||
|
||||
@@ -168,8 +137,6 @@ New and updated examples and miniapps
|
||||
mesh based on element attributes. Any newly exposed boundary elements are
|
||||
assigned attribute numbers related to the trimmed element attributes.
|
||||
|
||||
- Added device support in Example 5/5p.
|
||||
|
||||
Improved testing
|
||||
----------------
|
||||
- Added a GitLab pipeline that automates PR testing on supercomputing systems
|
||||
|
||||
+2
-2
@@ -211,10 +211,10 @@ endif()
|
||||
# SUNDIALS
|
||||
if (MFEM_USE_SUNDIALS)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
find_package(SUNDIALS REQUIRED NVector_Serial CVODES ARKODE KINSOL)
|
||||
find_package(SUNDIALS REQUIRED NVector_Serial CVODE ARKODE KINSOL)
|
||||
else()
|
||||
find_package(SUNDIALS REQUIRED
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODES ARKODE KINSOL)
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
||||
@@ -109,7 +109,6 @@ The MFEM source code has the following structure:
|
||||
├── linalg
|
||||
├── mesh
|
||||
├── miniapps
|
||||
│ ├── adjoint
|
||||
│ ├── common
|
||||
│ ├── electromagnetics
|
||||
│ ├── gslib
|
||||
|
||||
@@ -25,6 +25,5 @@ mfem_find_package(SUNDIALS SUNDIALS SUNDIALS_DIR
|
||||
ADD_COMPONENT NVector_ParHyp
|
||||
"include" nvector/nvector_parhyp.h "lib" sundials_nvecparhyp
|
||||
ADD_COMPONENT CVODE "include" cvode/cvode.h "lib" sundials_cvode
|
||||
ADD_COMPONENT CVODES "include" cvodes/cvodes.h "lib" sundials_cvodes
|
||||
ADD_COMPONENT ARKODE "include" arkode/arkode.h "lib" sundials_arkode
|
||||
ADD_COMPONENT KINSOL "include" kinsol/kinsol.h "lib" sundials_kinsol)
|
||||
|
||||
@@ -50,7 +50,7 @@ option(MFEM_USE_OCCA "Enable OCCA" OFF)
|
||||
option(MFEM_USE_RAJA "Enable RAJA" OFF)
|
||||
option(MFEM_USE_CEED "Enable CEED" OFF)
|
||||
option(MFEM_USE_UMPIRE "Enable Umpire" OFF)
|
||||
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" OFF)
|
||||
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" ON)
|
||||
option(MFEM_USE_ADIOS2 "Enable ADIOS2" OFF)
|
||||
|
||||
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
|
||||
@@ -88,8 +88,6 @@ set(METIS_DIR "${MFEM_DIR}/../metis-4.0" CACHE PATH "Path to the METIS library."
|
||||
|
||||
set(LIBUNWIND_DIR "" CACHE PATH "Path to Libunwind.")
|
||||
|
||||
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
|
||||
# and modify cmake variables for hypre for sundials
|
||||
set(SUNDIALS_DIR "${MFEM_DIR}/../sundials-5.0.0/instdir" CACHE PATH
|
||||
"Path to the SUNDIALS library.")
|
||||
# The following may be necessary, if SUNDIALS was built with KLU:
|
||||
|
||||
+4
-12
@@ -138,8 +138,7 @@ MFEM_USE_RAJA = NO
|
||||
MFEM_USE_OCCA = NO
|
||||
MFEM_USE_CEED = NO
|
||||
MFEM_USE_UMPIRE = NO
|
||||
MFEM_USE_CAMP = NO
|
||||
MFEM_USE_SIMD = NO
|
||||
MFEM_USE_SIMD = YES
|
||||
MFEM_USE_ADIOS2 = NO
|
||||
|
||||
# Compile and link options for zlib.
|
||||
@@ -190,12 +189,10 @@ OPENMP_LIB =
|
||||
POSIX_CLOCKS_LIB = -lrt
|
||||
|
||||
# SUNDIALS library configuration
|
||||
# For sundials_nvecparhyp and nvecparallel remember to build with MPI_ENABLED=ON
|
||||
# and modify cmake variables for hypre for sundials
|
||||
SUNDIALS_DIR = @MFEM_DIR@/../sundials-5.0.0/instdir
|
||||
SUNDIALS_OPT = -I$(SUNDIALS_DIR)/include
|
||||
SUNDIALS_LIB = -Wl,-rpath,$(SUNDIALS_DIR)/lib64 -L$(SUNDIALS_DIR)/lib64\
|
||||
-lsundials_arkode -lsundials_cvodes -lsundials_nvecserial -lsundials_kinsol
|
||||
-lsundials_arkode -lsundials_cvode -lsundials_nvecserial -lsundials_kinsol
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),YES)
|
||||
SUNDIALS_LIB += -lsundials_nvecparhyp -lsundials_nvecparallel
|
||||
@@ -342,9 +339,9 @@ GSLIB_DIR = @MFEM_DIR@/../gslib/build
|
||||
GSLIB_OPT = -I$(GSLIB_DIR)/include
|
||||
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
|
||||
|
||||
# CUDA library configuration
|
||||
# CUDA library configuration (currently not needed)
|
||||
CUDA_OPT =
|
||||
CUDA_LIB = -lcusparse
|
||||
CUDA_LIB =
|
||||
|
||||
# HIP library configuration (currently not needed)
|
||||
HIP_OPT =
|
||||
@@ -373,11 +370,6 @@ UMPIRE_DIR = @MFEM_DIR@/../umpire
|
||||
UMPIRE_OPT = -I$(UMPIRE_DIR)/include
|
||||
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -lumpire
|
||||
|
||||
# CAMP library configuration
|
||||
CAMP_DIR = @MFEM_DIR@/../camp
|
||||
CAMP_OPT = -I$(CAMP_DIR)/include
|
||||
CAMP_LIB = -L$(CAMP_DIR)/lib
|
||||
|
||||
# If YES, enable some informational messages
|
||||
VERBOSE = NO
|
||||
|
||||
|
||||
@@ -770,7 +770,6 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/examples/pumi \
|
||||
@MFEM_SOURCE_DIR@/examples/hiop \
|
||||
@MFEM_SOURCE_DIR@/examples/sundials \
|
||||
@MFEM_SOURCE_DIR@/miniapps/adjoint \
|
||||
@MFEM_SOURCE_DIR@/miniapps/common \
|
||||
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
|
||||
@MFEM_SOURCE_DIR@/miniapps/gslib \
|
||||
|
||||
@@ -88,8 +88,8 @@ namespace mfem {
|
||||
* - <a class="el" href="ex24p_8cpp_source.html">Example 24p</a>: parallel mixed finite element spaces and interpolators
|
||||
* - <a class="el" href="ex25_8cpp_source.html">Example 25</a>: simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex25p_8cpp_source.html">Example 25p</a>: parallel simulation of electromagnetic wave propagation using a Perfectly Matched Layer (PML)
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26_8cpp_source.html">Example 26</a>: multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
* - <a class="el" href="ex26p_8cpp_source.html">Example 26p</a>: parallel multigrid preconditioner for the Laplace problem using nodal H1 FEM
|
||||
*
|
||||
* <H4>SUNDIALS Examples</H4>
|
||||
* - Variants of Examples
|
||||
@@ -101,9 +101,6 @@ namespace mfem {
|
||||
* and
|
||||
* <a class="el" href="sundials_2ex16p_8cpp_source.html">16p</a>
|
||||
* demonstrating the use of MFEM's \link sundials.hpp SUNDIALS classes\endlink
|
||||
* - CVODES adjoint examples:
|
||||
* <a class="el" href="cvsRoberts__ASAi__dns_8cpp_source.html">serial ODE system</a>,
|
||||
* <a class="el" href="adjoint__advection__diffusion_8cpp_source.html">parallel advection-diffusion</a>
|
||||
*
|
||||
* <H4>PETSc Examples</H4>
|
||||
* - Variants of Examples
|
||||
@@ -143,9 +140,7 @@ namespace mfem {
|
||||
* - <a class="el" href="tesla_8cpp_source.html">Tesla</a>: simple magnetostatics simulation code
|
||||
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
|
||||
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
|
||||
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
|
||||
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
|
||||
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
|
||||
* - <a class="el" href="toroid_8cpp_source.html">Toroid</a>: generate simple toroidal meshes
|
||||
* - <a class="el" href="twist_8cpp_source.html">Twist</a>: generate simple periodic meshes
|
||||
@@ -162,6 +157,7 @@ namespace mfem {
|
||||
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
|
||||
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
|
||||
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
|
||||
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
|
||||
*
|
||||
|
||||
+31
-34
@@ -103,8 +103,8 @@ int main(int argc, char *argv[])
|
||||
// 3. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
|
||||
// the same code.
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 4. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
|
||||
@@ -112,10 +112,10 @@ int main(int argc, char *argv[])
|
||||
// elements.
|
||||
{
|
||||
int ref_levels =
|
||||
(int)floor(log(50000./mesh.GetNE())/log(2.)/dim);
|
||||
(int)floor(log(50000./mesh->GetNE())/log(2.)/dim);
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -123,70 +123,66 @@ int main(int argc, char *argv[])
|
||||
// Lagrange finite elements of the specified order. If order < 1, we
|
||||
// instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (mesh.GetNodes())
|
||||
else if (mesh->GetNodes())
|
||||
{
|
||||
fec = mesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
fec = mesh->GetNodes()->OwnFEC();
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
FiniteElementSpace fespace(&mesh, fec);
|
||||
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
|
||||
cout << "Number of finite element unknowns: "
|
||||
<< fespace.GetTrueVSize() << endl;
|
||||
<< fespace->GetTrueVSize() << endl;
|
||||
|
||||
// 6. Determine the list of true (i.e. conforming) essential boundary dofs.
|
||||
// In this example, the boundary conditions are defined by marking all
|
||||
// the boundary attributes from the mesh as essential (Dirichlet) and
|
||||
// converting them to a list of true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
if (mesh.bdr_attributes.Size())
|
||||
if (mesh->bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
Array<int> ess_bdr(mesh->bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 7. Set up the linear form b(.) which corresponds to the right-hand side of
|
||||
// the FEM linear system, which in this case is (1,phi_i) where phi_i are
|
||||
// the basis functions in the finite element fespace.
|
||||
LinearForm b(&fespace);
|
||||
LinearForm *b = new LinearForm(fespace);
|
||||
ConstantCoefficient one(1.0);
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b.Assemble();
|
||||
b->AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b->Assemble();
|
||||
|
||||
// 8. Define the solution vector x as a finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
GridFunction x(&fespace);
|
||||
GridFunction x(fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 9. Set up the bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
|
||||
// domain integrator.
|
||||
BilinearForm a(&fespace);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
BilinearForm *a = new BilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
|
||||
// 10. Assemble the bilinear form and the corresponding linear system,
|
||||
// applying any necessary transformations such as: eliminating boundary
|
||||
// conditions, applying conforming constraints for non-conforming AMR,
|
||||
// static condensation, etc.
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
|
||||
cout << "Size of linear system: " << A->Height() << endl;
|
||||
|
||||
@@ -207,9 +203,9 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
else // Jacobi preconditioning in partial assembly mode
|
||||
{
|
||||
if (UsesTensorBasis(fespace))
|
||||
if (UsesTensorBasis(*fespace))
|
||||
{
|
||||
OperatorJacobiSmoother M(a, ess_tdof_list);
|
||||
OperatorJacobiSmoother M(*a, ess_tdof_list);
|
||||
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
|
||||
}
|
||||
else
|
||||
@@ -219,13 +215,13 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 12. Recover the solution as a finite element grid function.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
|
||||
// 13. Save the refined mesh and the solution. This output can be viewed later
|
||||
// using GLVis: "glvis -m refined.mesh -g sol.gf".
|
||||
ofstream mesh_ofs("refined.mesh");
|
||||
mesh_ofs.precision(8);
|
||||
mesh.Print(mesh_ofs);
|
||||
mesh->Print(mesh_ofs);
|
||||
ofstream sol_ofs("sol.gf");
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
@@ -237,14 +233,15 @@ int main(int argc, char *argv[])
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << x << flush;
|
||||
sol_sock << "solution\n" << *mesh << x << flush;
|
||||
}
|
||||
|
||||
// 15. Free the used memory.
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
delete a;
|
||||
delete b;
|
||||
delete fespace;
|
||||
if (order > 0) { delete fec; }
|
||||
delete mesh;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
+35
-37
@@ -112,8 +112,8 @@ int main(int argc, char *argv[])
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement. We choose
|
||||
@@ -121,23 +121,23 @@ int main(int argc, char *argv[])
|
||||
// more than 10,000 elements.
|
||||
{
|
||||
int ref_levels =
|
||||
(int)floor(log(10000./mesh.GetNE())/log(2.)/dim);
|
||||
(int)floor(log(10000./mesh->GetNE())/log(2.)/dim);
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
mesh.Clear();
|
||||
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
{
|
||||
int par_ref_levels = 2;
|
||||
for (int l = 0; l < par_ref_levels; l++)
|
||||
{
|
||||
pmesh.UniformRefinement();
|
||||
pmesh->UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -145,16 +145,13 @@ int main(int argc, char *argv[])
|
||||
// use continuous Lagrange finite elements of the specified order. If
|
||||
// order < 1, we instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (pmesh.GetNodes())
|
||||
else if (pmesh->GetNodes())
|
||||
{
|
||||
fec = pmesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
fec = pmesh->GetNodes()->OwnFEC();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
@@ -163,10 +160,9 @@ int main(int argc, char *argv[])
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
ParFiniteElementSpace fespace(&pmesh, fec);
|
||||
HYPRE_Int size = fespace.GlobalTrueVSize();
|
||||
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
|
||||
HYPRE_Int size = fespace->GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element unknowns: " << size << endl;
|
||||
@@ -177,44 +173,44 @@ int main(int argc, char *argv[])
|
||||
// by marking all the boundary attributes from the mesh as essential
|
||||
// (Dirichlet) and converting them to a list of true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
if (pmesh->bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
|
||||
Array<int> ess_bdr(pmesh->bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 9. Set up the parallel linear form b(.) which corresponds to the
|
||||
// right-hand side of the FEM linear system, which in this case is
|
||||
// (1,phi_i) where phi_i are the basis functions in fespace.
|
||||
ParLinearForm b(&fespace);
|
||||
ParLinearForm *b = new ParLinearForm(fespace);
|
||||
ConstantCoefficient one(1.0);
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b.Assemble();
|
||||
b->AddDomainIntegrator(new DomainLFIntegrator(one));
|
||||
b->Assemble();
|
||||
|
||||
// 10. Define the solution vector x as a parallel finite element grid function
|
||||
// corresponding to fespace. Initialize x with initial guess of zero,
|
||||
// which satisfies the boundary conditions.
|
||||
ParGridFunction x(&fespace);
|
||||
ParGridFunction x(fespace);
|
||||
x = 0.0;
|
||||
|
||||
// 11. Set up the parallel bilinear form a(.,.) on the finite element space
|
||||
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
|
||||
// domain integrator.
|
||||
ParBilinearForm a(&fespace);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
ParBilinearForm *a = new ParBilinearForm(fespace);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a->AddDomainIntegrator(new DiffusionIntegrator(one));
|
||||
|
||||
// 12. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, static condensation, etc.
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
if (static_cond) { a->EnableStaticCondensation(); }
|
||||
a->Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
a->FormLinearSystem(ess_tdof_list, x, *b, A, X, B);
|
||||
|
||||
// 13. Solve the linear system A X = B.
|
||||
// * With full assembly, use the BoomerAMG preconditioner from hypre.
|
||||
@@ -222,9 +218,9 @@ int main(int argc, char *argv[])
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
{
|
||||
if (UsesTensorBasis(fespace))
|
||||
if (UsesTensorBasis(*fespace))
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
prec = new OperatorJacobiSmoother(*a, ess_tdof_list);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -242,7 +238,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
a->RecoverFEMSolution(X, *b, x);
|
||||
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
@@ -253,7 +249,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh.Print(mesh_ofs);
|
||||
pmesh->Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
@@ -268,14 +264,16 @@ int main(int argc, char *argv[])
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << x << flush;
|
||||
sol_sock << "solution\n" << *pmesh << x << flush;
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
delete a;
|
||||
delete b;
|
||||
delete fespace;
|
||||
if (order > 0) { delete fec; }
|
||||
delete pmesh;
|
||||
|
||||
MPI_Finalize();
|
||||
|
||||
return 0;
|
||||
|
||||
+28
-37
@@ -13,11 +13,6 @@
|
||||
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
|
||||
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
|
||||
//
|
||||
// With partial assembly:
|
||||
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1 -pa
|
||||
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2 -pa
|
||||
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0 -pa
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define and
|
||||
// solve simple complex-valued linear systems. It implements three
|
||||
// variants of a damped harmonic oscillator:
|
||||
@@ -81,7 +76,6 @@ int main(int argc, char *argv[])
|
||||
bool visualization = 1;
|
||||
bool herm_conv = true;
|
||||
bool exact_sol = true;
|
||||
bool pa = false;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -112,8 +106,6 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
@@ -290,7 +282,6 @@ int main(int argc, char *argv[])
|
||||
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
|
||||
|
||||
SesquilinearForm *a = new SesquilinearForm(fespace, conv);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -327,8 +318,6 @@ int main(int argc, char *argv[])
|
||||
// -Grad(a Div) - omega^2 b + omega c
|
||||
//
|
||||
BilinearForm *pcOp = new BilinearForm(fespace);
|
||||
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -359,8 +348,19 @@ int main(int argc, char *argv[])
|
||||
Vector B, U;
|
||||
|
||||
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
|
||||
u = 0.0;
|
||||
U = 0.0;
|
||||
|
||||
cout << "Size of linear system: " << A->Width() << endl << endl;
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
|
||||
{
|
||||
ComplexSparseMatrix * Asp =
|
||||
dynamic_cast<ComplexSparseMatrix*>(A.Ptr());
|
||||
|
||||
cout << "Size of linear system: "
|
||||
<< 2 * Asp->real().Width() << endl << endl;
|
||||
}
|
||||
|
||||
// 10. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// preconditioner based on the appropriate sparse smoother.
|
||||
@@ -368,8 +368,8 @@ int main(int argc, char *argv[])
|
||||
Array<int> blockOffsets;
|
||||
blockOffsets.SetSize(3);
|
||||
blockOffsets[0] = 0;
|
||||
blockOffsets[1] = A->Height() / 2;
|
||||
blockOffsets[2] = A->Height() / 2;
|
||||
blockOffsets[1] = PCOp.Ptr()->Height();
|
||||
blockOffsets[2] = PCOp.Ptr()->Height();
|
||||
blockOffsets.PartialSum();
|
||||
|
||||
BlockDiagonalPreconditioner BDP(blockOffsets);
|
||||
@@ -377,31 +377,22 @@ int main(int argc, char *argv[])
|
||||
Operator * pc_r = NULL;
|
||||
Operator * pc_i = NULL;
|
||||
|
||||
if (pa)
|
||||
double s = 1.0;
|
||||
switch (prob)
|
||||
{
|
||||
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
|
||||
case 0:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
|
||||
s = -1.0;
|
||||
break;
|
||||
case 2:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorHandle PCOp;
|
||||
pcOp->SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new GSSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
case 2:
|
||||
pc_r = new DSmoother(*PCOp.As<SparseMatrix>());
|
||||
break;
|
||||
default:
|
||||
break; // This should be unreachable
|
||||
}
|
||||
}
|
||||
double s = (prob != 1) ? 1.0 : -1.0;
|
||||
pc_i = new ScaledOperator(pc_r,
|
||||
(conv == ComplexOperator::HERMITIAN) ?
|
||||
s:-s);
|
||||
|
||||
+28
-39
@@ -13,11 +13,6 @@
|
||||
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
|
||||
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
|
||||
//
|
||||
// With partial assembly:
|
||||
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 1 -p 1 -pa
|
||||
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 1 -p 2 -pa
|
||||
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0 -pa
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define and
|
||||
// solve simple complex-valued linear systems. It implements three
|
||||
// variants of a damped harmonic oscillator:
|
||||
@@ -89,7 +84,6 @@ int main(int argc, char *argv[])
|
||||
bool visualization = 1;
|
||||
bool herm_conv = true;
|
||||
bool exact_sol = true;
|
||||
bool pa = false;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -122,8 +116,6 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
@@ -323,7 +315,6 @@ int main(int argc, char *argv[])
|
||||
ConstantCoefficient negMassCoef(omega_ * omega_ * epsilon_);
|
||||
|
||||
ParSesquilinearForm *a = new ParSesquilinearForm(fespace, conv);
|
||||
if (pa) { a->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -360,7 +351,6 @@ int main(int argc, char *argv[])
|
||||
// -Grad(a Div) - omega^2 b + omega c
|
||||
//
|
||||
ParBilinearForm *pcOp = new ParBilinearForm(fespace);
|
||||
if (pa) { pcOp->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
@@ -392,11 +382,19 @@ int main(int argc, char *argv[])
|
||||
Vector B, U;
|
||||
|
||||
a->FormLinearSystem(ess_tdof_list, u, b, A, U, B);
|
||||
u = 0.0;
|
||||
U = 0.0;
|
||||
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
ComplexHypreParMatrix * Ahyp =
|
||||
dynamic_cast<ComplexHypreParMatrix*>(A.Ptr());
|
||||
|
||||
cout << "Size of linear system: "
|
||||
<< 2 * fespace->GlobalTrueVSize() << endl << endl;
|
||||
<< 2 * Ahyp->real().GetGlobalNumRows() << endl << endl;
|
||||
}
|
||||
|
||||
// 12. Define and apply a parallel FGMRES solver for AU=B with a block
|
||||
@@ -406,8 +404,8 @@ int main(int argc, char *argv[])
|
||||
Array<int> blockTrueOffsets;
|
||||
blockTrueOffsets.SetSize(3);
|
||||
blockTrueOffsets[0] = 0;
|
||||
blockTrueOffsets[1] = A->Height() / 2;
|
||||
blockTrueOffsets[2] = A->Height() / 2;
|
||||
blockTrueOffsets[1] = PCOp.Ptr()->Height();
|
||||
blockTrueOffsets[2] = PCOp.Ptr()->Height();
|
||||
blockTrueOffsets.PartialSum();
|
||||
|
||||
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
|
||||
@@ -415,34 +413,25 @@ int main(int argc, char *argv[])
|
||||
Operator * pc_r = NULL;
|
||||
Operator * pc_i = NULL;
|
||||
|
||||
if (pa)
|
||||
switch (prob)
|
||||
{
|
||||
pc_r = new OperatorJacobiSmoother(*pcOp, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorHandle PCOp;
|
||||
pcOp->FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
case 0:
|
||||
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
break;
|
||||
case 2:
|
||||
if (dim == 2 )
|
||||
{
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
break;
|
||||
case 2:
|
||||
if (dim == 2 )
|
||||
{
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
else
|
||||
{
|
||||
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), fespace);
|
||||
}
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
pc_i = new ScaledOperator(pc_r,
|
||||
(conv == ComplexOperator::HERMITIAN) ?
|
||||
|
||||
+7
-86
@@ -7,7 +7,6 @@
|
||||
// ex24 -m ../data/beam-tet.mesh
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 1
|
||||
// ex24 -m ../data/beam-hex.mesh -o 2 -pa -p 2
|
||||
// ex24 -m ../data/escher.mesh
|
||||
// ex24 -m ../data/escher.mesh -o 2
|
||||
// ex24 -m ../data/fichera.mesh
|
||||
@@ -25,13 +24,12 @@
|
||||
// ex24 -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code illustrates usage of mixed finite element
|
||||
// spaces, with three variants:
|
||||
// spaces, with two variants:
|
||||
//
|
||||
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
|
||||
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
|
||||
// 3) (div v, q) for v in H(div) tested against q in L_2
|
||||
// 2) (div v, q) for v in H(div) tested against q in L_2
|
||||
//
|
||||
// Using different approaches, we project the gradient, curl, or
|
||||
// Using different approaches, we project the gradient or
|
||||
// divergence to the appropriate space.
|
||||
//
|
||||
// We recommend viewing examples 1, 3, and 5 before viewing this
|
||||
@@ -47,11 +45,8 @@ using namespace mfem;
|
||||
double p_exact(const Vector &x);
|
||||
void gradp_exact(const Vector &, Vector &);
|
||||
double div_gradp_exact(const Vector &x);
|
||||
void v_exact(const Vector &x, Vector &v);
|
||||
void curlv_exact(const Vector &x, Vector &cv);
|
||||
|
||||
int dim;
|
||||
double freq = 1.0, kappa;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
@@ -88,7 +83,6 @@ int main(int argc, char *argv[])
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
@@ -125,15 +119,10 @@ int main(int argc, char *argv[])
|
||||
trial_fec = new H1_FECollection(order, dim);
|
||||
test_fec = new ND_FECollection(order, dim);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
trial_fec = new ND_FECollection(order, dim);
|
||||
test_fec = new RT_FECollection(order-1, dim);
|
||||
}
|
||||
else
|
||||
{
|
||||
trial_fec = new RT_FECollection(order-1, dim);
|
||||
test_fec = new L2_FECollection(order-1, dim);
|
||||
trial_fec = new RT_FECollection(order - 1, dim);
|
||||
test_fec = new L2_FECollection(order - 1, dim);
|
||||
}
|
||||
|
||||
FiniteElementSpace trial_fes(mesh, trial_fec);
|
||||
@@ -147,12 +136,6 @@ int main(int argc, char *argv[])
|
||||
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
|
||||
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
|
||||
endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: "
|
||||
@@ -167,18 +150,12 @@ int main(int argc, char *argv[])
|
||||
GridFunction x(&test_fes);
|
||||
FunctionCoefficient p_coef(p_exact);
|
||||
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
|
||||
VectorFunctionCoefficient v_coef(sdim, v_exact);
|
||||
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
|
||||
FunctionCoefficient divgradp_coef(div_gradp_exact);
|
||||
|
||||
if (prob == 0)
|
||||
{
|
||||
gftrial.ProjectCoefficient(p_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
gftrial.ProjectCoefficient(v_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
gftrial.ProjectCoefficient(gradp_coef);
|
||||
@@ -202,11 +179,6 @@ int main(int argc, char *argv[])
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
|
||||
}
|
||||
else
|
||||
{
|
||||
a.AddDomainIntegrator(new MassIntegrator(one));
|
||||
@@ -272,10 +244,6 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
dlo.AddDomainInterpolator(new GradientInterpolator());
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
dlo.AddDomainInterpolator(new CurlInterpolator());
|
||||
}
|
||||
else
|
||||
{
|
||||
dlo.AddDomainInterpolator(new DivergenceInterpolator());
|
||||
@@ -290,10 +258,6 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
exact_proj.ProjectCoefficient(gradp_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
exact_proj.ProjectCoefficient(curlv_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
exact_proj.ProjectCoefficient(divgradp_coef);
|
||||
@@ -312,21 +276,8 @@ int main(int argc, char *argv[])
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
|
||||
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
|
||||
" ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
double errSol = x.ComputeL2Error(curlv_coef);
|
||||
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
|
||||
double errProj = exact_proj.ComputeL2Error(curlv_coef);
|
||||
|
||||
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in H(div): "
|
||||
"|| E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
else
|
||||
@@ -344,7 +295,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
|
||||
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v "
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
@@ -420,33 +371,3 @@ double div_gradp_exact(const Vector &x)
|
||||
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void v_exact(const Vector &x, Vector &v)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(2));
|
||||
v(2) = sin(kappa * x(0));
|
||||
}
|
||||
else
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(0));
|
||||
if (x.Size() == 3) { v(2) = 0.0; }
|
||||
}
|
||||
}
|
||||
|
||||
void curlv_exact(const Vector &x, Vector &cv)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
cv(0) = -kappa * cos(kappa * x(2));
|
||||
cv(1) = -kappa * cos(kappa * x(0));
|
||||
cv(2) = -kappa * cos(kappa * x(1));
|
||||
}
|
||||
else
|
||||
{
|
||||
cv = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
+11
-93
@@ -6,8 +6,7 @@
|
||||
// mpirun -np 4 ex24p -m ../data/square-disc.mesh -o 2
|
||||
// mpirun -np 4 ex24p -m ../data/beam-tet.mesh
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 1
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -pa -p 2
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -o 2 -p 1 -pa
|
||||
// mpirun -np 4 ex24p -m ../data/escher.mesh
|
||||
// mpirun -np 4 ex24p -m ../data/escher.mesh -o 2
|
||||
// mpirun -np 4 ex24p -m ../data/fichera.mesh
|
||||
@@ -25,13 +24,12 @@
|
||||
// mpirun -np 4 ex24p -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code illustrates usage of mixed finite element
|
||||
// spaces, with three variants:
|
||||
// spaces, with two variants:
|
||||
//
|
||||
// 1) (grad p, u) for p in H^1 tested against u in H(curl)
|
||||
// 2) (curl v, u) for v in H(curl) tested against u in H(div), 3D
|
||||
// 3) (div v, q) for v in H(div) tested against q in L_2
|
||||
// 2) (div v, q) for v in H(div) tested against q in L_2
|
||||
//
|
||||
// Using different approaches, we project the gradient, curl, or
|
||||
// Using different approaches, we project the gradient or
|
||||
// divergence to the appropriate space.
|
||||
//
|
||||
// We recommend viewing examples 1, 3, and 5 before viewing this
|
||||
@@ -47,11 +45,8 @@ using namespace mfem;
|
||||
double p_exact(const Vector &x);
|
||||
void gradp_exact(const Vector &, Vector &);
|
||||
double div_gradp_exact(const Vector &x);
|
||||
void v_exact(const Vector &x, Vector &v);
|
||||
void curlv_exact(const Vector &x, Vector &cv);
|
||||
|
||||
int dim;
|
||||
double freq = 1.0, kappa;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
@@ -101,7 +96,6 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
kappa = freq * M_PI;
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
@@ -153,15 +147,10 @@ int main(int argc, char *argv[])
|
||||
trial_fec = new H1_FECollection(order, dim);
|
||||
test_fec = new ND_FECollection(order, dim);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
trial_fec = new ND_FECollection(order, dim);
|
||||
test_fec = new RT_FECollection(order-1, dim);
|
||||
}
|
||||
else
|
||||
{
|
||||
trial_fec = new RT_FECollection(order-1, dim);
|
||||
test_fec = new L2_FECollection(order-1, dim);
|
||||
trial_fec = new RT_FECollection(order - 1, dim);
|
||||
test_fec = new L2_FECollection(order - 1, dim);
|
||||
}
|
||||
|
||||
ParFiniteElementSpace trial_fes(pmesh, trial_fec);
|
||||
@@ -177,12 +166,6 @@ int main(int argc, char *argv[])
|
||||
cout << "Number of Nedelec finite element unknowns: " << test_size << endl;
|
||||
cout << "Number of H1 finite element unknowns: " << trial_size << endl;
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
cout << "Number of Nedelec finite element unknowns: " << trial_size << endl;
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: " << test_size <<
|
||||
endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "Number of Raviart-Thomas finite element unknowns: "
|
||||
@@ -198,18 +181,12 @@ int main(int argc, char *argv[])
|
||||
ParGridFunction x(&test_fes);
|
||||
FunctionCoefficient p_coef(p_exact);
|
||||
VectorFunctionCoefficient gradp_coef(sdim, gradp_exact);
|
||||
VectorFunctionCoefficient v_coef(sdim, v_exact);
|
||||
VectorFunctionCoefficient curlv_coef(sdim, curlv_exact);
|
||||
FunctionCoefficient divgradp_coef(div_gradp_exact);
|
||||
|
||||
if (prob == 0)
|
||||
{
|
||||
gftrial.ProjectCoefficient(p_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
gftrial.ProjectCoefficient(v_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
gftrial.ProjectCoefficient(gradp_coef);
|
||||
@@ -233,11 +210,6 @@ int main(int argc, char *argv[])
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorGradientIntegrator(one));
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(one));
|
||||
a_mixed.AddDomainIntegrator(new MixedVectorCurlIntegrator(one));
|
||||
}
|
||||
else
|
||||
{
|
||||
a.AddDomainIntegrator(new MassIntegrator(one));
|
||||
@@ -321,10 +293,6 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
dlo.AddDomainInterpolator(new GradientInterpolator());
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
dlo.AddDomainInterpolator(new CurlInterpolator());
|
||||
}
|
||||
else
|
||||
{
|
||||
dlo.AddDomainInterpolator(new DivergenceInterpolator());
|
||||
@@ -339,10 +307,6 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
exact_proj.ProjectCoefficient(gradp_coef);
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
exact_proj.ProjectCoefficient(curlv_coef);
|
||||
}
|
||||
else
|
||||
{
|
||||
exact_proj.ProjectCoefficient(divgradp_coef);
|
||||
@@ -360,27 +324,11 @@ int main(int argc, char *argv[])
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl)"
|
||||
": || E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad"
|
||||
" p ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
}
|
||||
else if (prob == 1)
|
||||
{
|
||||
double errSol = x.ComputeL2Error(curlv_coef);
|
||||
double errInterp = discreteInterpolant.ComputeL2Error(curlv_coef);
|
||||
double errProj = exact_proj.ComputeL2Error(curlv_coef);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\n Solution of (E_h,w) = (curl v_h,w) for E_h and w in "
|
||||
"H(div): || E_h - curl v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Curl interpolant E_h = curl v_h in H(div): || E_h - curl v "
|
||||
cout << "\n Solution of (E_h,v) = (grad p_h,v) for E_h and v in H(curl): "
|
||||
"|| E_h - grad p ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Gradient interpolant E_h = grad p_h in H(curl): || E_h - grad p"
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection E_h of exact curl v in H(div): || E_h - curl v "
|
||||
cout << " Projection E_h of exact grad p in H(curl): || E_h - grad p "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
}
|
||||
@@ -402,7 +350,7 @@ int main(int argc, char *argv[])
|
||||
cout << "\n Solution of (f_h,q) = (div v_h,q) for f_h and q in L_2: "
|
||||
"|| f_h - div v ||_{L_2} = " << errSol << '\n' << endl;
|
||||
cout << " Divergence interpolant f_h = div v_h in L_2: || f_h - div v"
|
||||
" ||_{L_2} = " << errInterp << '\n' << endl;
|
||||
"||_{L_2} = " << errInterp << '\n' << endl;
|
||||
cout << " Projection f_h of exact div v in L_2: || f_h - div v "
|
||||
"||_{L_2} = " << errProj << '\n' << endl;
|
||||
}
|
||||
@@ -488,33 +436,3 @@ double div_gradp_exact(const Vector &x)
|
||||
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void v_exact(const Vector &x, Vector &v)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(2));
|
||||
v(2) = sin(kappa * x(0));
|
||||
}
|
||||
else
|
||||
{
|
||||
v(0) = sin(kappa * x(1));
|
||||
v(1) = sin(kappa * x(0));
|
||||
if (x.Size() == 3) { v(2) = 0.0; }
|
||||
}
|
||||
}
|
||||
|
||||
void curlv_exact(const Vector &x, Vector &cv)
|
||||
{
|
||||
if (dim == 3)
|
||||
{
|
||||
cv(0) = -kappa * cos(kappa * x(2));
|
||||
cv(1) = -kappa * cos(kappa * x(0));
|
||||
cv(2) = -kappa * cos(kappa * x(1));
|
||||
}
|
||||
else
|
||||
{
|
||||
cv = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
+23
-15
@@ -389,22 +389,27 @@ int main(int argc, char *argv[])
|
||||
// applying any necessary transformations such as: assembly, eliminating
|
||||
// boundary conditions, applying conforming constraints for
|
||||
// non-conforming AMR, etc.
|
||||
a.Assemble(0);
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
OperatorHandle Ah;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
|
||||
// 13. Solve using a direct or an iterative solver
|
||||
// 13. Transform to monolithic SparseMatrix
|
||||
SparseMatrix *A = Ah.As<ComplexSparseMatrix>()->GetSystemMatrix();
|
||||
|
||||
cout << "Size of linear system: " << A->Height() << endl;
|
||||
|
||||
// 14. Solve using a direct or an iterative solver
|
||||
#ifdef MFEM_USE_SUITESPARSE
|
||||
{
|
||||
ComplexUMFPackSolver csolver(*A.As<ComplexSparseMatrix>());
|
||||
csolver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
csolver.SetPrintLevel(1);
|
||||
csolver.Mult(B, X);
|
||||
UMFPackSolver solver(*A);
|
||||
solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
solver.Mult(B, X);
|
||||
}
|
||||
#else
|
||||
// 13a. Set up the Bilinear form a(.,.) for the preconditioner
|
||||
|
||||
// 14a. Set up the Bilinear form a(.,.) for the preconditioner
|
||||
//
|
||||
// In Comp
|
||||
// Domain: 1/mu (Curl E, Curl F) + omega^2 * epsilon (E,F)
|
||||
@@ -432,10 +437,10 @@ int main(int argc, char *argv[])
|
||||
|
||||
prec.Assemble();
|
||||
|
||||
OperatorPtr PCOpAh;
|
||||
OperatorHandle PCOpAh;
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// 13b. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// 14b. Define and apply a GMRES solver for AU=B with a block diagonal
|
||||
// preconditioner based on the Gauss-Seidel sparse smoother.
|
||||
Array<int> offsets(3);
|
||||
offsets[0] = 0;
|
||||
@@ -462,15 +467,17 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
#endif
|
||||
|
||||
// 14. Recover the solution as a finite element grid function and compute the
|
||||
// 15. Recover the solution as a finite element grid function and compute the
|
||||
// errors if the exact solution is known.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// If exact is known compute the error
|
||||
if (exact_known)
|
||||
{
|
||||
ComplexGridFunction x_gf(fespace);
|
||||
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
|
||||
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
|
||||
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
|
||||
int order_quad = max(2, 2 * order + 1);
|
||||
const IntegrationRule *irs[Geometry::NumGeom];
|
||||
for (int i = 0; i < Geometry::NumGeom; ++i)
|
||||
@@ -499,7 +506,7 @@ int main(int argc, char *argv[])
|
||||
<< sqrt(L2Error_Re*L2Error_Re + L2Error_Im*L2Error_Im) << "\n\n";
|
||||
}
|
||||
|
||||
// 15. Save the refined mesh and the solution. This output can be viewed
|
||||
// 16. Save the refined mesh and the solution. This output can be viewed
|
||||
// later using GLVis: "glvis -m mesh -g sol".
|
||||
{
|
||||
ofstream mesh_ofs("ex25.mesh");
|
||||
@@ -514,7 +521,7 @@ int main(int argc, char *argv[])
|
||||
x.imag().Save(sol_i_ofs);
|
||||
}
|
||||
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
// 17. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
// Define visualization keys for GLVis (see GLVis documentation)
|
||||
@@ -565,7 +572,8 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
// 18. Free the used memory.
|
||||
delete A;
|
||||
delete pml;
|
||||
delete fespace;
|
||||
delete fec;
|
||||
|
||||
+16
-7
@@ -419,15 +419,21 @@ int main(int argc, char *argv[])
|
||||
// constraints for non-conforming AMR, etc.
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr Ah;
|
||||
OperatorHandle Ah;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
|
||||
|
||||
// 15. Solve using a direct or an iterative solver
|
||||
// 15. Transform to monolithic HypreParMatrix
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Size of linear system: " << A->GetGlobalNumRows() << endl;
|
||||
}
|
||||
|
||||
// 16. Solve using a direct or an iterative solver
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
{
|
||||
// Transform to monolithic HypreParMatrix
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
SuperLURowLocMatrix SA(*A);
|
||||
SuperLUSolver superlu(MPI_COMM_WORLD);
|
||||
superlu.SetPrintStatistics(false);
|
||||
@@ -435,9 +441,9 @@ int main(int argc, char *argv[])
|
||||
superlu.SetColumnPermutation(superlu::PARMETIS);
|
||||
superlu.SetOperator(SA);
|
||||
superlu.Mult(B, X);
|
||||
delete A;
|
||||
}
|
||||
#else
|
||||
|
||||
// 16a. Set up the parallel Bilinear form a(.,.) for the preconditioner
|
||||
//
|
||||
// In Comp
|
||||
@@ -466,7 +472,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
prec.Assemble();
|
||||
|
||||
OperatorPtr PCOpAh;
|
||||
OperatorHandle PCOpAh;
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// 16b. Define and apply a parallel GMRES solver for AU=B with a block
|
||||
@@ -490,7 +496,7 @@ int main(int argc, char *argv[])
|
||||
gmres.SetMaxIter(2000);
|
||||
gmres.SetRelTol(1e-5);
|
||||
gmres.SetAbsTol(0.0);
|
||||
gmres.SetOperator(*Ah);
|
||||
gmres.SetOperator(*A);
|
||||
gmres.SetPreconditioner(BlockAMS);
|
||||
gmres.Mult(B, X);
|
||||
}
|
||||
@@ -503,8 +509,10 @@ int main(int argc, char *argv[])
|
||||
// If exact is known compute the error
|
||||
if (exact_known)
|
||||
{
|
||||
ParComplexGridFunction x_gf(fespace);
|
||||
VectorFunctionCoefficient E_ex_Re(dim, E_exact_Re);
|
||||
VectorFunctionCoefficient E_ex_Im(dim, E_exact_Im);
|
||||
x_gf.ProjectCoefficient(E_ex_Re, E_ex_Im);
|
||||
int order_quad = max(2, 2 * order + 1);
|
||||
const IntegrationRule *irs[Geometry::NumGeom];
|
||||
for (int i = 0; i < Geometry::NumGeom; ++i)
|
||||
@@ -621,6 +629,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 20. Free the used memory.
|
||||
delete A;
|
||||
delete pml;
|
||||
delete fespace;
|
||||
delete fec;
|
||||
|
||||
+17
-36
@@ -11,12 +11,6 @@
|
||||
// ex5 -m ../data/escher.mesh
|
||||
// ex5 -m ../data/fichera.mesh
|
||||
//
|
||||
// Device sample runs:
|
||||
// ex5 -m ../data/star.mesh -pa -d cuda
|
||||
// ex5 -m ../data/star.mesh -pa -d raja-cuda
|
||||
// ex5 -m ../data/star.mesh -pa -d raja-omp
|
||||
// ex5 -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple 2D/3D mixed Darcy problem
|
||||
// corresponding to the saddle point system
|
||||
// k*u + grad p = f
|
||||
@@ -56,7 +50,6 @@ int main(int argc, char *argv[])
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int order = 1;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
@@ -66,8 +59,6 @@ int main(int argc, char *argv[])
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
@@ -79,18 +70,13 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
device.Print();
|
||||
|
||||
// 3. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// 2. Read the mesh from the given mesh file. We can handle triangular,
|
||||
// quadrilateral, tetrahedral, hexahedral, surface and volume meshes with
|
||||
// the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 4. Refine the mesh to increase the resolution. In this example we do
|
||||
// 3. Refine the mesh to increase the resolution. In this example we do
|
||||
// 'ref_levels' of uniform refinement. We choose 'ref_levels' to be the
|
||||
// largest number that gives a final mesh with no more than 10,000
|
||||
// elements.
|
||||
@@ -103,7 +89,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Define a finite element space on the mesh. Here we use the
|
||||
// 4. Define a finite element space on the mesh. Here we use the
|
||||
// Raviart-Thomas finite elements of the specified order.
|
||||
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
|
||||
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
|
||||
@@ -111,7 +97,7 @@ int main(int argc, char *argv[])
|
||||
FiniteElementSpace *R_space = new FiniteElementSpace(mesh, hdiv_coll);
|
||||
FiniteElementSpace *W_space = new FiniteElementSpace(mesh, l2_coll);
|
||||
|
||||
// 6. Define the BlockStructure of the problem, i.e. define the array of
|
||||
// 5. Define the BlockStructure of the problem, i.e. define the array of
|
||||
// offsets for each variable. The last component of the Array is the sum
|
||||
// of the dimensions of each block.
|
||||
Array<int> block_offsets(3); // number of variables + 1
|
||||
@@ -126,7 +112,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "dim(R+W) = " << block_offsets.Last() << "\n";
|
||||
std::cout << "***********************************************************\n";
|
||||
|
||||
// 7. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
// 6. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
ConstantCoefficient k(1.0);
|
||||
|
||||
VectorFunctionCoefficient fcoeff(dim, fFun);
|
||||
@@ -136,28 +122,25 @@ int main(int argc, char *argv[])
|
||||
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
|
||||
FunctionCoefficient pcoeff(pFun_ex);
|
||||
|
||||
// 8. Allocate memory (x, rhs) for the analytical solution and the right hand
|
||||
// 7. Allocate memory (x, rhs) for the analytical solution and the right hand
|
||||
// side. Define the GridFunction u,p for the finite element solution and
|
||||
// linear forms fform and gform for the right hand side. The data
|
||||
// allocated by x and rhs are passed as a reference to the grid functions
|
||||
// (u,p) and the linear forms (fform, gform).
|
||||
MemoryType mt = device.GetMemoryType();
|
||||
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
|
||||
BlockVector x(block_offsets), rhs(block_offsets);
|
||||
|
||||
LinearForm *fform(new LinearForm);
|
||||
fform->Update(R_space, rhs.GetBlock(0), 0);
|
||||
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
|
||||
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
|
||||
fform->Assemble();
|
||||
fform->SyncAliasMemory(rhs);
|
||||
|
||||
LinearForm *gform(new LinearForm);
|
||||
gform->Update(W_space, rhs.GetBlock(1), 0);
|
||||
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
|
||||
gform->Assemble();
|
||||
gform->SyncAliasMemory(rhs);
|
||||
|
||||
// 9. Assemble the finite element matrices for the Darcy operator
|
||||
// 8. Assemble the finite element matrices for the Darcy operator
|
||||
//
|
||||
// D = [ M B^T ]
|
||||
// [ B 0 ]
|
||||
@@ -202,7 +185,7 @@ int main(int argc, char *argv[])
|
||||
darcyOp.SetBlock(1,0, &B);
|
||||
}
|
||||
|
||||
// 10. Construct the operators for preconditioner
|
||||
// 9. Construct the operators for preconditioner
|
||||
//
|
||||
// P = [ diag(M) 0 ]
|
||||
// [ 0 B diag(M)^-1 B^T ]
|
||||
@@ -219,11 +202,10 @@ int main(int argc, char *argv[])
|
||||
if (pa)
|
||||
{
|
||||
mVarf->AssembleDiagonal(Md);
|
||||
auto Md_host = Md.HostRead();
|
||||
Vector invMd(mVarf->Height());
|
||||
for (int i=0; i<mVarf->Height(); ++i)
|
||||
{
|
||||
invMd(i) = 1.0 / Md_host[i];
|
||||
invMd(i) = 1.0 / Md(i);
|
||||
}
|
||||
|
||||
Vector BMBt_diag(bVarf->Height());
|
||||
@@ -264,7 +246,7 @@ int main(int argc, char *argv[])
|
||||
darcyPrec.SetDiagonalBlock(0, invM);
|
||||
darcyPrec.SetDiagonalBlock(1, invS);
|
||||
|
||||
// 11. Solve the linear system with MINRES.
|
||||
// 10. Solve the linear system with MINRES.
|
||||
// Check the norm of the unpreconditioned residual.
|
||||
int maxIter(1000);
|
||||
double rtol(1.e-6);
|
||||
@@ -281,7 +263,6 @@ int main(int argc, char *argv[])
|
||||
solver.SetPrintLevel(1);
|
||||
x = 0.0;
|
||||
solver.Mult(rhs, x);
|
||||
if (device.IsEnabled()) { x.HostRead(); }
|
||||
chrono.Stop();
|
||||
|
||||
if (solver.GetConverged())
|
||||
@@ -292,7 +273,7 @@ int main(int argc, char *argv[])
|
||||
<< " iterations. Residual norm is " << solver.GetFinalNorm() << ".\n";
|
||||
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
|
||||
|
||||
// 12. Create the grid functions u and p. Compute the L2 error norms.
|
||||
// 11. Create the grid functions u and p. Compute the L2 error norms.
|
||||
GridFunction u, p;
|
||||
u.MakeRef(R_space, x.GetBlock(0), 0);
|
||||
p.MakeRef(W_space, x.GetBlock(1), 0);
|
||||
@@ -312,7 +293,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "|| u_h - u_ex || / || u_ex || = " << err_u / norm_u << "\n";
|
||||
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
|
||||
|
||||
// 13. Save the mesh and the solution. This output can be viewed later using
|
||||
// 12. Save the mesh and the solution. This output can be viewed later using
|
||||
// GLVis: "glvis -m ex5.mesh -g sol_u.gf" or "glvis -m ex5.mesh -g
|
||||
// sol_p.gf".
|
||||
{
|
||||
@@ -329,13 +310,13 @@ int main(int argc, char *argv[])
|
||||
p.Save(p_ofs);
|
||||
}
|
||||
|
||||
// 14. Save data in the VisIt format
|
||||
// 13. Save data in the VisIt format
|
||||
VisItDataCollection visit_dc("Example5", mesh);
|
||||
visit_dc.RegisterField("velocity", &u);
|
||||
visit_dc.RegisterField("pressure", &p);
|
||||
visit_dc.Save();
|
||||
|
||||
// 15. Save data in the ParaView format
|
||||
// 14. Save data in the ParaView format
|
||||
ParaViewDataCollection paraview_dc("Example5", mesh);
|
||||
paraview_dc.SetPrefixPath("ParaView");
|
||||
paraview_dc.SetLevelsOfDetail(order);
|
||||
@@ -347,7 +328,7 @@ int main(int argc, char *argv[])
|
||||
paraview_dc.RegisterField("pressure",&p);
|
||||
paraview_dc.Save();
|
||||
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
// 15. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
@@ -360,7 +341,7 @@ int main(int argc, char *argv[])
|
||||
p_sock << "solution\n" << *mesh << p << "window_title 'Pressure'" << endl;
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
// 16. Free the used memory.
|
||||
delete fform;
|
||||
delete gform;
|
||||
delete invM;
|
||||
|
||||
+21
-42
@@ -11,12 +11,6 @@
|
||||
// mpirun -np 4 ex5p -m ../data/escher.mesh
|
||||
// mpirun -np 4 ex5p -m ../data/fichera.mesh
|
||||
//
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d cuda
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-cuda
|
||||
// mpirun -np 4 ex5p -m ../data/star.mesh -r 2 -pa -d raja-omp
|
||||
// mpirun -np 4 ex5p -m ../data/beam-hex.mesh -pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple 2D/3D mixed Darcy problem
|
||||
// corresponding to the saddle point system
|
||||
// k*u + grad p = f
|
||||
@@ -66,7 +60,6 @@ int main(int argc, char *argv[])
|
||||
int order = 1;
|
||||
bool par_format = false;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = 1;
|
||||
bool adios2 = false;
|
||||
|
||||
@@ -82,8 +75,6 @@ int main(int argc, char *argv[])
|
||||
"Format to use when saving the results for VisIt.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
@@ -105,18 +96,13 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
if (myid == 0) { device.Print(); }
|
||||
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// 3. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// 4. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement. We choose
|
||||
// 'ref_levels' to be the largest number that gives a final mesh with no
|
||||
// more than 10,000 elements, unless the user specifies it as input.
|
||||
@@ -132,7 +118,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// 5. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
|
||||
@@ -145,7 +131,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
}
|
||||
|
||||
// 7. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// 6. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// use the Raviart-Thomas finite elements of the specified order.
|
||||
FiniteElementCollection *hdiv_coll(new RT_FECollection(order, dim));
|
||||
FiniteElementCollection *l2_coll(new L2_FECollection(order, dim));
|
||||
@@ -165,7 +151,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "***********************************************************\n";
|
||||
}
|
||||
|
||||
// 8. Define the two BlockStructure of the problem. block_offsets is used
|
||||
// 7. Define the two BlockStructure of the problem. block_offsets is used
|
||||
// for Vector based on dof (like ParGridFunction or ParLinearForm),
|
||||
// block_trueOffstes is used for Vector based on trueDof (HypreParVector
|
||||
// for the rhs and solution of the linear system). The offsets computed
|
||||
@@ -182,7 +168,7 @@ int main(int argc, char *argv[])
|
||||
block_trueOffsets[2] = W_space->TrueVSize();
|
||||
block_trueOffsets.PartialSum();
|
||||
|
||||
// 9. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
// 8. Define the coefficients, analytical solution, and rhs of the PDE.
|
||||
ConstantCoefficient k(1.0);
|
||||
|
||||
VectorFunctionCoefficient fcoeff(dim, fFun);
|
||||
@@ -192,30 +178,25 @@ int main(int argc, char *argv[])
|
||||
VectorFunctionCoefficient ucoeff(dim, uFun_ex);
|
||||
FunctionCoefficient pcoeff(pFun_ex);
|
||||
|
||||
// 10. Define the parallel grid function and parallel linear forms, solution
|
||||
// vector and rhs.
|
||||
MemoryType mt = device.GetMemoryType();
|
||||
BlockVector x(block_offsets, mt), rhs(block_offsets, mt);
|
||||
BlockVector trueX(block_trueOffsets, mt), trueRhs(block_trueOffsets, mt);
|
||||
// 9. Define the parallel grid function and parallel linear forms, solution
|
||||
// vector and rhs.
|
||||
BlockVector x(block_offsets), rhs(block_offsets);
|
||||
BlockVector trueX(block_trueOffsets), trueRhs(block_trueOffsets);
|
||||
|
||||
ParLinearForm *fform(new ParLinearForm);
|
||||
fform->Update(R_space, rhs.GetBlock(0), 0);
|
||||
fform->AddDomainIntegrator(new VectorFEDomainLFIntegrator(fcoeff));
|
||||
fform->AddBoundaryIntegrator(new VectorFEBoundaryFluxLFIntegrator(fnatcoeff));
|
||||
fform->Assemble();
|
||||
fform->SyncAliasMemory(rhs);
|
||||
fform->ParallelAssemble(trueRhs.GetBlock(0));
|
||||
trueRhs.GetBlock(0).SyncAliasMemory(trueRhs);
|
||||
|
||||
ParLinearForm *gform(new ParLinearForm);
|
||||
gform->Update(W_space, rhs.GetBlock(1), 0);
|
||||
gform->AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
|
||||
gform->Assemble();
|
||||
gform->SyncAliasMemory(rhs);
|
||||
gform->ParallelAssemble(trueRhs.GetBlock(1));
|
||||
trueRhs.GetBlock(1).SyncAliasMemory(trueRhs);
|
||||
|
||||
// 11. Assemble the finite element matrices for the Darcy operator
|
||||
// 10. Assemble the finite element matrices for the Darcy operator
|
||||
//
|
||||
// D = [ M B^T ]
|
||||
// [ B 0 ]
|
||||
@@ -268,7 +249,7 @@ int main(int argc, char *argv[])
|
||||
darcyOp->SetBlock(1,0, B);
|
||||
}
|
||||
|
||||
// 12. Construct the operators for preconditioner
|
||||
// 11. Construct the operators for preconditioner
|
||||
//
|
||||
// P = [ diag(M) 0 ]
|
||||
// [ 0 B diag(M)^-1 B^T ]
|
||||
@@ -285,11 +266,10 @@ int main(int argc, char *argv[])
|
||||
{
|
||||
Md_PA.SetSize(R_space->GetTrueVSize());
|
||||
mVarf->AssembleDiagonal(Md_PA);
|
||||
auto Md_host = Md_PA.HostRead();
|
||||
Vector invMd(Md_PA.Size());
|
||||
for (int i=0; i<Md_PA.Size(); ++i)
|
||||
{
|
||||
invMd(i) = 1.0 / Md_host[i];
|
||||
invMd(i) = 1.0 / Md_PA(i);
|
||||
}
|
||||
|
||||
Vector BMBt_diag(W_space->GetTrueVSize());
|
||||
@@ -322,7 +302,7 @@ int main(int argc, char *argv[])
|
||||
darcyPr->SetDiagonalBlock(0, invM);
|
||||
darcyPr->SetDiagonalBlock(1, invS);
|
||||
|
||||
// 13. Solve the linear system with MINRES.
|
||||
// 12. Solve the linear system with MINRES.
|
||||
// Check the norm of the unpreconditioned residual.
|
||||
int maxIter(pa ? 1000 : 500);
|
||||
double rtol(1.e-6);
|
||||
@@ -339,7 +319,6 @@ int main(int argc, char *argv[])
|
||||
solver.SetPrintLevel(verbose);
|
||||
trueX = 0.0;
|
||||
solver.Mult(trueRhs, trueX);
|
||||
if (device.IsEnabled()) { trueX.HostRead(); }
|
||||
chrono.Stop();
|
||||
|
||||
if (verbose)
|
||||
@@ -353,7 +332,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "MINRES solver took " << chrono.RealTime() << "s. \n";
|
||||
}
|
||||
|
||||
// 14. Extract the parallel grid function corresponding to the finite element
|
||||
// 13. Extract the parallel grid function corresponding to the finite element
|
||||
// approximation X. This is the local solution on each processor. Compute
|
||||
// L2 error norms.
|
||||
ParGridFunction *u(new ParGridFunction);
|
||||
@@ -381,7 +360,7 @@ int main(int argc, char *argv[])
|
||||
std::cout << "|| p_h - p_ex || / || p_ex || = " << err_p / norm_p << "\n";
|
||||
}
|
||||
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can be
|
||||
// 14. Save the refined mesh and the solution in parallel. This output can be
|
||||
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol_*".
|
||||
{
|
||||
ostringstream mesh_name, u_name, p_name;
|
||||
@@ -402,7 +381,7 @@ int main(int argc, char *argv[])
|
||||
p->Save(p_ofs);
|
||||
}
|
||||
|
||||
// 16. Save data in the VisIt format
|
||||
// 15. Save data in the VisIt format
|
||||
VisItDataCollection visit_dc("Example5-Parallel", pmesh);
|
||||
visit_dc.RegisterField("velocity", u);
|
||||
visit_dc.RegisterField("pressure", p);
|
||||
@@ -411,7 +390,7 @@ int main(int argc, char *argv[])
|
||||
DataCollection::PARALLEL_FORMAT);
|
||||
visit_dc.Save();
|
||||
|
||||
// 17. Save data in the ParaView format
|
||||
// 16. Save data in the ParaView format
|
||||
ParaViewDataCollection paraview_dc("Example5P", pmesh);
|
||||
paraview_dc.SetPrefixPath("ParaView");
|
||||
paraview_dc.SetLevelsOfDetail(order);
|
||||
@@ -423,7 +402,7 @@ int main(int argc, char *argv[])
|
||||
paraview_dc.RegisterField("pressure",p);
|
||||
paraview_dc.Save();
|
||||
|
||||
// 18. Optionally output a BP (binary pack) file using ADIOS2. This can be
|
||||
// 17. Optionally output a BP (binary pack) file using ADIOS2. This can be
|
||||
// visualized with the ParaView VTX reader.
|
||||
#ifdef MFEM_USE_ADIOS2
|
||||
if (adios2)
|
||||
@@ -443,7 +422,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
#endif
|
||||
|
||||
// 19. Send the solution by socket to a GLVis server.
|
||||
// 18. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
@@ -463,7 +442,7 @@ int main(int argc, char *argv[])
|
||||
<< endl;
|
||||
}
|
||||
|
||||
// 20. Free the used memory.
|
||||
// 19. Free the used memory.
|
||||
delete fform;
|
||||
delete gform;
|
||||
delete u;
|
||||
|
||||
+17
-19
@@ -20,11 +20,8 @@
|
||||
// Device sample runs:
|
||||
// ex9 -pa
|
||||
// ex9 -ea
|
||||
// ex9 -fa
|
||||
// ex9 -pa -m ../data/periodic-cube.mesh
|
||||
// ex9 -pa -m ../data/periodic-cube.mesh -d cuda
|
||||
// ex9 -ea -m ../data/periodic-cube.mesh -d cuda
|
||||
// ex9 -fa -m ../data/periodic-cube.mesh -d cuda
|
||||
//
|
||||
// Description: This example code solves the time-dependent advection equation
|
||||
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
|
||||
@@ -147,7 +144,6 @@ int main(int argc, char *argv[])
|
||||
int order = 3;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
bool fa = false;
|
||||
const char *device_config = "cpu";
|
||||
int ode_solver_type = 4;
|
||||
double t_final = 10.0;
|
||||
@@ -174,8 +170,6 @@ int main(int argc, char *argv[])
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
|
||||
"--no-full-assembly", "Enable Full Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
|
||||
@@ -260,7 +254,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
// 5. Define the discontinuous DG finite element space of the given
|
||||
// polynomial order on the refined mesh.
|
||||
DG_FECollection fec(order, dim, BasisType::GaussLobatto);
|
||||
DG_FECollection fec(order, dim, BasisType::Positive);
|
||||
FiniteElementSpace fes(&mesh, &fec);
|
||||
|
||||
cout << "Number of unknowns: " << fes.GetVSize() << endl;
|
||||
@@ -284,11 +278,6 @@ int main(int argc, char *argv[])
|
||||
m.SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
k.SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
}
|
||||
else if (fa)
|
||||
{
|
||||
m.SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
k.SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
}
|
||||
m.AddDomainIntegrator(new MassIntegrator);
|
||||
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k.AddInteriorFaceIntegrator(
|
||||
@@ -389,6 +378,10 @@ int main(int argc, char *argv[])
|
||||
// iterations, ti, with a time-step dt).
|
||||
FE_Evolution adv(m, k, b);
|
||||
|
||||
Vector masses(u.Size());
|
||||
m.SpMat().Mult(u, masses);
|
||||
double mass = masses.Sum();
|
||||
|
||||
double t = 0.0;
|
||||
adv.SetTime(t);
|
||||
ode_solver->Init(adv);
|
||||
@@ -435,6 +428,9 @@ int main(int argc, char *argv[])
|
||||
u.Save(osol);
|
||||
}
|
||||
|
||||
m.SpMat().Mult(u, masses);
|
||||
cout << "Mass difference:" << abs(mass - masses.Sum()) << endl;
|
||||
|
||||
// 10. Free the used memory.
|
||||
delete ode_solver;
|
||||
delete pd;
|
||||
@@ -448,19 +444,21 @@ int main(int argc, char *argv[])
|
||||
FE_Evolution::FE_Evolution(BilinearForm &_M, BilinearForm &_K, const Vector &_b)
|
||||
: TimeDependentOperator(_M.Height()), M(_M), K(_K), b(_b), z(_M.Height())
|
||||
{
|
||||
bool pa = M.GetAssemblyLevel() == AssemblyLevel::PARTIAL;
|
||||
bool ea = M.GetAssemblyLevel() == AssemblyLevel::ELEMENT;
|
||||
Array<int> ess_tdof_list;
|
||||
if (M.GetAssemblyLevel() == AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
M_prec = new DSmoother(M.SpMat());
|
||||
M_solver.SetOperator(M.SpMat());
|
||||
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
|
||||
}
|
||||
else
|
||||
if (pa || ea)
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(M, ess_tdof_list);
|
||||
M_solver.SetOperator(M);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
else
|
||||
{
|
||||
M_prec = new DSmoother(M.SpMat());
|
||||
dg_solver = new DG_Solver(M.SpMat(), K.SpMat(), *M.FESpace());
|
||||
M_solver.SetOperator(M.SpMat());
|
||||
}
|
||||
M_solver.SetPreconditioner(*M_prec);
|
||||
M_solver.iterative_mode = false;
|
||||
M_solver.SetRelTol(1e-9);
|
||||
|
||||
+15
-24
@@ -21,11 +21,8 @@
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex9p -pa
|
||||
// mpirun -np 4 ex9p -ea
|
||||
// mpirun -np 4 ex9p -fa
|
||||
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh
|
||||
// mpirun -np 4 ex9p -pa -m ../data/periodic-cube.mesh -d cuda
|
||||
// mpirun -np 4 ex9p -ea -m ../data/periodic-cube.mesh -d cuda
|
||||
// mpirun -np 4 ex9p -fa -m ../data/periodic-cube.mesh -d cuda
|
||||
//
|
||||
// Description: This example code solves the time-dependent advection equation
|
||||
// du/dt + v.grad(u) = 0, where v is a given fluid velocity, and
|
||||
@@ -167,7 +164,6 @@ int main(int argc, char *argv[])
|
||||
int order = 3;
|
||||
bool pa = false;
|
||||
bool ea = false;
|
||||
bool fa = false;
|
||||
const char *device_config = "cpu";
|
||||
int ode_solver_type = 4;
|
||||
double t_final = 10.0;
|
||||
@@ -197,8 +193,6 @@ int main(int argc, char *argv[])
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&ea, "-ea", "--element-assembly", "-no-ea",
|
||||
"--no-element-assembly", "Enable Element Assembly.");
|
||||
args.AddOption(&fa, "-fa", "--full-assembly", "-no-fa",
|
||||
"--no-full-assembly", "Enable Full Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&ode_solver_type, "-s", "--ode-solver",
|
||||
@@ -335,12 +329,6 @@ int main(int argc, char *argv[])
|
||||
m->SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
k->SetAssemblyLevel(AssemblyLevel::ELEMENT);
|
||||
}
|
||||
else if (fa)
|
||||
{
|
||||
m->SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
k->SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
}
|
||||
|
||||
m->AddDomainIntegrator(new MassIntegrator);
|
||||
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k->AddInteriorFaceIntegrator(
|
||||
@@ -577,21 +565,29 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
|
||||
M_solver(_M.ParFESpace()->GetComm()),
|
||||
z(_M.Height())
|
||||
{
|
||||
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
|
||||
{
|
||||
M.Reset(_M.ParallelAssemble(), true);
|
||||
K.Reset(_K.ParallelAssemble(), true);
|
||||
}
|
||||
else
|
||||
bool pa = _M.GetAssemblyLevel()==AssemblyLevel::PARTIAL;
|
||||
bool ea = _M.GetAssemblyLevel()==AssemblyLevel::ELEMENT;
|
||||
|
||||
if (pa || ea)
|
||||
{
|
||||
M.Reset(&_M, false);
|
||||
K.Reset(&_K, false);
|
||||
}
|
||||
else
|
||||
{
|
||||
M.Reset(_M.ParallelAssemble(), true);
|
||||
K.Reset(_K.ParallelAssemble(), true);
|
||||
}
|
||||
|
||||
M_solver.SetOperator(*M);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
if (_M.GetAssemblyLevel()==AssemblyLevel::LEGACYFULL)
|
||||
if (pa || ea)
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
else
|
||||
{
|
||||
HypreParMatrix &M_mat = *M.As<HypreParMatrix>();
|
||||
HypreParMatrix &K_mat = *K.As<HypreParMatrix>();
|
||||
@@ -600,11 +596,6 @@ FE_Evolution::FE_Evolution(ParBilinearForm &_M, ParBilinearForm &_K,
|
||||
|
||||
dg_solver = new DG_Solver(M_mat, K_mat, *_M.FESpace());
|
||||
}
|
||||
else
|
||||
{
|
||||
M_prec = new OperatorJacobiSmoother(_M, ess_tdof_list);
|
||||
dg_solver = NULL;
|
||||
}
|
||||
|
||||
M_solver.SetPreconditioner(*M_prec);
|
||||
M_solver.iterative_mode = false;
|
||||
|
||||
@@ -50,43 +50,10 @@ set(SRCS
|
||||
fespacehierarchy.cpp
|
||||
nonlininteg_vectorconvection.cpp
|
||||
quadinterpolator.cpp
|
||||
quadinterpolator_det.cpp
|
||||
quadinterpolator_eval_by_nodes.cpp
|
||||
quadinterpolator_eval_by_vdim.cpp
|
||||
quadinterpolator_grad_by_nodes.cpp
|
||||
quadinterpolator_grad_by_vdim.cpp
|
||||
quadinterpolator_grad_phys_by_nodes.cpp
|
||||
quadinterpolator_grad_phys_by_vdim.cpp
|
||||
quadinterpolator_face.cpp
|
||||
restriction.cpp
|
||||
staticcond.cpp
|
||||
tmop.cpp
|
||||
tmop_pa.cpp
|
||||
tmop_pa_h2d.cpp
|
||||
tmop_pa_h2d_c0.cpp
|
||||
tmop_pa_h2m.cpp
|
||||
tmop_pa_h2m_c0.cpp
|
||||
tmop_pa_h2s.cpp
|
||||
tmop_pa_h2s_c0.cpp
|
||||
tmop_pa_h3d.cpp
|
||||
tmop_pa_h3d_c0.cpp
|
||||
tmop_pa_h3m.cpp
|
||||
tmop_pa_h3m_c0.cpp
|
||||
tmop_pa_h3s.cpp
|
||||
tmop_pa_h3s_c0.cpp
|
||||
tmop_pa_jp2.cpp
|
||||
tmop_pa_jp3.cpp
|
||||
tmop_pa_jt2_tc.cpp
|
||||
tmop_pa_jt3_datc.cpp
|
||||
tmop_pa_jt3_tc.cpp
|
||||
tmop_pa_p2.cpp
|
||||
tmop_pa_p2_c0.cpp
|
||||
tmop_pa_p3.cpp
|
||||
tmop_pa_p3_c0.cpp
|
||||
tmop_pa_w2.cpp
|
||||
tmop_pa_w2_c0.cpp
|
||||
tmop_pa_w3.cpp
|
||||
tmop_pa_w3_c0.cpp
|
||||
tmop_tools.cpp
|
||||
gslib.cpp
|
||||
transfer.cpp
|
||||
@@ -116,10 +83,7 @@ set(HDRS
|
||||
nonlinearform_ext.hpp
|
||||
nonlininteg.hpp
|
||||
quadinterpolator.hpp
|
||||
quadinterpolator_eval.hpp
|
||||
quadinterpolator_face.hpp
|
||||
quadinterpolator_grad.hpp
|
||||
quadinterpolator_grad_phys.hpp
|
||||
restriction.hpp
|
||||
fespacehierarchy.hpp
|
||||
staticcond.hpp
|
||||
@@ -132,7 +96,6 @@ set(HDRS
|
||||
tfespace.hpp
|
||||
tintrules.hpp
|
||||
tmop.hpp
|
||||
tmop_pa.hpp
|
||||
tmop_tools.hpp
|
||||
gslib.hpp
|
||||
transfer.hpp
|
||||
|
||||
+14
-16
@@ -76,7 +76,7 @@ BilinearForm::BilinearForm(FiniteElementSpace * f)
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
@@ -94,7 +94,7 @@ BilinearForm::BilinearForm (FiniteElementSpace * f, BilinearForm * bf, int ps)
|
||||
precompute_sparsity = ps;
|
||||
diag_policy = DIAG_KEEP;
|
||||
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
|
||||
@@ -121,10 +121,9 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
assembly = assembly_level;
|
||||
switch (assembly)
|
||||
{
|
||||
case AssemblyLevel::LEGACYFULL:
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
ext = new FABilinearFormExtension(this);
|
||||
// ext = new FABilinearFormExtension(this);
|
||||
// Use the original BilinearForm implementation for now
|
||||
break;
|
||||
case AssemblyLevel::ELEMENT:
|
||||
ext = new EABilinearFormExtension(this);
|
||||
@@ -144,7 +143,7 @@ void BilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
void BilinearForm::EnableStaticCondensation()
|
||||
{
|
||||
delete static_cond;
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
{
|
||||
static_cond = NULL;
|
||||
MFEM_WARNING("Static condensation not supported for this assembly level");
|
||||
@@ -169,7 +168,7 @@ void BilinearForm::EnableHybridization(FiniteElementSpace *constr_space,
|
||||
const Array<int> &ess_tdof_list)
|
||||
{
|
||||
delete hybridization;
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
{
|
||||
delete constr_integ;
|
||||
hybridization = NULL;
|
||||
@@ -224,7 +223,7 @@ MatrixInverse * BilinearForm::Inverse() const
|
||||
|
||||
void BilinearForm::Finalize (int skip_zeros)
|
||||
{
|
||||
if (assembly == AssemblyLevel::LEGACYFULL)
|
||||
if (assembly == AssemblyLevel::FULL)
|
||||
{
|
||||
if (!static_cond) { mat->Finalize(skip_zeros); }
|
||||
if (mat_e) { mat_e->Finalize(skip_zeros); }
|
||||
@@ -640,7 +639,8 @@ void BilinearForm::AssembleDiagonal(Vector &diag) const
|
||||
}
|
||||
else
|
||||
{
|
||||
mat->GetDiag(diag);
|
||||
MFEM_ABORT("Not implemented. Maybe assemble your bilinear form into a "
|
||||
"matrix and use SparseMatrix::GetDiag?");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1083,7 +1083,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
|
||||
mat = NULL;
|
||||
mat_e = NULL;
|
||||
extern_bfs = 0;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
@@ -1108,7 +1108,7 @@ MixedBilinearForm::MixedBilinearForm (FiniteElementSpace *tr_fes,
|
||||
bbfi_marker = mbf->bbfi_marker;
|
||||
btfbfi_marker = mbf->btfbfi_marker;
|
||||
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
@@ -1121,8 +1121,6 @@ void MixedBilinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
assembly = assembly_level;
|
||||
switch (assembly)
|
||||
{
|
||||
case AssemblyLevel::LEGACYFULL:
|
||||
break;
|
||||
case AssemblyLevel::FULL:
|
||||
// ext = new FAMixedBilinearFormExtension(this);
|
||||
// Use the original BilinearForm implementation for now
|
||||
@@ -1193,7 +1191,7 @@ void MixedBilinearForm::AddMultTranspose(const Vector & x, Vector & y,
|
||||
|
||||
MatrixInverse * MixedBilinearForm::Inverse() const
|
||||
{
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
{
|
||||
MFEM_WARNING("MixedBilinearForm::Inverse not possible with this assembly level!");
|
||||
return NULL;
|
||||
@@ -1206,7 +1204,7 @@ MatrixInverse * MixedBilinearForm::Inverse() const
|
||||
|
||||
void MixedBilinearForm::Finalize (int skip_zeros)
|
||||
{
|
||||
if (assembly == AssemblyLevel::LEGACYFULL)
|
||||
if (assembly == AssemblyLevel::FULL)
|
||||
{
|
||||
mat -> Finalize (skip_zeros);
|
||||
}
|
||||
@@ -1483,7 +1481,7 @@ void MixedBilinearForm::AssembleDiagonal_ADAt(const Vector &D,
|
||||
|
||||
void MixedBilinearForm::ConformingAssemble()
|
||||
{
|
||||
if (assembly != AssemblyLevel::LEGACYFULL)
|
||||
if (assembly != AssemblyLevel::FULL)
|
||||
{
|
||||
MFEM_WARNING("Conforming assemble not supported for this assembly level!");
|
||||
return;
|
||||
|
||||
@@ -29,11 +29,8 @@ namespace mfem
|
||||
form classes derived from Operator. */
|
||||
enum class AssemblyLevel
|
||||
{
|
||||
/// Legacy fully assembled form, i.e. a global sparse matrix in MFEM, Hypre
|
||||
/// or PETSC format. This assembly is ALWAYS performed on the host.
|
||||
LEGACYFULL = 0,
|
||||
/// Fully assembled form, i.e. a global sparse matrix in MFEM format. This
|
||||
/// assembly is compatible with device execution.
|
||||
/// Fully assembled form, i.e. a global sparse matrix in MFEM, Hypre or PETSC
|
||||
/// format.
|
||||
FULL,
|
||||
/// Form assembled at element level, which computes and stores dense element
|
||||
/// matrices.
|
||||
@@ -122,7 +119,7 @@ protected:
|
||||
static_cond = NULL; hybridization = NULL;
|
||||
precompute_sparsity = 0;
|
||||
diag_policy = DIAG_KEEP;
|
||||
assembly = AssemblyLevel::LEGACYFULL;
|
||||
assembly = AssemblyLevel::FULL;
|
||||
batch = 1;
|
||||
ext = NULL;
|
||||
}
|
||||
|
||||
+36
-188
@@ -15,7 +15,6 @@
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "libceed/ceed.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -116,7 +115,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
|
||||
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
|
||||
|
||||
const int iSz = integrators.Size();
|
||||
if (elem_restrict && !DeviceCanUseCeed())
|
||||
if (elem_restrict)
|
||||
{
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
@@ -293,8 +292,7 @@ void PABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
|
||||
// Data and methods for element-assembled bilinear forms
|
||||
EABilinearFormExtension::EABilinearFormExtension(BilinearForm *form)
|
||||
: PABilinearFormExtension(form),
|
||||
factorize_face_terms(form->FESpace()->IsDGSpace())
|
||||
: PABilinearFormExtension(form)
|
||||
{
|
||||
}
|
||||
|
||||
@@ -349,17 +347,6 @@ void EABilinearFormExtension::Assemble()
|
||||
{
|
||||
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr);
|
||||
}
|
||||
|
||||
if (factorize_face_terms && int_face_restrict_lex)
|
||||
{
|
||||
auto restFint = dynamic_cast<const L2FaceRestriction&>(*int_face_restrict_lex);
|
||||
restFint.AddFaceMatricesToElementMatrices(ea_data_int, ea_data);
|
||||
}
|
||||
if (factorize_face_terms && bdr_face_restrict_lex)
|
||||
{
|
||||
auto restFbdr = dynamic_cast<const L2FaceRestriction&>(*bdr_face_restrict_lex);
|
||||
restFbdr.AddFaceMatricesToElementMatrices(ea_data_bdr, ea_data);
|
||||
}
|
||||
}
|
||||
|
||||
void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
@@ -412,27 +399,24 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
|
||||
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
|
||||
if (!factorize_face_terms)
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
}
|
||||
res += A_int(i, j, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(i, j, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
@@ -459,7 +443,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
// Treatment of boundary faces
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
if (bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, faceBdrX);
|
||||
@@ -538,27 +522,24 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
const int NDOFS = faceDofs;
|
||||
auto X = Reshape(faceIntX.Read(), NDOFS, 2, nf_int);
|
||||
auto Y = Reshape(faceIntY.ReadWrite(), NDOFS, 2, nf_int);
|
||||
if (!factorize_face_terms)
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
double res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
}
|
||||
res += A_int(j, i, 0, f)*X(i, 0, f);
|
||||
}
|
||||
Y(j, 0, f) += res;
|
||||
res = 0.0;
|
||||
for (int i = 0; i < NDOFS; i++)
|
||||
{
|
||||
res += A_int(j, i, 1, f)*X(i, 1, f);
|
||||
}
|
||||
Y(j, 1, f) += res;
|
||||
});
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
@@ -585,7 +566,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
// Treatment of boundary faces
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (!factorize_face_terms && bdr_face_restrict_lex && bFISz>0)
|
||||
if (bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
// Apply the Boundary Face Restriction
|
||||
bdr_face_restrict_lex->Mult(x, faceBdrX);
|
||||
@@ -614,139 +595,6 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
// Data and methods for fully-assembled bilinear forms
|
||||
FABilinearFormExtension::FABilinearFormExtension(BilinearForm *form)
|
||||
: EABilinearFormExtension(form),
|
||||
mat(form->FESpace()->GetVSize(),form->FESpace()->GetVSize(),0),
|
||||
face_mat(form->FESpace()->GetVSize(),0,0),
|
||||
use_face_mat(false)
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if ( ParFiniteElementSpace* pfes =
|
||||
dynamic_cast<ParFiniteElementSpace*>(form->FESpace()) )
|
||||
{
|
||||
if (pfes->IsDGSpace())
|
||||
{
|
||||
use_face_mat = true;
|
||||
pfes->ExchangeFaceNbrData();
|
||||
face_mat.SetWidth(pfes->GetFaceNbrVSize());
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::Assemble()
|
||||
{
|
||||
EABilinearFormExtension::Assemble();
|
||||
FiniteElementSpace &fes = *a->FESpace();
|
||||
if (fes.IsDGSpace())
|
||||
{
|
||||
const L2ElementRestriction *restE =
|
||||
static_cast<const L2ElementRestriction*>(elem_restrict);
|
||||
const L2FaceRestriction *restF =
|
||||
static_cast<const L2FaceRestriction*>(int_face_restrict_lex);
|
||||
// 1. Fill I
|
||||
// 1.1 Increment with restE
|
||||
restE->FillI(mat);
|
||||
// 1.2 Increment with restF
|
||||
if (restF) { restF->FillI(mat, face_mat); }
|
||||
// 1.3 Sum the non-zeros in I
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
int cpt = 0;
|
||||
const int vd = fes.GetVDim();
|
||||
const int ndofs = ne*elemDofs*vd;
|
||||
for (int i = 0; i < ndofs; i++)
|
||||
{
|
||||
const int nnz = h_I[i];
|
||||
h_I[i] = cpt;
|
||||
cpt += nnz;
|
||||
}
|
||||
const int nnz = cpt;
|
||||
h_I[ndofs] = nnz;
|
||||
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
|
||||
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
|
||||
if (use_face_mat && restF)
|
||||
{
|
||||
auto h_I_face = face_mat.HostReadWriteI();
|
||||
int cpt = 0;
|
||||
for (int i = 0; i < ndofs; i++)
|
||||
{
|
||||
const int nnz = h_I_face[i];
|
||||
h_I_face[i] = cpt;
|
||||
cpt += nnz;
|
||||
}
|
||||
const int nnz_face = cpt;
|
||||
h_I_face[ndofs] = nnz_face;
|
||||
face_mat.GetMemoryJ().New(nnz_face,
|
||||
face_mat.GetMemoryJ().GetMemoryType());
|
||||
face_mat.GetMemoryData().New(nnz_face,
|
||||
face_mat.GetMemoryData().GetMemoryType());
|
||||
}
|
||||
// 2. Fill J and Data
|
||||
// 2.1 Fill J and Data with Elem ea_data
|
||||
restE->FillJAndData(ea_data, mat);
|
||||
// 2.2 Fill J and Data with Face ea_data_ext
|
||||
if (restF) { restF->FillJAndData(ea_data_ext, mat, face_mat); }
|
||||
// 2.3 Shift indirections in I back to original
|
||||
auto I = mat.HostReadWriteI();
|
||||
for (int i = ndofs; i > 0; i--)
|
||||
{
|
||||
I[i] = I[i-1];
|
||||
}
|
||||
I[0] = 0;
|
||||
if (use_face_mat && restF)
|
||||
{
|
||||
auto I_face = face_mat.HostReadWriteI();
|
||||
for (int i = ndofs; i > 0; i--)
|
||||
{
|
||||
I_face[i] = I_face[i-1];
|
||||
}
|
||||
I_face[0] = 0;
|
||||
}
|
||||
}
|
||||
else // continuous Galerkin case
|
||||
{
|
||||
const ElementRestriction &rest =
|
||||
static_cast<const ElementRestriction&>(*elem_restrict);
|
||||
rest.FillSparseMatrix(ea_data, mat);
|
||||
}
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
mat.Mult(x, y);
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (const ParFiniteElementSpace *pfes =
|
||||
dynamic_cast<const ParFiniteElementSpace*>(testFes))
|
||||
{
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
|
||||
const_cast<Vector&>(x),0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
Vector &shared_x = x_gf.FaceNbrData();
|
||||
if (shared_x.Size()) { face_mat.AddMult(shared_x, y); }
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void FABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
mat.MultTranspose(x, y);
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (const ParFiniteElementSpace *pfes =
|
||||
dynamic_cast<const ParFiniteElementSpace*>(testFes))
|
||||
{
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(pfes),
|
||||
const_cast<Vector&>(x),0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
Vector &shared_x = x_gf.FaceNbrData();
|
||||
if (shared_x.Size()) { face_mat.AddMultTranspose(shared_x, y); }
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
MixedBilinearFormExtension::MixedBilinearFormExtension(MixedBilinearForm *form)
|
||||
: Operator(form->Height(), form->Width()), a(form)
|
||||
{
|
||||
|
||||
+21
-19
@@ -62,6 +62,27 @@ public:
|
||||
virtual void Update() = 0;
|
||||
};
|
||||
|
||||
/** @brief Data and methods for fully-assembled bilinear forms.
|
||||
Not yet implemented! Use the BilinearForm Class instead. */
|
||||
class FABilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
public:
|
||||
FABilinearFormExtension(BilinearForm *form)
|
||||
: BilinearFormExtension(form) { }
|
||||
|
||||
/// TODO
|
||||
void Assemble() {}
|
||||
void FormSystemMatrix(const Array<int> &ess_tdof_list, OperatorHandle &A) {}
|
||||
void FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
Vector &x, Vector &b,
|
||||
OperatorHandle &A, Vector &X, Vector &B,
|
||||
int copy_interior = 0) {}
|
||||
void Mult(const Vector &x, Vector &y) const {}
|
||||
void MultTranspose(const Vector &x, Vector &y) const {}
|
||||
void Update() {}
|
||||
~FABilinearFormExtension() {}
|
||||
};
|
||||
|
||||
/// Data and methods for partially-assembled bilinear forms
|
||||
class PABilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
@@ -98,12 +119,10 @@ class EABilinearFormExtension : public PABilinearFormExtension
|
||||
protected:
|
||||
int ne;
|
||||
int elemDofs;
|
||||
// The element matrices are stored row major
|
||||
Vector ea_data;
|
||||
int nf_int, nf_bdr;
|
||||
int faceDofs;
|
||||
Vector ea_data_int, ea_data_ext, ea_data_bdr;
|
||||
bool factorize_face_terms;
|
||||
|
||||
public:
|
||||
EABilinearFormExtension(BilinearForm *form);
|
||||
@@ -113,23 +132,6 @@ public:
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
/// Data and methods for fully-assembled bilinear forms
|
||||
class FABilinearFormExtension : public EABilinearFormExtension
|
||||
{
|
||||
private:
|
||||
SparseMatrix mat;
|
||||
/// face_mat handles parallelism for DG face terms.
|
||||
SparseMatrix face_mat;
|
||||
bool use_face_mat;
|
||||
|
||||
public:
|
||||
FABilinearFormExtension(BilinearForm *form);
|
||||
|
||||
void Assemble();
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
/// Data and methods for matrix-free bilinear forms NOT YET IMPLEMENTED.
|
||||
class MFBilinearFormExtension : public BilinearFormExtension
|
||||
{
|
||||
|
||||
+2
-32
@@ -1685,22 +1685,6 @@ protected:
|
||||
{
|
||||
trial_fe.CalcPhysCurlShape(Trans, shape);
|
||||
}
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes);
|
||||
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
private:
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
|
||||
const DofToQuad *mapsOtest; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsCtest; ///< Not owned. DOF-to-quad map, closed.
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, dofs1D, dofs1Dtest,quad1D, testType, trialType, coeffDim;
|
||||
};
|
||||
|
||||
/** Class for integrating the bilinear form a(u,v) := (Q u, curl v) in 3D and
|
||||
@@ -1740,20 +1724,6 @@ protected:
|
||||
{
|
||||
test_fe.CalcPhysCurlShape(Trans, shape);
|
||||
}
|
||||
|
||||
using BilinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes);
|
||||
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
private:
|
||||
// PA extension
|
||||
Vector pa_data;
|
||||
const DofToQuad *mapsO; ///< Not owned. DOF-to-quad map, open.
|
||||
const DofToQuad *mapsC; ///< Not owned. DOF-to-quad map, closed.
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, dofs1D, quad1D, testType, trialType, coeffDim;
|
||||
};
|
||||
|
||||
/** Class for integrating the bilinear form a(u,v) := - (Q u, grad v) in either
|
||||
@@ -1954,7 +1924,7 @@ public:
|
||||
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe);
|
||||
|
||||
void SetupPA(const FiniteElementSpace &fes);
|
||||
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
|
||||
};
|
||||
|
||||
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
|
||||
@@ -2030,7 +2000,7 @@ public:
|
||||
const FiniteElement &test_fe,
|
||||
ElementTransformation &Trans);
|
||||
|
||||
void SetupPA(const FiniteElementSpace &fes);
|
||||
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
|
||||
};
|
||||
|
||||
/** Mass integrator (u, v) restricted to the boundary of a domain */
|
||||
|
||||
@@ -32,7 +32,7 @@ static void EAConvectionAssemble1D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -54,7 +54,7 @@ static void EAConvectionAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Bj[k1] * D(k1, e) * r_Gi[k1];
|
||||
}
|
||||
A(i1, j1, e) += val;
|
||||
A(i1, j1, e) = val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -76,7 +76,7 @@ static void EAConvectionAssemble2D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -121,7 +121,7 @@ static void EAConvectionAssemble2D(const int NE,
|
||||
* r_B[k1][j1]* r_B[k2][j2];
|
||||
}
|
||||
}
|
||||
A(i1, i2, j1, j2, e) += val;
|
||||
A(i1, i2, j1, j2, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -145,7 +145,7 @@ static void EAConvectionAssemble3D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -191,7 +191,7 @@ static void EAConvectionAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
A(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
A(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,10 +13,6 @@
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
#include "restriction.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
@@ -72,53 +68,47 @@ static void PAConvectionSetup3D(const int Q1D,
|
||||
const double alpha,
|
||||
Vector &op)
|
||||
{
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const int NQ = Q1D*Q1D*Q1D;
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
const bool const_v = vel.Size() == 3;
|
||||
const auto V = const_v ?
|
||||
Reshape(vel.Read(), 3,1,1,1,1) :
|
||||
Reshape(vel.Read(), 3,Q1D,Q1D,Q1D,NE);
|
||||
auto y = Reshape(op.Write(), Q1D,Q1D,Q1D,3,NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
auto V =
|
||||
const_v ? Reshape(vel.Read(), 3,1,1) : Reshape(vel.Read(), 3,NQ,NE);
|
||||
auto y = Reshape(op.Write(), NQ, 3, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double w = alpha * W(qx,qy,qz);
|
||||
const double v0 = const_v ? V(0,0,0,0,0) : V(0,qx,qy,qz,e);
|
||||
const double v1 = const_v ? V(1,0,0,0,0) : V(1,qx,qy,qz,e);
|
||||
const double v2 = const_v ? V(2,0,0,0,0) : V(2,qx,qy,qz,e);
|
||||
const double wx = w * v0;
|
||||
const double wy = w * v1;
|
||||
const double wz = w * v2;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// q . J^{-1} = q . adj(J)
|
||||
y(qx,qy,qz,0,e) = wx * A11 + wy * A12 + wz * A13;
|
||||
y(qx,qy,qz,1,e) = wx * A21 + wy * A22 + wz * A23;
|
||||
y(qx,qy,qz,2,e) = wx * A31 + wy * A32 + wz * A33;
|
||||
}
|
||||
}
|
||||
const double J11 = J(q,0,0,e);
|
||||
const double J21 = J(q,1,0,e);
|
||||
const double J31 = J(q,2,0,e);
|
||||
const double J12 = J(q,0,1,e);
|
||||
const double J22 = J(q,1,1,e);
|
||||
const double J32 = J(q,2,1,e);
|
||||
const double J13 = J(q,0,2,e);
|
||||
const double J23 = J(q,1,2,e);
|
||||
const double J33 = J(q,2,2,e);
|
||||
const double w = alpha * W[q];
|
||||
const double v0 = const_v ? V(0,0,0) : V(0,q,e);
|
||||
const double v1 = const_v ? V(1,0,0) : V(1,q,e);
|
||||
const double v2 = const_v ? V(2,0,0) : V(2,q,e);
|
||||
const double wx = w * v0;
|
||||
const double wy = w * v1;
|
||||
const double wz = w * v2;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// q . J^{-1} = q . adj(J)
|
||||
y(q,0,e) = wx * A11 + wy * A12 + wz * A13;
|
||||
y(q,1,e) = wx * A21 + wy * A22 + wz * A23;
|
||||
y(q,2,e) = wx * A31 + wy * A32 + wz * A33;
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -194,8 +184,8 @@ void PAConvectionApply2D(const int ne,
|
||||
Gu[dy][qx] = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dy][dx];
|
||||
Bu[dy][qx] += bx * x;
|
||||
Gu[dy][qx] += gx * x;
|
||||
@@ -212,8 +202,8 @@ void PAConvectionApply2D(const int ne,
|
||||
BGu[qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
GBu[qy][qx] += gx * Bu[dy][qx];
|
||||
BGu[qy][qx] += bx * Gu[dy][qx];
|
||||
}
|
||||
@@ -242,7 +232,7 @@ void PAConvectionApply2D(const int ne,
|
||||
BDGu[dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BDGu[dy][qx] += w * DGu[qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -254,7 +244,7 @@ void PAConvectionApply2D(const int ne,
|
||||
double BBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBDGu += w * BDGu[dy][qx];
|
||||
}
|
||||
y(dx,dy,e) += BBDGu;
|
||||
@@ -320,7 +310,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[tidz][dy][dx];
|
||||
const double x = u[tidz][dy][dx];
|
||||
Bu[tidz][dy][qx] += bx * x;
|
||||
Gu[tidz][dy][qx] += gx * x;
|
||||
}
|
||||
@@ -337,8 +327,8 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
BGu[tidz][qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
GBu[tidz][qy][qx] += gx * Bu[tidz][dy][qx];
|
||||
BGu[tidz][qy][qx] += bx * Gu[tidz][dy][qx];
|
||||
}
|
||||
@@ -369,7 +359,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
BDGu[tidz][dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BDGu[tidz][dy][qx] += w * DGu[tidz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -382,7 +372,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
double BBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBDGu += w * BDGu[tidz][dy][qx];
|
||||
}
|
||||
y(dx,dy,e) += BBDGu;
|
||||
@@ -446,8 +436,8 @@ void PAConvectionApply3D(const int ne,
|
||||
Gu[dz][dy][qx] = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dz][dy][dx];
|
||||
Bu[dz][dy][qx] += bx * x;
|
||||
Gu[dz][dy][qx] += gx * x;
|
||||
@@ -469,8 +459,8 @@ void PAConvectionApply3D(const int ne,
|
||||
BGu[dz][qy][qx] = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
BBu[dz][qy][qx] += bx * Bu[dz][dy][qx];
|
||||
GBu[dz][qy][qx] += gx * Bu[dz][dy][qx];
|
||||
BGu[dz][qy][qx] += bx * Gu[dz][dy][qx];
|
||||
@@ -492,8 +482,8 @@ void PAConvectionApply3D(const int ne,
|
||||
BBGu[qz][qy][qx] = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
GBBu[qz][qy][qx] += gx * BBu[dz][qy][qx];
|
||||
BGBu[qz][qy][qx] += bx * GBu[dz][qy][qx];
|
||||
BBGu[qz][qy][qx] += bx * BGu[dz][qy][qx];
|
||||
@@ -531,7 +521,7 @@ void PAConvectionApply3D(const int ne,
|
||||
BDGu[dz][qy][qx] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double w = Bt(dz,qz);
|
||||
const double w = Bt(dz,qz);
|
||||
BDGu[dz][qy][qx] += w * DGu[qz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -547,7 +537,7 @@ void PAConvectionApply3D(const int ne,
|
||||
BBDGu[dz][dy][qx] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BBDGu[dz][dy][qx] += w * BDGu[dz][qy][qx];
|
||||
}
|
||||
}
|
||||
@@ -562,7 +552,7 @@ void PAConvectionApply3D(const int ne,
|
||||
double BBBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBBDGu += w * BBDGu[dz][dy][qx];
|
||||
}
|
||||
y(dx,dy,dz,e) += BBBDGu;
|
||||
@@ -635,8 +625,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double Gu_ = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double bx = B(qx,dx);
|
||||
const double gx = G(qx,dx);
|
||||
const double x = u[dz][dy][dx];
|
||||
Bu_ += bx * x;
|
||||
Gu_ += gx * x;
|
||||
@@ -661,8 +651,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BGu_ = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
const double bx = B(qy,dy);
|
||||
const double gx = G(qy,dy);
|
||||
BBu_ += bx * Bu[dz][dy][qx];
|
||||
GBu_ += gx * Bu[dz][dy][qx];
|
||||
BGu_ += bx * Gu[dz][dy][qx];
|
||||
@@ -688,8 +678,8 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBGu_ = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
const double bx = B(qz,dz);
|
||||
const double gx = G(qz,dz);
|
||||
GBBu_ += gx * BBu[dz][qy][qx];
|
||||
BGBu_ += bx * GBu[dz][qy][qx];
|
||||
BBGu_ += bx * BGu[dz][qy][qx];
|
||||
@@ -731,7 +721,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BDGu_ = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double w = Bt(dz,qz);
|
||||
const double w = Bt(dz,qz);
|
||||
BDGu_ += w * DGu[qz][qy][qx];
|
||||
}
|
||||
BDGu[dz][qy][qx] = BDGu_;
|
||||
@@ -749,7 +739,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBDGu_ = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double w = Bt(dy,qy);
|
||||
const double w = Bt(dy,qy);
|
||||
BBDGu_ += w * BDGu[dz][qy][qx];
|
||||
}
|
||||
BBDGu[dz][dy][qx] = BBDGu_;
|
||||
@@ -766,7 +756,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
double BBBDGu = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double w = Bt(dx,qx);
|
||||
const double w = Bt(dx,qx);
|
||||
BBBDGu += w * BBDGu[dz][dy][qx];
|
||||
}
|
||||
y(dx,dy,dz,e) = BBBDGu;
|
||||
@@ -776,117 +766,6 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
|
||||
static void QEvalVGF2D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto X = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto C = Reshape(y_, VDIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
MFEM_SHARED double DD[NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
mfem::kernels::LoadX<MD1,NBZ>(e,D1D,c,X,DD);
|
||||
mfem::kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,DD,DQ);
|
||||
mfem::kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
double G;
|
||||
mfem::kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,G);
|
||||
C(c,qx,qy,e) = G;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0>
|
||||
static void QEvalVGF3D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto X = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto C = Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
mfem::kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
MFEM_SHARED double DDD[MD1*MD1*MD1];
|
||||
MFEM_SHARED double DDQ[MD1*MD1*MQ1];
|
||||
MFEM_SHARED double DQQ[MD1*MQ1*MQ1];
|
||||
MFEM_SHARED double QQQ[MQ1*MQ1*MQ1];
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
mfem::kernels::LoadX<MD1>(e,D1D,c,X,DDD);
|
||||
mfem::kernels::EvalX<MD1,MQ1>(D1D,Q1D,B,DDD,DDQ);
|
||||
mfem::kernels::EvalY<MD1,MQ1>(D1D,Q1D,B,DDQ,DQQ);
|
||||
mfem::kernels::EvalZ<MD1,MQ1>(D1D,Q1D,B,DQQ,QQQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
double G;
|
||||
mfem::kernels::PullEval<MQ1>(qx,qy,qz,QQQ,G);
|
||||
C(c,qx,qy,qz,e) = G;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assumes tensor-product elements
|
||||
@@ -899,104 +778,16 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
const int nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode, temp_type);
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, temp_type);
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
|
||||
Vector vel;
|
||||
if (VectorConstantCoefficient *cQ =
|
||||
dynamic_cast<VectorConstantCoefficient*>(Q))
|
||||
if (VectorConstantCoefficient *cQ = dynamic_cast<VectorConstantCoefficient*>(Q))
|
||||
{
|
||||
vel = cQ->GetVec();
|
||||
}
|
||||
else if (VectorGridFunctionCoefficient *vgfQ =
|
||||
dynamic_cast<VectorGridFunctionCoefficient*>(Q))
|
||||
{
|
||||
Vector xe;
|
||||
vel.SetSize(dim * nq * ne, temp_type);
|
||||
|
||||
const GridFunction *gf = vgfQ->GetGridFunction();
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
const FiniteElementSpace &gf_fes = *gf->FESpace();
|
||||
|
||||
const int vdim = gf_fes.GetVDim();
|
||||
const Operator *R = gf_fes.GetElementRestriction(ordering);
|
||||
const FiniteElement &el_gf = *gf_fes.GetFE(0);
|
||||
const DofToQuad *maps_gf = &el_gf.GetDofToQuad(*ir, mode);
|
||||
const int D1D = maps_gf->ndof;
|
||||
const int Q1D = maps_gf->nqpt;
|
||||
|
||||
MFEM_VERIFY(R,"");
|
||||
MFEM_VERIFY(vdim == dim, "");
|
||||
MFEM_VERIFY(dim==2 || dim==3,"");
|
||||
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
xe.UseDevice(true);
|
||||
R->Mult(*gf, xe);
|
||||
|
||||
const auto B = maps_gf->B.Read();
|
||||
const auto x = xe.Read();
|
||||
auto y = vel.Write();
|
||||
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: QEvalVGF2D<2,2,2>(ne,B,x,y); break;
|
||||
case 0x33: QEvalVGF2D<2,3,3>(ne,B,x,y); break;
|
||||
case 0x34: QEvalVGF2D<2,3,4>(ne,B,x,y); break;
|
||||
default:
|
||||
{
|
||||
constexpr int MAX_DQ = 8;
|
||||
MFEM_VERIFY(D1D <= MAX_DQ, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_DQ, "");
|
||||
QEvalVGF2D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x23: QEvalVGF3D<3,2,3>(ne,B,x,y); break;
|
||||
case 0x34: QEvalVGF3D<3,3,4>(ne,B,x,y); break;
|
||||
case 0x35: QEvalVGF3D<3,3,5>(ne,B,x,y); break;
|
||||
case 0x46: QEvalVGF3D<3,4,6>(ne,B,x,y); break;
|
||||
case 0x48: QEvalVGF3D<3,4,8>(ne,B,x,y); break;
|
||||
default:
|
||||
{
|
||||
constexpr int MAX_DQ = 6;
|
||||
MFEM_VERIFY(D1D <= MAX_DQ, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_DQ, "");
|
||||
QEvalVGF3D<0,0,0,MAX_DQ>(ne,B,x,y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (VectorQuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<VectorQuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == dim * nq * ne,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
|
||||
qFun.Read();
|
||||
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
vel.SetSize(dim * nq * ne);
|
||||
@@ -1036,12 +827,9 @@ static void PAConvectionApply(const int dim,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPAConvectionApply2D<2,2,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x33: return SmemPAConvectionApply2D<3,3,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x34: return SmemPAConvectionApply2D<3,4,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x44: return SmemPAConvectionApply2D<4,4,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x46: return SmemPAConvectionApply2D<4,6,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x33: return SmemPAConvectionApply2D<3,3,3>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x44: return SmemPAConvectionApply2D<4,4,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x55: return SmemPAConvectionApply2D<5,5,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x58: return SmemPAConvectionApply2D<5,8,2>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x66: return SmemPAConvectionApply2D<6,6,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x77: return SmemPAConvectionApply2D<7,7,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x88: return SmemPAConvectionApply2D<8,8,1>(NE,B,G,Bt,Gt,op,x,y);
|
||||
@@ -1054,12 +842,8 @@ static void PAConvectionApply(const int dim,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPAConvectionApply3D<2,3>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x24: return SmemPAConvectionApply3D<2,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x26: return SmemPAConvectionApply3D<2,6>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x34: return SmemPAConvectionApply3D<3,4>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x35: return SmemPAConvectionApply3D<3,5>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x45: return SmemPAConvectionApply3D<4,5>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x48: return SmemPAConvectionApply3D<4,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x56: return SmemPAConvectionApply3D<5,6>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x67: return SmemPAConvectionApply3D<6,7>(NE,B,G,Bt,Gt,op,x,y);
|
||||
case 0x78: return SmemPAConvectionApply3D<7,8>(NE,B,G,Bt,Gt,op,x,y);
|
||||
|
||||
@@ -167,19 +167,6 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
|
||||
r.SetSize(1);
|
||||
r(0) = c_rho->constant;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* c_rho =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(rho))
|
||||
{
|
||||
const QuadratureFunction &qFun = c_rho->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == nq * nf,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
r.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
r.SetSize(nq * nf);
|
||||
@@ -213,20 +200,6 @@ void DGTraceIntegrator::SetupPA(const FiniteElementSpace &fes, FaceType type)
|
||||
{
|
||||
vel = c_u->GetVec();
|
||||
}
|
||||
else if (VectorQuadratureFunctionCoefficient* c_u =
|
||||
dynamic_cast<VectorQuadratureFunctionCoefficient*>(u))
|
||||
{
|
||||
// Assumed to be in lexicographical ordering
|
||||
const QuadratureFunction &qFun = c_u->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == dim * nq * nf,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
vel.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
vel.SetSize(dim * nq * nf);
|
||||
|
||||
@@ -31,7 +31,7 @@ static void EADiffusionAssemble1D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -53,7 +53,7 @@ static void EADiffusionAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Gj[k1] * D(k1, e) * r_Gi[k1];
|
||||
}
|
||||
A(i1, j1, e) += val;
|
||||
A(i1, j1, e) = val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -75,7 +75,7 @@ static void EADiffusionAssemble2D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -120,7 +120,7 @@ static void EADiffusionAssemble2D(const int NE,
|
||||
+ gbi * D11 * gbj;
|
||||
}
|
||||
}
|
||||
A(i1, i2, j1, j2, e) += val;
|
||||
A(i1, i2, j1, j2, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -144,7 +144,7 @@ static void EADiffusionAssemble3D(const int NE,
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 6, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto A = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -208,7 +208,7 @@ static void EADiffusionAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
A(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
A(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+215
-301
@@ -170,53 +170,47 @@ static void PADiffusionSetup3D(const int Q1D,
|
||||
const Vector &c,
|
||||
Vector &d)
|
||||
{
|
||||
const int NQ = Q1D*Q1D*Q1D;
|
||||
const bool const_c = c.Size() == 1;
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
|
||||
Reshape(c.Read(), Q1D,Q1D,Q1D,NE);
|
||||
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, 6, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
|
||||
auto D = Reshape(d.Write(), NQ, 6, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
|
||||
const double c_detJ = W(qx,qy,qz) * coeff / detJ;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
D(qx,qy,qz,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
D(qx,qy,qz,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
D(qx,qy,qz,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
D(qx,qy,qz,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
D(qx,qy,qz,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
D(qx,qy,qz,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
}
|
||||
}
|
||||
const double J11 = J(q,0,0,e);
|
||||
const double J21 = J(q,1,0,e);
|
||||
const double J31 = J(q,2,0,e);
|
||||
const double J12 = J(q,0,1,e);
|
||||
const double J22 = J(q,1,1,e);
|
||||
const double J32 = J(q,2,1,e);
|
||||
const double J13 = J(q,0,2,e);
|
||||
const double J23 = J(q,1,2,e);
|
||||
const double J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0) : C(q,e);
|
||||
const double c_detJ = W[q] * coeff / detJ;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
D(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
|
||||
D(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
|
||||
D(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
|
||||
D(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
|
||||
D(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
|
||||
D(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -259,7 +253,8 @@ static void PADiffusionSetup(const int dim,
|
||||
}
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
|
||||
const bool force)
|
||||
{
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
@@ -268,7 +263,7 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
if (DeviceCanUseCeed() && !force)
|
||||
{
|
||||
if (ceedDataPtr) { delete ceedDataPtr; }
|
||||
CeedData* ptr = new CeedData();
|
||||
@@ -276,16 +271,17 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
InitCeedCoeff(Q, ptr);
|
||||
return CeedPADiffusionAssemble(fes, *ir, *ptr);
|
||||
}
|
||||
#else
|
||||
MFEM_CONTRACT_VAR(force);
|
||||
#endif
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
const int nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
|
||||
@@ -300,19 +296,6 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = cQ->constant;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == ne*nq,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
coeff.MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
}
|
||||
else
|
||||
{
|
||||
coeff.SetSize(nq * ne);
|
||||
@@ -740,7 +723,6 @@ static void PADiffusionAssembleDiagonal(const int dim,
|
||||
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,B,G,D,Y);
|
||||
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,B,G,D,Y);
|
||||
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,B,G,D,Y);
|
||||
case 0x46: return SmemPADiffusionDiagonal3D<4,6>(NE,B,G,D,Y);
|
||||
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,B,G,D,Y);
|
||||
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,B,G,D,Y);
|
||||
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,B,G,D,Y);
|
||||
@@ -754,17 +736,9 @@ static void PADiffusionAssembleDiagonal(const int dim,
|
||||
|
||||
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAssembleDiagonalPA(ceedDataPtr, diag);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G, pa_data, diag);
|
||||
}
|
||||
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
|
||||
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G, pa_data, diag);
|
||||
}
|
||||
|
||||
|
||||
@@ -1333,33 +1307,7 @@ static void PADiffusionApply3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
// Half of B and G are stored in shared to get B, Bt, G and Gt.
|
||||
// Indices computation for SmemPADiffusionApply3D.
|
||||
static MFEM_HOST_DEVICE inline int qi(const int q, const int d, const int Q)
|
||||
{
|
||||
return (q<=d) ? q : Q-1-q;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int dj(const int q, const int d, const int D)
|
||||
{
|
||||
return (q<=d) ? d : D-1-d;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int qk(const int q, const int d, const int Q)
|
||||
{
|
||||
return (q<=d) ? Q-1-q : q;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int dl(const int q, const int d, const int D)
|
||||
{
|
||||
return (q<=d) ? D-1-d : d;
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline double sign(const int q, const int d)
|
||||
{
|
||||
return (q<=d) ? -1.0 : 1.0;
|
||||
}
|
||||
|
||||
// Shared memory PA Diffusion Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void SmemPADiffusionApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
@@ -1372,27 +1320,28 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= M1D, "");
|
||||
MFEM_VERIFY(Q1D <= M1Q, "");
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, 6, NE);
|
||||
auto d = Reshape(d_.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, 1,
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
MFEM_SHARED double sBG[MQ1*MD1];
|
||||
double (*B)[MD1] = (double (*)[MD1]) sBG;
|
||||
double (*G)[MD1] = (double (*)[MD1]) sBG;
|
||||
double (*Bt)[MQ1] = (double (*)[MQ1]) sBG;
|
||||
double (*Gt)[MQ1] = (double (*)[MQ1]) sBG;
|
||||
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
|
||||
MFEM_SHARED double sBG[2][MQ1*MD1];
|
||||
double (*B)[MD1] = (double (*)[MD1]) (sBG+0);
|
||||
double (*G)[MD1] = (double (*)[MD1]) (sBG+1);
|
||||
double (*Bt)[MQ1] = (double (*)[MQ1]) (sBG+0);
|
||||
double (*Gt)[MQ1] = (double (*)[MQ1]) (sBG+1);
|
||||
MFEM_SHARED double sm0[3][MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[3][MDQ*MDQ*MDQ];
|
||||
double (*X)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
|
||||
@@ -1410,127 +1359,108 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
double (*QDD0)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+0);
|
||||
double (*QDD1)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+1);
|
||||
double (*QDD2)[MD1][MD1] = (double (*)[MD1][MD1]) (sm0+2);
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
X[dz][dy][dx] = x(dx,dy,dz,e);
|
||||
}
|
||||
}
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
}
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
const int i = qi(qx,dy,Q1D);
|
||||
const int j = dj(qx,dy,D1D);
|
||||
const int k = qk(qx,dy,Q1D);
|
||||
const int l = dl(qx,dy,D1D);
|
||||
B[i][j] = b(qx,dy);
|
||||
G[k][l] = g(qx,dy) * sign(qx,dy);
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
G[q][d] = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
double u[D1D], v[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const int i = qi(qx,dx,Q1D);
|
||||
const int j = dj(qx,dx,D1D);
|
||||
const int k = qk(qx,dx,Q1D);
|
||||
const int l = dl(qx,dx,D1D);
|
||||
const double s = sign(qx,dx);
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double coords = X[dz][dy][dx];
|
||||
u[dz] += coords * B[i][j];
|
||||
v[dz] += coords * G[k][l] * s;
|
||||
u += coords * B[qx][dx];
|
||||
v += coords * G[qx][dx];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
DDQ0[dz][dy][qx] = u[dz];
|
||||
DDQ1[dz][dy][qx] = v[dz];
|
||||
DDQ0[dz][dy][qx] = u;
|
||||
DDQ1[dz][dy][qx] = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
double u[D1D], v[D1D], w[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++) { u[dz] = v[dz] = w[dz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const int i = qi(qy,dy,Q1D);
|
||||
const int j = dj(qy,dy,D1D);
|
||||
const int k = qk(qy,dy,Q1D);
|
||||
const int l = dl(qy,dy,D1D);
|
||||
const double s = sign(qy,dy);
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++)
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u[dz] += DDQ1[dz][dy][qx] * B[i][j];
|
||||
v[dz] += DDQ0[dz][dy][qx] * G[k][l] * s;
|
||||
w[dz] += DDQ0[dz][dy][qx] * B[i][j];
|
||||
u += DDQ1[dz][dy][qx] * B[qy][dy];
|
||||
v += DDQ0[dz][dy][qx] * G[qy][dy];
|
||||
w += DDQ0[dz][dy][qx] * B[qy][dy];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; dz++)
|
||||
{
|
||||
DQQ0[dz][qy][qx] = u[dz];
|
||||
DQQ1[dz][qy][qx] = v[dz];
|
||||
DQQ2[dz][qy][qx] = w[dz];
|
||||
DQQ0[dz][qy][qx] = u;
|
||||
DQQ1[dz][qy][qx] = v;
|
||||
DQQ2[dz][qy][qx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const int i = qi(qz,dz,Q1D);
|
||||
const int j = dj(qz,dz,D1D);
|
||||
const int k = qk(qz,dz,Q1D);
|
||||
const int l = dl(qz,dz,D1D);
|
||||
const double s = sign(qz,dz);
|
||||
u[qz] += DQQ0[dz][qy][qx] * B[i][j];
|
||||
v[qz] += DQQ1[dz][qy][qx] * B[i][j];
|
||||
w[qz] += DQQ2[dz][qy][qx] * G[k][l] * s;
|
||||
u += DQQ0[dz][qy][qx] * B[qz][dz];
|
||||
v += DQQ1[dz][qy][qx] * B[qz][dz];
|
||||
w += DQQ2[dz][qy][qx] * G[qz][dz];
|
||||
}
|
||||
QQQ0[qz][qy][qx] = u;
|
||||
QQQ1[qz][qy][qx] = v;
|
||||
QQQ2[qz][qy][qx] = w;
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double O11 = d(qx,qy,qz,0,e);
|
||||
const double O12 = d(qx,qy,qz,1,e);
|
||||
const double O13 = d(qx,qy,qz,2,e);
|
||||
const double O22 = d(qx,qy,qz,3,e);
|
||||
const double O23 = d(qx,qy,qz,4,e);
|
||||
const double O33 = d(qx,qy,qz,5,e);
|
||||
const double gX = u[qz];
|
||||
const double gY = v[qz];
|
||||
const double gZ = w[qz];
|
||||
const int q = qx + ((qy*Q1D) + (qz*Q1D*Q1D));
|
||||
const double O11 = d(q,0,e);
|
||||
const double O12 = d(q,1,e);
|
||||
const double O13 = d(q,2,e);
|
||||
const double O22 = d(q,3,e);
|
||||
const double O23 = d(q,4,e);
|
||||
const double O33 = d(q,5,e);
|
||||
const double gX = QQQ0[qz][qy][qx];
|
||||
const double gY = QQQ1[qz][qy][qx];
|
||||
const double gZ = QQQ2[qz][qy][qx];
|
||||
QQQ0[qz][qy][qx] = (O11*gX) + (O12*gY) + (O13*gZ);
|
||||
QQQ1[qz][qy][qx] = (O12*gX) + (O22*gY) + (O23*gZ);
|
||||
QQQ2[qz][qy][qx] = (O13*gX) + (O23*gY) + (O33*gZ);
|
||||
@@ -1538,112 +1468,78 @@ static void SmemPADiffusionApply3D(const int NE,
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
const int i = qi(q,d,Q1D);
|
||||
const int j = dj(q,d,D1D);
|
||||
const int k = qk(q,d,Q1D);
|
||||
const int l = dl(q,d,D1D);
|
||||
Bt[j][i] = b(q,d);
|
||||
Gt[l][k] = g(q,d) * sign(q,d);
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
Bt[d][q] = b(q,d);
|
||||
Gt[d][q] = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
const int i = qi(qx,dx,Q1D);
|
||||
const int j = dj(qx,dx,D1D);
|
||||
const int k = qk(qx,dx,Q1D);
|
||||
const int l = dl(qx,dx,D1D);
|
||||
const double s = sign(qx,dx);
|
||||
MFEM_UNROLL(MQ1)
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
u += QQQ0[qz][qy][qx] * Gt[dx][qx];
|
||||
v += QQQ1[qz][qy][qx] * Bt[dx][qx];
|
||||
w += QQQ2[qz][qy][qx] * Bt[dx][qx];
|
||||
}
|
||||
QQD0[qz][qy][dx] = u;
|
||||
QQD1[qz][qy][dx] = v;
|
||||
QQD2[qz][qy][dx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
u += QQD0[qz][qy][dx] * Bt[dy][qy];
|
||||
v += QQD1[qz][qy][dx] * Gt[dy][qy];
|
||||
w += QQD2[qz][qy][dx] * Bt[dy][qy];
|
||||
}
|
||||
QDD0[qz][dy][dx] = u;
|
||||
QDD1[qz][dy][dx] = v;
|
||||
QDD2[qz][dy][dx] = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u[qz] += QQQ0[qz][qy][qx] * Gt[l][k] * s;
|
||||
v[qz] += QQQ1[qz][qy][qx] * Bt[j][i];
|
||||
w[qz] += QQQ2[qz][qy][qx] * Bt[j][i];
|
||||
u += QDD0[qz][dy][dx] * Bt[dz][qz];
|
||||
v += QDD1[qz][dy][dx] * Bt[dz][qz];
|
||||
w += QDD2[qz][dy][dx] * Gt[dz][qz];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD0[qz][qy][dx] = u[qz];
|
||||
QQD1[qz][qy][dx] = v[qz];
|
||||
QQD2[qz][qy][dx] = w[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u[Q1D], v[Q1D], w[Q1D];
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz) { u[qz] = v[qz] = w[qz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const int i = qi(qy,dy,Q1D);
|
||||
const int j = dj(qy,dy,D1D);
|
||||
const int k = qk(qy,dy,Q1D);
|
||||
const int l = dl(qy,dy,D1D);
|
||||
const double s = sign(qy,dy);
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
u[qz] += QQD0[qz][qy][dx] * Bt[j][i];
|
||||
v[qz] += QQD1[qz][qy][dx] * Gt[l][k] * s;
|
||||
w[qz] += QQD2[qz][qy][dx] * Bt[j][i];
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QDD0[qz][dy][dx] = u[qz];
|
||||
QDD1[qz][dy][dx] = v[qz];
|
||||
QDD2[qz][dy][dx] = w[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double u[D1D], v[D1D], w[D1D];
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz) { u[dz] = v[dz] = w[dz] = 0.0; }
|
||||
MFEM_UNROLL(MQ1)
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
const int i = qi(qz,dz,Q1D);
|
||||
const int j = dj(qz,dz,D1D);
|
||||
const int k = qk(qz,dz,Q1D);
|
||||
const int l = dl(qz,dz,D1D);
|
||||
const double s = sign(qz,dz);
|
||||
u[dz] += QDD0[qz][dy][dx] * Bt[j][i];
|
||||
v[dz] += QDD1[qz][dy][dx] * Bt[j][i];
|
||||
w[dz] += QDD2[qz][dy][dx] * Gt[l][k] * s;
|
||||
}
|
||||
}
|
||||
MFEM_UNROLL(MD1)
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
y(dx,dy,dz,e) += (u[dz] + v[dz] + w[dz]);
|
||||
y(dx,dy,dz,e) += (u + v + w);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1678,11 +1574,9 @@ static void PADiffusionApply(const int dim,
|
||||
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
const int ID = (D1D << 4 ) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (ID)
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,B,G,D,X,Y);
|
||||
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,B,G,D,X,Y);
|
||||
@@ -1695,13 +1589,11 @@ static void PADiffusionApply(const int dim,
|
||||
default: return PADiffusionApply2D(NE,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
|
||||
if (dim == 3)
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch (ID)
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,B,G,D,X,Y);
|
||||
case 0x24: return SmemPADiffusionApply3D<2,4>(NE,B,G,D,X,Y);
|
||||
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,B,G,D,X,Y);
|
||||
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,B,G,D,X,Y);
|
||||
case 0x46: return SmemPADiffusionApply3D<4,6>(NE,B,G,D,X,Y);
|
||||
@@ -1722,7 +1614,29 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAddMultPA(ceedDataPtr, x, y);
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorSyncArray(ceedDataPtr->v, mem);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
|
||||
+30
-1330
File diff suppressed because it is too large
Load Diff
+1
-11
@@ -114,8 +114,6 @@ void PAHdivMassApply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
|
||||
@@ -240,7 +238,6 @@ void PAHdivMassAssembleDiagonal2D(const int D1D,
|
||||
Vector &_diag)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bc = Reshape(_Bc.Read(), Q1D, D1D);
|
||||
@@ -617,8 +614,6 @@ static void PADivDivApply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Bot = Reshape(_Bot.Read(), D1D-1, Q1D);
|
||||
@@ -982,7 +977,6 @@ static void PADivDivAssembleDiagonal2D(const int D1D,
|
||||
Vector &_diag)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
|
||||
@@ -1406,8 +1400,6 @@ static void PAHdivL2Apply2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto Bo = Reshape(_Bo.Read(), Q1D, D1D-1);
|
||||
auto Gc = Reshape(_Gc.Read(), Q1D, D1D);
|
||||
@@ -1674,8 +1666,6 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
|
||||
Vector &_y)
|
||||
{
|
||||
constexpr static int VDIM = 2;
|
||||
constexpr static int MAX_D1D = HDIV_MAX_D1D;
|
||||
constexpr static int MAX_Q1D = HDIV_MAX_Q1D;
|
||||
|
||||
auto L2Bo = Reshape(_L2Bo.Read(), Q1D, L2D1D);
|
||||
auto Gct = Reshape(_Gct.Read(), D1D, Q1D);
|
||||
@@ -1734,7 +1724,7 @@ static void PAHdivL2ApplyTranspose2D(const int D1D,
|
||||
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
double aX[MAX_D1D];
|
||||
double aX[HDIV_MAX_D1D];
|
||||
|
||||
int osc = 0;
|
||||
for (int c = 0; c < VDIM; ++c) // loop over x, y components
|
||||
|
||||
@@ -30,7 +30,7 @@ static void EAMassAssemble1D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -52,7 +52,7 @@ static void EAMassAssemble1D(const int NE,
|
||||
{
|
||||
val += r_Bi[k1] * r_Bj[k1] * D(k1, e);
|
||||
}
|
||||
M(i1, j1, e) += val;
|
||||
M(i1, j1, e) = val;
|
||||
}
|
||||
}
|
||||
});
|
||||
@@ -72,7 +72,7 @@ static void EAMassAssemble2D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -114,7 +114,7 @@ static void EAMassAssemble2D(const int NE,
|
||||
* s_D[k1][k2];
|
||||
}
|
||||
}
|
||||
M(i1, i2, j1, j2, e) += val;
|
||||
M(i1, i2, j1, j2, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -136,7 +136,7 @@ static void EAMassAssemble3D(const int NE,
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
auto M = Reshape(eadata.Write(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -189,7 +189,7 @@ static void EAMassAssemble3D(const int NE,
|
||||
}
|
||||
}
|
||||
}
|
||||
M(i1, i2, i3, j1, j2, j3, e) += val;
|
||||
M(i1, i2, i3, j1, j2, j3, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+57
-101
@@ -23,9 +23,8 @@ namespace mfem
|
||||
|
||||
// PA Mass Assemble kernel
|
||||
|
||||
void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
|
||||
{
|
||||
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
@@ -34,7 +33,7 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
ElementTransformation *T = mesh->GetElementTransformation(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
if (DeviceCanUseCeed() && !force)
|
||||
{
|
||||
if (ceedDataPtr) { delete ceedDataPtr; }
|
||||
CeedData* ptr = new CeedData();
|
||||
@@ -46,57 +45,27 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetMesh()->GetNE();
|
||||
nq = ir->GetNPoints();
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const int flags = GeometricFactors::JACOBIANS |
|
||||
GeometricFactors::COORDINATES;
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
geom = mesh->GetGeometricFactors(*ir, flags, mode, temp_type);
|
||||
maps = &el.GetDofToQuad(*ir, mode);
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
|
||||
GeometricFactors::JACOBIANS);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
Vector *coeff{nullptr};
|
||||
bool own_coeff{true};
|
||||
Vector coeff;
|
||||
if (Q == nullptr)
|
||||
{
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(1);
|
||||
(*coeff)(0) = 1.0;
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = 1.0;
|
||||
}
|
||||
else if (ConstantCoefficient* cQ = dynamic_cast<ConstantCoefficient*>(Q))
|
||||
{
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(1);
|
||||
(*coeff)(0) = cQ->constant;
|
||||
}
|
||||
else if (QuadratureCoefficient* cQ = dynamic_cast<QuadratureCoefficient*>(Q))
|
||||
{
|
||||
coeff = cQ->Data();
|
||||
own_coeff = false;
|
||||
}
|
||||
else if (QuadratureFunctionCoefficient* cQ =
|
||||
dynamic_cast<QuadratureFunctionCoefficient*>(Q))
|
||||
{
|
||||
const QuadratureFunction &qFun = cQ->GetQuadFunction();
|
||||
MFEM_VERIFY(qFun.Size() == nq * ne,
|
||||
"Incompatible QuadratureFunction dimension \n");
|
||||
|
||||
MFEM_VERIFY(ir == &qFun.GetSpace()->GetElementIntRule(0),
|
||||
"IntegrationRule used within integrator and in"
|
||||
" QuadratureFunction appear to be different");
|
||||
qFun.Read();
|
||||
coeff->MakeRef(const_cast<QuadratureFunction &>(qFun),0);
|
||||
coeff.SetSize(1);
|
||||
coeff(0) = cQ->constant;
|
||||
}
|
||||
else
|
||||
{
|
||||
coeff = new Vector;
|
||||
coeff->SetSize(nq * ne);
|
||||
auto C = Reshape(coeff->HostWrite(), nq, ne);
|
||||
coeff.SetSize(nq * ne);
|
||||
auto C = Reshape(coeff.HostWrite(), nq, ne);
|
||||
for (int e = 0; e < ne; ++e)
|
||||
{
|
||||
ElementTransformation& T = *fes.GetElementTransformation(e);
|
||||
@@ -111,11 +80,11 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int NQ = nq;
|
||||
const bool const_c = coeff->Size() == 1;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
auto w = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
|
||||
auto C =
|
||||
const_c ? Reshape(coeff->Read(), 1,1) : Reshape(coeff->Read(), NQ,NE);
|
||||
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
@@ -134,43 +103,28 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
|
||||
if (dim==3)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int Q1D = quad1D;
|
||||
const bool const_c = coeff->Size() == 1;
|
||||
const auto W = Reshape(ir->GetWeights().Read(),Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(geom->J.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const auto C = const_c ?
|
||||
Reshape(coeff->Read(), 1,1,1,1) :
|
||||
Reshape(coeff->Read(), Q1D,Q1D,Q1D,NE);
|
||||
auto V = Reshape(pa_data.Write(), Q1D,Q1D,Q1D,NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
const int NQ = nq;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
auto W = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
|
||||
auto C =
|
||||
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ,NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
|
||||
V(qx,qy,qz,e) = W(qx,qy,qz) * coeff * detJ;
|
||||
}
|
||||
}
|
||||
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
|
||||
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
|
||||
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double coeff = const_c ? C(0,0) : C(q,e);
|
||||
v(q,e) = W[q] * coeff * detJ;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
if (own_coeff) { delete coeff; }
|
||||
}
|
||||
|
||||
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
@@ -472,12 +426,8 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
|
||||
case 0x24: return SmemPAMassAssembleDiagonal3D<2,4>(NE,B,D,Y);
|
||||
case 0x26: return SmemPAMassAssembleDiagonal3D<2,6>(NE,B,D,Y);
|
||||
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
|
||||
case 0x35: return SmemPAMassAssembleDiagonal3D<3,5>(NE,B,D,Y);
|
||||
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
|
||||
case 0x48: return SmemPAMassAssembleDiagonal3D<4,8>(NE,B,D,Y);
|
||||
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
|
||||
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
|
||||
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
|
||||
@@ -490,16 +440,8 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
|
||||
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAssembleDiagonalPA(ceedDataPtr, diag);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
}
|
||||
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
|
||||
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
}
|
||||
|
||||
|
||||
@@ -697,7 +639,6 @@ static void SmemPAMassApply2D(const int NE,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
@@ -961,7 +902,6 @@ static void SmemPAMassApply3D(const int NE,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
@@ -1211,13 +1151,10 @@ static void PAMassApply(const int dim,
|
||||
case 0x24: return SmemPAMassApply2D<2,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply2D<3,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply2D<3,5,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply2D<3,6,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply2D<4,6,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply2D<4,8,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x57: return SmemPAMassApply2D<5,7,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x58: return SmemPAMassApply2D<5,8,2>(NE,B,Bt,D,X,Y);
|
||||
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
|
||||
@@ -1225,7 +1162,6 @@ static void PAMassApply(const int dim,
|
||||
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
mfem::out << "Unknown 2D kernel 0x" << std::hex << id << std::endl;
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
@@ -1234,9 +1170,7 @@ static void PAMassApply(const int dim,
|
||||
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
|
||||
case 0x24: return SmemPAMassApply3D<2,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply3D<3,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply3D<3,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x37: return SmemPAMassApply3D<3,7>(NE,B,Bt,D,X,Y);
|
||||
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply3D<4,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply3D<4,8>(NE,B,Bt,D,X,Y);
|
||||
@@ -1248,8 +1182,8 @@ static void PAMassApply(const int dim,
|
||||
case 0x9A: return SmemPAMassApply3D<9,10>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
mfem::out << "Unknown 3D kernel 0x" << std::hex << id << std::endl;
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
@@ -1258,7 +1192,29 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
#ifdef MFEM_USE_CEED
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
CeedAddMultPA(ceedDataPtr, x, y);
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorSyncArray(ceedDataPtr->v, mem);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
|
||||
@@ -25,7 +25,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
if (ne == 0) { return; }
|
||||
const int dofs = fes.GetFE(0)->GetDof();
|
||||
auto A = Reshape(ea_data_tmp.Write(), dofs, dofs, ne);
|
||||
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
|
||||
auto AT = Reshape(ea_data.Write(), dofs, dofs, ne);
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < dofs; i++)
|
||||
|
||||
@@ -9,14 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void PAHcurlSetup2D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
@@ -24,7 +22,6 @@ void PAHcurlSetup2D(const int Q1D,
|
||||
Vector &op);
|
||||
|
||||
void PAHcurlSetup3D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
@@ -175,36 +172,16 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
|
||||
pa_data.SetSize(symmDims * nq * ne, Device::GetMemoryType());
|
||||
|
||||
const int coeffDim = VQ ? VQ->GetVDim() : 1;
|
||||
|
||||
Vector coeff(coeffDim * ne * nq);
|
||||
Vector coeff(ne * nq);
|
||||
coeff = 1.0;
|
||||
auto coeffh = Reshape(coeff.HostWrite(), coeffDim, nq, ne);
|
||||
if (Q || VQ)
|
||||
if (Q)
|
||||
{
|
||||
Vector D(VQ ? coeffDim : 0);
|
||||
if (VQ)
|
||||
{
|
||||
MFEM_VERIFY(coeffDim == dim, "");
|
||||
}
|
||||
|
||||
for (int e=0; e<ne; ++e)
|
||||
{
|
||||
ElementTransformation *tr = mesh->GetElementTransformation(e);
|
||||
for (int p=0; p<nq; ++p)
|
||||
{
|
||||
if (VQ)
|
||||
{
|
||||
VQ->Eval(D, *tr, ir->IntPoint(p));
|
||||
for (int i=0; i<coeffDim; ++i)
|
||||
{
|
||||
coeffh(i, p, e) = D[i];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
coeffh(0, p, e) = Q->Eval(*tr, ir->IntPoint(p));
|
||||
}
|
||||
coeff[p + (e * nq)] = Q->Eval(*tr, ir->IntPoint(p));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -213,12 +190,12 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
|
||||
if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
|
||||
{
|
||||
PAHcurlSetup3D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
|
||||
{
|
||||
PAHcurlSetup2D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (el->GetDerivType() == mfem::FiniteElement::DIV && dim == 3)
|
||||
@@ -371,12 +348,12 @@ void MixedVectorGradientIntegrator::AssemblePA(const FiniteElementSpace
|
||||
// Use the same setup functions as VectorFEMassIntegrator.
|
||||
if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 3)
|
||||
{
|
||||
PAHcurlSetup3D(quad1D, 1, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else if (test_el->GetDerivType() == mfem::FiniteElement::CURL && dim == 2)
|
||||
{
|
||||
PAHcurlSetup2D(quad1D, 1, ne, ir->GetWeights(), geom->J,
|
||||
PAHcurlSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
|
||||
coeff, pa_data);
|
||||
}
|
||||
else
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
// Implementation of Coefficient class
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
@@ -22,13 +21,6 @@ namespace mfem
|
||||
|
||||
using namespace std;
|
||||
|
||||
double QuadratureCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
auto coeff = mfem::Reshape(qData->HostRead(), nip, NE);
|
||||
return coeff(ip.index, T.ElementNo);
|
||||
}
|
||||
|
||||
double PWConstCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
|
||||
+1
-31
@@ -30,10 +30,7 @@ class ParMesh;
|
||||
/** @brief Base class Coefficients that optionally depend on space and time.
|
||||
These are used by the BilinearFormIntegrator, LinearFormIntegrator, and
|
||||
NonlinearFormIntegrator classes to represent the physical coefficients in
|
||||
the PDEs that are being discretized. This class can also be used in a more
|
||||
general way to represent functions that don't necessarily belong to a FE
|
||||
space, e.g., to project onto GridFunctions to use as initial conditions,
|
||||
exact solutions, etc. See, e.g., ex4 or ex22 for these uses. */
|
||||
the PDEs that are being discretized. */
|
||||
class Coefficient
|
||||
{
|
||||
protected:
|
||||
@@ -87,33 +84,6 @@ public:
|
||||
{ return (constant); }
|
||||
};
|
||||
|
||||
|
||||
/// class for quadrature coefficient
|
||||
class QuadratureCoefficient : public Coefficient
|
||||
{
|
||||
|
||||
private:
|
||||
const int nip;
|
||||
const int NE;
|
||||
public:
|
||||
Vector *qData{nullptr};
|
||||
|
||||
//Set external data
|
||||
QuadratureCoefficient(Vector *Data, int in_nip, int in_NE)
|
||||
: qData(Data), nip(in_nip), NE(in_NE)
|
||||
{ }
|
||||
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
|
||||
Vector *Data()
|
||||
{
|
||||
return qData;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
/// class for piecewise constant coefficient
|
||||
/** @brief A piecewise constant coefficient with the constants keyed
|
||||
off the element attribute numbers. */
|
||||
class PWConstCoefficient : public Coefficient
|
||||
|
||||
+53
-135
@@ -342,10 +342,11 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
int ci)
|
||||
{
|
||||
FiniteElementSpace * fes = blfr->FESpace();
|
||||
|
||||
int vsize = fes->GetVSize();
|
||||
|
||||
// Allocate temporary vectors
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
|
||||
// Extract the real and imaginary parts of the input vectors
|
||||
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
|
||||
@@ -359,7 +360,8 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC) { b_i *= -1.0; }
|
||||
|
||||
int tvsize = fes->GetTrueVSize();
|
||||
OperatorHandle A_r, A_i;
|
||||
SparseMatrix * A_r = nullptr;
|
||||
SparseMatrix * A_i = nullptr;
|
||||
|
||||
X.SetSize(2 * tvsize);
|
||||
B.SetSize(2 * tvsize);
|
||||
@@ -372,39 +374,42 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
|
||||
if (RealInteg())
|
||||
{
|
||||
A_r = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
|
||||
b_0 = b_r;
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, A_r, X_0, B_0, ci);
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_r, X_0, B_0, ci);
|
||||
X_r = X_0; B_r = B_0;
|
||||
|
||||
b_0 = b_i;
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, A_r, X_0, B_0, ci);
|
||||
blfr->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_r, X_0, B_0, ci);
|
||||
X_i = X_0; B_i = B_0;
|
||||
|
||||
if (ImagInteg())
|
||||
{
|
||||
A_i = new SparseMatrix;
|
||||
blfi->SetDiagonalPolicy(mfem::Matrix::DiagonalPolicy::DIAG_ZERO);
|
||||
|
||||
b_0 = 0.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, false);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, false);
|
||||
B_r -= B_0;
|
||||
|
||||
b_0 = 0.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, false);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, false);
|
||||
B_i += B_0;
|
||||
}
|
||||
}
|
||||
else if (ImagInteg())
|
||||
{
|
||||
A_i = new SparseMatrix;
|
||||
blfi->SetDiagonalPolicy(diag_policy);
|
||||
|
||||
b_0 = b_i;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, A_i, X_0, B_0, ci);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_r, b_0, *A_i, X_0, B_0, ci);
|
||||
X_r = X_0; B_i = B_0;
|
||||
|
||||
b_0 = b_r; b_0 *= -1.0;
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, A_i, X_0, B_0, ci);
|
||||
blfi->FormLinearSystem(ess_tdof_list, x_i, b_0, *A_i, X_0, B_0, ci);
|
||||
X_i = X_0; B_r = B_0; B_r *= -1.0;
|
||||
}
|
||||
else
|
||||
@@ -412,55 +417,16 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify RHS and offdiagonal blocks (imaginary parts of the matrix) to
|
||||
// conform with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
int j = ess_tdof_list[k];
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
|
||||
{
|
||||
B_i *= -1.0;
|
||||
b_i *= -1.0;
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
|
||||
A_i.Type() == Operator::MFEM_SPARSEMAT )
|
||||
{
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
|
||||
A_i.As<SparseMatrix>(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
ComplexOperator * A_op =
|
||||
new ComplexOperator(A_r.Ptr(),
|
||||
A_i.Ptr(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
ComplexSparseMatrix * A_sp;
|
||||
A_sp = new ComplexSparseMatrix(A_r, A_i, true, true, conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -468,60 +434,31 @@ SesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
OperatorHandle &A)
|
||||
|
||||
{
|
||||
OperatorHandle A_r, A_i;
|
||||
SparseMatrix * A_r = nullptr;
|
||||
SparseMatrix * A_i = nullptr;
|
||||
|
||||
if (RealInteg())
|
||||
{
|
||||
A_r = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
blfr->FormSystemMatrix(ess_tdof_list, A_r);
|
||||
blfr->FormSystemMatrix(ess_tdof_list, *A_r);
|
||||
}
|
||||
if (ImagInteg())
|
||||
{
|
||||
blfi->SetDiagonalPolicy(RealInteg() ?
|
||||
mfem::Matrix::DiagonalPolicy::DIAG_ZERO :
|
||||
diag_policy);
|
||||
blfi->FormSystemMatrix(ess_tdof_list, A_i);
|
||||
A_i = new SparseMatrix;
|
||||
blfr->SetDiagonalPolicy(diag_policy);
|
||||
blfi->FormSystemMatrix(ess_tdof_list, *A_i);
|
||||
}
|
||||
if (!RealInteg() && !ImagInteg())
|
||||
{
|
||||
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if (A_i.Is<ConstrainedOperator>())
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
if ( A_r.Type() == Operator::MFEM_SPARSEMAT ||
|
||||
A_i.Type() == Operator::MFEM_SPARSEMAT )
|
||||
{
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r.As<SparseMatrix>(),
|
||||
A_i.As<SparseMatrix>(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
ComplexOperator * A_op =
|
||||
new ComplexOperator(A_r.Ptr(),
|
||||
A_i.Ptr(),
|
||||
A_r.OwnsOperator(),
|
||||
A_i.OwnsOperator(),
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
ComplexSparseMatrix * A_sp =
|
||||
new ComplexSparseMatrix(A_r, A_i, true, true, conv);
|
||||
A.Reset<ComplexSparseMatrix>(A_sp, true);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -709,7 +646,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
|
||||
tdof_offsets = new HYPRE_Int[n+1];
|
||||
|
||||
for (int i = 0; i <= n; i++)
|
||||
for (int i=0; i<=n; i++)
|
||||
{
|
||||
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
|
||||
}
|
||||
@@ -717,8 +654,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
|
||||
|
||||
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
ParLinearForm *plf_r,
|
||||
ParLinearForm *plf_i,
|
||||
ParLinearForm *plf_r, ParLinearForm *plf_i,
|
||||
ComplexOperator::Convention
|
||||
convention)
|
||||
: Vector(2*(pfes->GetVSize())),
|
||||
@@ -734,7 +670,7 @@ ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
|
||||
int n = (HYPRE_AssumedPartitionCheck()) ? 2 : pfes->GetNRanks();
|
||||
tdof_offsets = new HYPRE_Int[n+1];
|
||||
|
||||
for (int i = 0; i <= n; i++)
|
||||
for (int i=0; i<=n; i++)
|
||||
{
|
||||
tdof_offsets[i] = 2 * tdof_offsets_fes[i];
|
||||
}
|
||||
@@ -881,8 +817,7 @@ ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
{}
|
||||
|
||||
ParSesquilinearForm::ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
ParBilinearForm *pbfr,
|
||||
ParBilinearForm *pbfi,
|
||||
ParBilinearForm *pbfr, ParBilinearForm *pbfi,
|
||||
ComplexOperator::Convention convention)
|
||||
: conv(convention),
|
||||
pblfr(new ParBilinearForm(pf,pbfr)),
|
||||
@@ -978,10 +913,9 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
int vsize = pfes->GetVSize();
|
||||
|
||||
// Allocate temporary vectors
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
Vector b_0(vsize); b_0 = 0.0;
|
||||
|
||||
// Extract the real and imaginary parts of the input vectors
|
||||
MFEM_ASSERT(x.Size() == 2 * vsize, "Input GridFunction of incorrect size!");
|
||||
Vector x_r(x.GetData(), vsize);
|
||||
Vector x_i(&(x.GetData())[vsize], vsize);
|
||||
|
||||
@@ -1040,34 +974,25 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
// Modify RHS and offdiagonal blocks (Imaginary parts of the matrix) to
|
||||
// conform with standard essential BC treatment i.e. zero out rows and
|
||||
// columns and place ones on the diagonal.
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
// Modify RHS to conform with standard essential BC treatment
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
int j=ess_tdof_list[k];
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if ( A_i.Type() == Operator::Hypre_ParCSR )
|
||||
{
|
||||
HypreParMatrix * Ah;
|
||||
A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix *Aih = *Ah;
|
||||
for (int k = 0; k < n; k++)
|
||||
HypreParMatrix * Ah; A_i.Get(Ah);
|
||||
int n = ess_tdof_list.Size();
|
||||
hypre_ParCSRMatrix * Aih =
|
||||
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
|
||||
for (int k=0; k<n; k++)
|
||||
{
|
||||
int j = ess_tdof_list[k];
|
||||
int j=ess_tdof_list[k];
|
||||
Aih->diag->data[Aih->diag->i[j]] = 0.0;
|
||||
B_r(j) = X_r(j);
|
||||
B_i(j) = X_i(j);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
if (conv == ComplexOperator::BLOCK_SYMMETRIC)
|
||||
@@ -1075,7 +1000,6 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
B_i *= -1.0;
|
||||
b_i *= -1.0;
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
A.Clear();
|
||||
if ( A_r.Type() == Operator::Hypre_ParCSR ||
|
||||
@@ -1099,8 +1023,6 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
@@ -1121,27 +1043,25 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
MFEM_ABORT("Both Real and Imaginary part of the Sesquilinear form are empty");
|
||||
}
|
||||
|
||||
// Modify offdiagonal blocks (Imaginary parts of the matrix) to conform with
|
||||
// standard essential BC treatment i.e. zero out rows and columns and place
|
||||
// ones on the diagonal.
|
||||
if (RealInteg() && ImagInteg())
|
||||
{
|
||||
// Modify offdiagonal blocks (imaginary parts of the matrix) to conform
|
||||
// with standard essential BC treatment
|
||||
if ( A_i.Type() == Operator::Hypre_ParCSR )
|
||||
{
|
||||
int n = ess_tdof_list.Size();
|
||||
HypreParMatrix * Ah;
|
||||
A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix * Aih = *Ah;
|
||||
for (int k = 0; k < n; k++)
|
||||
int j;
|
||||
|
||||
HypreParMatrix * Ah; A_i.Get(Ah);
|
||||
hypre_ParCSRMatrix * Aih =
|
||||
(hypre_ParCSRMatrix *)const_cast<HypreParMatrix&>(*Ah);
|
||||
for (int k=0; k<n; k++)
|
||||
{
|
||||
int j = ess_tdof_list[k];
|
||||
j=ess_tdof_list[k];
|
||||
Aih->diag->data[Aih->diag->i[j]] = 0.0;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
A_i.As<ConstrainedOperator>()->SetDiagonalPolicy
|
||||
(mfem::Operator::DiagonalPolicy::DIAG_ZERO);
|
||||
}
|
||||
}
|
||||
|
||||
// A = A_r + i A_i
|
||||
@@ -1167,8 +1087,6 @@ ParSesquilinearForm::FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
conv);
|
||||
A.Reset<ComplexOperator>(A_op, true);
|
||||
}
|
||||
A_r.SetOperatorOwner(false);
|
||||
A_i.SetOperatorOwner(false);
|
||||
}
|
||||
|
||||
void
|
||||
|
||||
+1
-31
@@ -219,21 +219,6 @@ public:
|
||||
void SetConvention(const ComplexOperator::Convention &
|
||||
convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::FULL (default)
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
blfr->SetAssemblyLevel(assembly_level);
|
||||
blfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
BilinearForm & real() { return *blfr; }
|
||||
BilinearForm & imag() { return *blfi; }
|
||||
const BilinearForm & real() const { return *blfr; }
|
||||
@@ -493,7 +478,7 @@ public:
|
||||
/** Class for a parallel sesquilinear form
|
||||
|
||||
A sesquilinear form is a generalization of a bilinear form to complex-valued
|
||||
fields. Sesquilinear forms are linear in the second argument but the
|
||||
fields. Sesquilinear forms are linear in the second argument but but the
|
||||
first argument involves a complex conjugate in the sense that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
@@ -539,21 +524,6 @@ public:
|
||||
void SetConvention(const ComplexOperator::Convention &
|
||||
convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::FULL (default)
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
pblfr->SetAssemblyLevel(assembly_level);
|
||||
pblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
ParBilinearForm & real() { return *pblfr; }
|
||||
ParBilinearForm & imag() { return *pblfi; }
|
||||
const ParBilinearForm & real() const { return *pblfr; }
|
||||
|
||||
+8
-40
@@ -415,6 +415,9 @@ void VisItDataCollection::SetMesh(MPI_Comm comm, Mesh *new_mesh)
|
||||
void VisItDataCollection::RegisterField(const std::string& name,
|
||||
GridFunction *gf)
|
||||
{
|
||||
DataCollection::RegisterField(name, gf);
|
||||
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim());
|
||||
|
||||
int LOD = 1;
|
||||
if (gf->FESpace()->GetNURBSext())
|
||||
{
|
||||
@@ -428,27 +431,6 @@ void VisItDataCollection::RegisterField(const std::string& name,
|
||||
}
|
||||
}
|
||||
|
||||
DataCollection::RegisterField(name, gf);
|
||||
field_info_map[name] = VisItFieldInfo("nodes", gf->VectorDim(), LOD);
|
||||
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
|
||||
}
|
||||
|
||||
void VisItDataCollection::RegisterQField(const std::string& name,
|
||||
QuadratureFunction *qf)
|
||||
{
|
||||
int LOD = -1;
|
||||
Mesh *mesh = qf->GetSpace()->GetMesh();
|
||||
for (int e=0; e<qf->GetSpace()->GetNE(); e++)
|
||||
{
|
||||
int locLOD = GlobGeometryRefiner.GetRefinementLevelFromElems(
|
||||
mesh->GetElementBaseGeometry(e),
|
||||
qf->GetElementIntRule(e).GetNPoints());
|
||||
|
||||
LOD = std::max(LOD,locLOD);
|
||||
}
|
||||
|
||||
DataCollection::RegisterQField(name, qf);
|
||||
field_info_map[name] = VisItFieldInfo("elements", 1, LOD);
|
||||
visit_levels_of_detail = std::max(visit_levels_of_detail, LOD);
|
||||
}
|
||||
|
||||
@@ -616,28 +598,14 @@ void VisItDataCollection::LoadFields()
|
||||
// TODO: 1) load parallel GridFunction on one processor
|
||||
if (serial)
|
||||
{
|
||||
if ((it->second).association == "nodes")
|
||||
{
|
||||
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
|
||||
}
|
||||
else if ((it->second).association == "elements")
|
||||
{
|
||||
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
|
||||
}
|
||||
field_map.Register(it->first, new GridFunction(mesh, file), own_data);
|
||||
}
|
||||
else
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if ((it->second).association == "nodes")
|
||||
{
|
||||
field_map.Register(
|
||||
it->first,
|
||||
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
|
||||
}
|
||||
else if ((it->second).association == "elements")
|
||||
{
|
||||
q_field_map.Register(it->first, new QuadratureFunction(mesh, file), own_data);
|
||||
}
|
||||
field_map.Register(
|
||||
it->first,
|
||||
new ParGridFunction(dynamic_cast<ParMesh*>(mesh), file), own_data);
|
||||
#else
|
||||
error = READ_ERROR;
|
||||
MFEM_WARNING("Reading parallel format in serial is not supported");
|
||||
@@ -672,7 +640,7 @@ std::string VisItDataCollection::GetVisItRootString()
|
||||
{
|
||||
ftags["assoc"] = picojson::value((it->second).association);
|
||||
ftags["comps"] = picojson::value(to_string((it->second).num_components));
|
||||
ftags["lod"] = picojson::value(to_string((it->second).lod));
|
||||
ftags["lod"] = picojson::value(to_string(visit_levels_of_detail));
|
||||
field["path"] = picojson::value(path_str + it->first + file_ext_format);
|
||||
field["tags"] = picojson::value(ftags);
|
||||
fields[it->first] = picojson::value(field);
|
||||
|
||||
+3
-10
@@ -391,10 +391,9 @@ class VisItFieldInfo
|
||||
public:
|
||||
std::string association;
|
||||
int num_components;
|
||||
int lod;
|
||||
VisItFieldInfo() { association = ""; num_components = 0; lod = 1;}
|
||||
VisItFieldInfo(std::string _association, int _num_components, int _lod = 1)
|
||||
{ association = _association; num_components = _num_components; lod =_lod;}
|
||||
VisItFieldInfo() { association = ""; num_components = 0; }
|
||||
VisItFieldInfo(std::string _association, int _num_components)
|
||||
{ association = _association; num_components = _num_components; }
|
||||
};
|
||||
|
||||
/// Data collection with VisIt I/O routines
|
||||
@@ -446,12 +445,6 @@ public:
|
||||
/// Add a grid function to the collection and update the root file
|
||||
virtual void RegisterField(const std::string& field_name, GridFunction *gf);
|
||||
|
||||
/// Add a quadrature function to the collection and update the root file.
|
||||
/** Visualization of quadrature function is not supported in VisIt(3.12).
|
||||
A patch has been sent to VisIt developers in June 2020. */
|
||||
virtual void RegisterQField(const std::string& q_field_name,
|
||||
QuadratureFunction *qf);
|
||||
|
||||
/// Set VisIt parameter: default levels of detail for the MultiresControl
|
||||
void SetLevelsOfDetail(int levels_of_detail);
|
||||
|
||||
|
||||
+1
-4
@@ -37,8 +37,7 @@ public:
|
||||
ClosedUniform = 4, ///< Nodes: x_i = i/(n-1), i=0,...,n-1
|
||||
OpenHalfUniform = 5, ///< Nodes: x_i = (i+1/2)/n, i=0,...,n-1
|
||||
Serendipity = 6, ///< Serendipity basis (squares / cubes)
|
||||
ClosedGL = 7, ///< Closed GaussLegendre
|
||||
NumBasisTypes = 8 /**< Keep track of maximum types to prevent
|
||||
NumBasisTypes = 7 /**< Keep track of maximum types to prevent
|
||||
hard-coding */
|
||||
};
|
||||
/** @brief If the input does not represents a valid BasisType, abort with an
|
||||
@@ -70,7 +69,6 @@ public:
|
||||
case ClosedUniform: return Quadrature1D::ClosedUniform;
|
||||
case OpenHalfUniform: return Quadrature1D::OpenHalfUniform;
|
||||
case Serendipity: return Quadrature1D::GaussLobatto;
|
||||
case ClosedGL: return Quadrature1D::ClosedGL;
|
||||
}
|
||||
return Quadrature1D::Invalid;
|
||||
}
|
||||
@@ -84,7 +82,6 @@ public:
|
||||
case Quadrature1D::OpenUniform: return OpenUniform;
|
||||
case Quadrature1D::ClosedUniform: return ClosedUniform;
|
||||
case Quadrature1D::OpenHalfUniform: return OpenHalfUniform;
|
||||
case Quadrature1D::ClosedGL: return ClosedGL;
|
||||
}
|
||||
return Invalid;
|
||||
}
|
||||
|
||||
+6
-6
@@ -944,7 +944,7 @@ const Operator *FiniteElementSpace::GetFaceRestriction(
|
||||
}
|
||||
|
||||
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
const IntegrationRule &ir, const DofToQuad::Mode mode) const
|
||||
const IntegrationRule &ir) const
|
||||
{
|
||||
for (int i = 0; i < E2Q_array.Size(); i++)
|
||||
{
|
||||
@@ -952,13 +952,13 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
if (qi->IntRule == &ir) { return qi; }
|
||||
}
|
||||
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir, mode);
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, ir);
|
||||
E2Q_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
|
||||
const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
const QuadratureSpace &qs, const DofToQuad::Mode mode) const
|
||||
const QuadratureSpace &qs) const
|
||||
{
|
||||
for (int i = 0; i < E2Q_array.Size(); i++)
|
||||
{
|
||||
@@ -966,7 +966,7 @@ const QuadratureInterpolator *FiniteElementSpace::GetQuadratureInterpolator(
|
||||
if (qi->qspace == &qs) { return qi; }
|
||||
}
|
||||
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs, mode);
|
||||
QuadratureInterpolator *qi = new QuadratureInterpolator(*this, qs);
|
||||
E2Q_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
@@ -983,8 +983,8 @@ const FaceQuadratureInterpolator
|
||||
if (qi->IntRule == &ir) { return qi; }
|
||||
}
|
||||
|
||||
FaceQuadratureInterpolator *qi =
|
||||
new FaceQuadratureInterpolator(*this, ir, type);
|
||||
FaceQuadratureInterpolator *qi = new FaceQuadratureInterpolator(*this, ir,
|
||||
type);
|
||||
E2IFQ_array.Append(qi);
|
||||
return qi;
|
||||
}
|
||||
|
||||
+2
-8
@@ -367,7 +367,7 @@ public:
|
||||
All elements will use the same IntegrationRule, @a ir as the target
|
||||
quadrature points. */
|
||||
const QuadratureInterpolator *GetQuadratureInterpolator(
|
||||
const IntegrationRule &ir, const DofToQuad::Mode = DofToQuad::FULL) const;
|
||||
const IntegrationRule &ir) const;
|
||||
|
||||
/** @brief Return a QuadratureInterpolator that interpolates E-vectors to
|
||||
quadrature point values and/or derivatives (Q-vectors). */
|
||||
@@ -378,7 +378,7 @@ public:
|
||||
The target quadrature points in the elements are described by the given
|
||||
QuadratureSpace, @a qs. */
|
||||
const QuadratureInterpolator *GetQuadratureInterpolator(
|
||||
const QuadratureSpace &qs, const DofToQuad::Mode = DofToQuad::FULL) const;
|
||||
const QuadratureSpace &qs) const;
|
||||
|
||||
/** @brief Return a FaceQuadratureInterpolator that interpolates E-vectors to
|
||||
quadrature point values and/or derivatives (Q-vectors). */
|
||||
@@ -756,12 +756,6 @@ public:
|
||||
/// Return the total number of quadrature points.
|
||||
int GetSize() const { return size; }
|
||||
|
||||
/// Returns the mesh
|
||||
inline Mesh *GetMesh() const { return mesh; }
|
||||
|
||||
/// Returns number of elements in the mesh.
|
||||
inline int GetNE() const { return mesh->GetNE(); }
|
||||
|
||||
/// Get the IntegrationRule associated with mesh element @a idx.
|
||||
const IntegrationRule &GetElementIntRule(int idx) const
|
||||
{ return *int_rule[mesh->GetElementBaseGeometry(idx)]; }
|
||||
|
||||
+33
-141
@@ -1344,14 +1344,15 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, Times-1);
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule(Times-1);
|
||||
for (int i = 1; i < Times; i++)
|
||||
if (ir == NULL)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(i-1);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = ip.z = 0.0;
|
||||
ir = new IntegrationRule(Times-1);
|
||||
for (int i = 1; i < Times; i++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(i-1);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -1363,17 +1364,18 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, ((Times-1)*(Times-2))/2);
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
|
||||
for (int k = 0, j = 1; j < Times-1; j++)
|
||||
for (int i = 1; i < Times-j; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
if (ir == NULL)
|
||||
{
|
||||
ir = new IntegrationRule(((Times-1)*(Times-2))/2);
|
||||
for (int k = 0, j = 1; j < Times-1; j++)
|
||||
for (int i = 1; i < Times-j; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -1384,17 +1386,18 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
return NULL;
|
||||
}
|
||||
ir = FindInIntPts(Geom, (Times-1)*(Times-1));
|
||||
if (ir) { return ir; }
|
||||
|
||||
ir = new IntegrationRule((Times-1)*(Times-1));
|
||||
for (int k = 0, j = 1; j < Times; j++)
|
||||
for (int i = 1; i < Times; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
if (ir == NULL)
|
||||
{
|
||||
ir = new IntegrationRule((Times-1)*(Times-1));
|
||||
for (int k = 0, j = 1; j < Times; j++)
|
||||
for (int i = 1; i < Times; i++, k++)
|
||||
{
|
||||
IntegrationPoint &ip = ir->IntPoint(k);
|
||||
ip.x = double(i) / Times;
|
||||
ip.y = double(j) / Times;
|
||||
ip.z = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -1402,121 +1405,10 @@ const IntegrationRule *GeometryRefiner::RefineInterior(Geometry::Type Geom,
|
||||
mfem_error("GeometryRefiner::RefineInterior(...)");
|
||||
}
|
||||
|
||||
MFEM_ASSERT(ir != NULL, "Failed to construct the refined IntegrationRule.");
|
||||
IntPts[Geom].Append(ir);
|
||||
|
||||
if (ir) { IntPts[Geom].Append(ir); }
|
||||
return ir;
|
||||
}
|
||||
|
||||
|
||||
int GeometryRefiner::GetRefinementLevelFromPoints(Geometry::Type geom, int Npts)
|
||||
{
|
||||
switch (geom)
|
||||
{
|
||||
case Geometry::POINT:
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SEGMENT:
|
||||
{
|
||||
return Npts -1;
|
||||
}
|
||||
case Geometry::TRIANGLE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+2)/2;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SQUARE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1);
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::CUBE:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1)*(n+1);
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::TETRAHEDRON:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+3)*(n+2)*(n+1)/6;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::PRISM:
|
||||
{
|
||||
for (int n = 0, np = 0; (n < 15) && (np < Npts) ; n++)
|
||||
{
|
||||
np = (n+1)*(n+1)*(n+2)/2;
|
||||
if (np == Npts) { return n; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
default:
|
||||
{
|
||||
mfem_error("Non existing Geometry.");
|
||||
}
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
int GeometryRefiner::GetRefinementLevelFromElems(Geometry::Type geom, int Nels)
|
||||
{
|
||||
switch (geom)
|
||||
{
|
||||
case Geometry::POINT:
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
case Geometry::SEGMENT:
|
||||
{
|
||||
return Nels;
|
||||
}
|
||||
case Geometry::TRIANGLE:
|
||||
case Geometry::SQUARE:
|
||||
{
|
||||
for (int n = 0; (n < 15) && (n*n < Nels+1) ; n++)
|
||||
{
|
||||
if (n*n == Nels) { return n-1; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
case Geometry::CUBE:
|
||||
case Geometry::TETRAHEDRON:
|
||||
case Geometry::PRISM:
|
||||
{
|
||||
for (int n = 0; (n < 15) && (n*n*n < Nels+1) ; n++)
|
||||
{
|
||||
if (n*n*n == Nels) { return n-1; }
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
default:
|
||||
{
|
||||
mfem_error("Non existing Geometry.");
|
||||
}
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
GeometryRefiner GlobGeometryRefiner;
|
||||
|
||||
}
|
||||
|
||||
@@ -273,12 +273,6 @@ public:
|
||||
/// @note This method always uses Quadrature1D::OpenUniform points.
|
||||
const IntegrationRule *RefineInterior(Geometry::Type Geom, int Times);
|
||||
|
||||
/// Get the Refinement level based on number of points
|
||||
virtual int GetRefinementLevelFromPoints(Geometry::Type Geom, int Npts);
|
||||
|
||||
/// Get the Refinement level based on number of elements
|
||||
virtual int GetRefinementLevelFromElems(Geometry::Type geom, int Npts);
|
||||
|
||||
~GeometryRefiner();
|
||||
};
|
||||
|
||||
|
||||
+2
-2
@@ -2984,7 +2984,7 @@ double GridFunction::ComputeLpError(const double p, Coefficient &exsol,
|
||||
}
|
||||
else
|
||||
{
|
||||
int intorder = 2*fe->GetOrder() + 3; // <----------
|
||||
int intorder = 2*fe->GetOrder() + 1; // <----------
|
||||
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
|
||||
}
|
||||
GetValues(i, *ir, vals);
|
||||
@@ -3117,7 +3117,7 @@ double GridFunction::ComputeLpError(const double p, VectorCoefficient &exsol,
|
||||
}
|
||||
else
|
||||
{
|
||||
int intorder = 2*fe->GetOrder() + 3; // <----------
|
||||
int intorder = 2*fe->GetOrder() + 1; // <----------
|
||||
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
|
||||
}
|
||||
T = fes->GetElementTransformation(i);
|
||||
|
||||
+2
-2
@@ -105,7 +105,7 @@ public:
|
||||
have the same size.
|
||||
|
||||
@note Defining this method overwrites the implicitly defined copy
|
||||
assignment operator. */
|
||||
assignemnt operator. */
|
||||
GridFunction &operator=(const GridFunction &rhs)
|
||||
{ return operator=((const Vector &)rhs); }
|
||||
|
||||
@@ -714,7 +714,7 @@ public:
|
||||
the same size.
|
||||
|
||||
@note Defining this method overwrites the implicitly defined copy
|
||||
assignment operator. */
|
||||
assignemnt operator. */
|
||||
QuadratureFunction &operator=(const QuadratureFunction &v);
|
||||
|
||||
/// Get the IntegrationRule associated with mesh element @a idx.
|
||||
|
||||
@@ -618,26 +618,6 @@ void QuadratureFunctions1D::OpenHalfUniform(const int np, IntegrationRule* ir)
|
||||
CalculateUniformWeights(ir, Quadrature1D::OpenHalfUniform);
|
||||
}
|
||||
|
||||
void QuadratureFunctions1D::ClosedGL(const int np, IntegrationRule* ir)
|
||||
{
|
||||
ir->SetSize(np);
|
||||
ir->IntPoint(0).x = 0.0;
|
||||
ir->IntPoint(np-1).x = 1.0;
|
||||
|
||||
if ( np > 2 )
|
||||
{
|
||||
IntegrationRule gl_ir;
|
||||
GaussLegendre(np-1, &gl_ir);
|
||||
|
||||
for (int i = 1; i < np-1; ++i)
|
||||
{
|
||||
ir->IntPoint(i).x = (gl_ir.IntPoint(i-1).x + gl_ir.IntPoint(i).x)/2;
|
||||
}
|
||||
}
|
||||
|
||||
CalculateUniformWeights(ir, Quadrature1D::ClosedGL);
|
||||
}
|
||||
|
||||
void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
|
||||
const int type)
|
||||
{
|
||||
@@ -670,11 +650,6 @@ void QuadratureFunctions1D::GivePolyPoints(const int np, double *pts,
|
||||
OpenHalfUniform(np, &ir);
|
||||
break;
|
||||
}
|
||||
case Quadrature1D::ClosedGL:
|
||||
{
|
||||
ClosedGL(np, &ir);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
{
|
||||
MFEM_ABORT("Asking for an unknown type of 1D Quadrature points, "
|
||||
|
||||
+1
-3
@@ -272,7 +272,6 @@ public:
|
||||
void OpenUniform(const int np, IntegrationRule *ir);
|
||||
void ClosedUniform(const int np, IntegrationRule *ir);
|
||||
void OpenHalfUniform(const int np, IntegrationRule *ir);
|
||||
void ClosedGL(const int np, IntegrationRule *ir);
|
||||
///@}
|
||||
|
||||
/// A helper function that will play nice with Poly_1D::OpenPoints and
|
||||
@@ -294,8 +293,7 @@ public:
|
||||
GaussLobatto = 1,
|
||||
OpenUniform = 2, ///< aka open Newton-Cotes
|
||||
ClosedUniform = 3, ///< aka closed Newton-Cotes
|
||||
OpenHalfUniform = 4, ///< aka "open half" Newton-Cotes
|
||||
ClosedGL = 5 ///< aka closed Gauss Legendre
|
||||
OpenHalfUniform = 4 ///< aka "open half" Newton-Cotes
|
||||
};
|
||||
/** @brief If the Quadrature1D type is not closed return Invalid; otherwise
|
||||
return type. */
|
||||
|
||||
-1465
File diff suppressed because it is too large
Load Diff
@@ -415,59 +415,6 @@ void CeedPAAssemble(const CeedPAOperator& op,
|
||||
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
|
||||
}
|
||||
|
||||
void CeedAddMultPA(const CeedData *ceedDataPtr,
|
||||
const Vector &x,
|
||||
Vector &y)
|
||||
{
|
||||
const CeedScalar *x_ptr;
|
||||
CeedScalar *y_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
x_ptr = x.Read();
|
||||
y_ptr = y.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
x_ptr = x.HostRead();
|
||||
y_ptr = y.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
|
||||
const_cast<CeedScalar*>(x_ptr));
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
|
||||
|
||||
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorTakeArray(ceedDataPtr->u, mem, const_cast<CeedScalar**>(&x_ptr));
|
||||
CeedVectorTakeArray(ceedDataPtr->v, mem, &y_ptr);
|
||||
}
|
||||
|
||||
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
|
||||
Vector &diag)
|
||||
{
|
||||
CeedScalar *d_ptr;
|
||||
CeedMemType mem;
|
||||
CeedGetPreferredMemType(internal::ceed, &mem);
|
||||
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
|
||||
{
|
||||
d_ptr = diag.ReadWrite();
|
||||
}
|
||||
else
|
||||
{
|
||||
d_ptr = diag.HostReadWrite();
|
||||
mem = CEED_MEM_HOST;
|
||||
}
|
||||
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, d_ptr);
|
||||
|
||||
CeedOperatorLinearAssembleAddDiagonal(ceedDataPtr->oper, ceedDataPtr->v,
|
||||
CEED_REQUEST_IMMEDIATE);
|
||||
|
||||
CeedVectorTakeArray(ceedDataPtr->v, mem, &d_ptr);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_CEED
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
|
||||
#ifdef MFEM_USE_CEED
|
||||
#include "../../general/device.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include <ceed.h>
|
||||
|
||||
namespace mfem
|
||||
@@ -145,15 +144,6 @@ const std::string &GetCeedPath();
|
||||
void CeedPAAssemble(const CeedPAOperator& op,
|
||||
CeedData& ceedData);
|
||||
|
||||
/** @brief Function that applies a libCEED PA operator. */
|
||||
void CeedAddMultPA(const CeedData *ceedDataPtr,
|
||||
const Vector &x,
|
||||
Vector &y);
|
||||
|
||||
/** @brief Function that assembles a libCEED PA operator diagonal. */
|
||||
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
|
||||
Vector &diag);
|
||||
|
||||
/** @brief Function that determines if a CEED kernel should be used, based on
|
||||
the current mfem::Device configuration. */
|
||||
inline bool DeviceCanUseCeed()
|
||||
|
||||
+1
-2
@@ -199,8 +199,7 @@ void LinearForm::Assemble()
|
||||
void LinearForm::Update(FiniteElementSpace *f, Vector &v, int v_offset)
|
||||
{
|
||||
fes = f;
|
||||
NewMemoryAndSize(Memory<double>(v.GetMemory(), v_offset, f->GetVSize()),
|
||||
f->GetVSize(), false);
|
||||
NewDataAndSize((double *)v + v_offset, fes->GetVSize());
|
||||
ResetDeltaLocations();
|
||||
}
|
||||
|
||||
|
||||
+4
-11
@@ -135,18 +135,11 @@ void Multigrid::SetOperator(const Operator& op)
|
||||
MFEM_ABORT("SetOperator not supported in Multigrid");
|
||||
}
|
||||
|
||||
void Multigrid::SmoothingStep(int level, bool transpose) const
|
||||
void Multigrid::SmoothingStep(int level) const
|
||||
{
|
||||
GetOperatorAtLevel(level)->Mult(*Y[level], *R[level]); // r = A x
|
||||
subtract(*X[level], *R[level], *R[level]); // r = b - A x
|
||||
if (transpose)
|
||||
{
|
||||
GetSmootherAtLevel(level)->MultTranspose(*R[level], *Z[level]); // z = S r
|
||||
}
|
||||
else
|
||||
{
|
||||
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
|
||||
}
|
||||
GetSmootherAtLevel(level)->Mult(*R[level], *Z[level]); // z = S r
|
||||
add(*Y[level], 1.0, *Z[level], *Y[level]); // x = x + S (b - A x)
|
||||
}
|
||||
|
||||
@@ -160,7 +153,7 @@ void Multigrid::Cycle(int level) const
|
||||
|
||||
for (int i = 0; i < preSmoothingSteps; i++)
|
||||
{
|
||||
SmoothingStep(level, false);
|
||||
SmoothingStep(level);
|
||||
}
|
||||
|
||||
// Compute residual
|
||||
@@ -194,7 +187,7 @@ void Multigrid::Cycle(int level) const
|
||||
// Post-smooth
|
||||
for (int i = 0; i < postSmoothingSteps; i++)
|
||||
{
|
||||
SmoothingStep(level, true);
|
||||
SmoothingStep(level);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -108,7 +108,7 @@ public:
|
||||
|
||||
private:
|
||||
/// Application of a smoothing step at particular level
|
||||
void SmoothingStep(int level, bool transpose) const;
|
||||
void SmoothingStep(int level) const;
|
||||
|
||||
/// Application of a cycle at particular level
|
||||
void Cycle(int level) const;
|
||||
|
||||
+14
-63
@@ -10,7 +10,6 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -28,7 +27,7 @@ void NonlinearForm::SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
// This is the default behavior.
|
||||
break;
|
||||
case AssemblyLevel::PARTIAL:
|
||||
ext = new PANonlinearForm(this);
|
||||
ext = new PANonlinearFormExtension(this);
|
||||
break;
|
||||
default:
|
||||
mfem_error("Unknown assembly level for this form.");
|
||||
@@ -81,13 +80,6 @@ void NonlinearForm::SetEssentialVDofs(const Array<int> &ess_vdofs_list)
|
||||
|
||||
double NonlinearForm::GetGridFunctionEnergy(const Vector &x) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
MFEM_VERIFY(!fnfi.Size(), "Interior faces terms not yet implemented!");
|
||||
MFEM_VERIFY(!bfnfi.Size(), "Boundary face terms not yet implemented!");
|
||||
return ext->GetGridFunctionEnergy(x);
|
||||
}
|
||||
|
||||
Array<int> vdofs;
|
||||
Vector el_x;
|
||||
const FiniteElement *fe;
|
||||
@@ -146,14 +138,6 @@ void NonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
if (ext)
|
||||
{
|
||||
ext->Mult(px, py);
|
||||
if (Serial())
|
||||
{
|
||||
if (cP) { cP->MultTranspose(py, y); }
|
||||
const int N = ess_tdof_list.Size();
|
||||
const auto tdof = ess_tdof_list.Read();
|
||||
auto Y = y.ReadWrite();
|
||||
MFEM_FORALL(i, N, Y[tdof[i]] = 0.0; );
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -280,16 +264,7 @@ Operator &NonlinearForm::GetGradient(const Vector &x) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
Operator &grad = ext->GetGradient(Prolongate(x));
|
||||
hGrad.Reset(&grad, false);
|
||||
if (Serial())
|
||||
{
|
||||
Operator *Gop;
|
||||
if (cP) { hGrad.Reset(new RAPOperator(*cP, grad, *cP)); }
|
||||
hGrad.Ptr()->Operator::FormSystemOperator(ess_tdof_list, Gop);
|
||||
hGrad.Reset(Gop);
|
||||
}
|
||||
return *hGrad.Ptr();
|
||||
MFEM_ABORT("Not yet implemented!");
|
||||
}
|
||||
|
||||
const int skip_zeros = 0;
|
||||
@@ -451,31 +426,7 @@ void NonlinearForm::Update()
|
||||
|
||||
void NonlinearForm::Setup()
|
||||
{
|
||||
if (ext) { return ext->Setup(); }
|
||||
}
|
||||
|
||||
void NonlinearForm::AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
|
||||
"Vector for holding diagonal has wrong size!");
|
||||
const Operator *P = fes->GetProlongationMatrix();
|
||||
if (!IsIdentityProlongation(P))
|
||||
{
|
||||
Vector local_diag(P->Height());
|
||||
ext->AssembleGradientDiagonal(local_diag);
|
||||
P->MultTranspose(local_diag, diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
ext->AssembleGradientDiagonal(diag);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Not implemented. Can be obtained through GetGradient().");
|
||||
}
|
||||
if (ext) { return ext->AssemblePA(); }
|
||||
}
|
||||
|
||||
NonlinearForm::~NonlinearForm()
|
||||
@@ -982,17 +933,6 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
|
||||
}
|
||||
}
|
||||
|
||||
if (!Grads(0,0)->Finalized())
|
||||
{
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
{
|
||||
Grads(i,j)->Finalize(skip_zeros);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int s=0; s<fes.Size(); ++s)
|
||||
{
|
||||
for (int i = 0; i < ess_vdofs[s]->Size(); ++i)
|
||||
@@ -1012,6 +952,17 @@ Operator &BlockNonlinearForm::GetGradientBlocked(const BlockVector &bx) const
|
||||
}
|
||||
}
|
||||
|
||||
if (!Grads(0,0)->Finalized())
|
||||
{
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
{
|
||||
Grads(i,j)->Finalize(skip_zeros);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int i=0; i<fes.Size(); ++i)
|
||||
{
|
||||
for (int j=0; j<fes.Size(); ++j)
|
||||
|
||||
@@ -45,7 +45,6 @@ protected:
|
||||
Array<Array<int>*> bfnfi_marker; // not owned
|
||||
|
||||
mutable SparseMatrix *Grad, *cGrad; // owned
|
||||
mutable OperatorHandle hGrad;
|
||||
|
||||
/// A list of all essential true dofs
|
||||
Array<int> ess_tdof_list;
|
||||
@@ -166,15 +165,6 @@ public:
|
||||
/// Setup the NonlinearForm
|
||||
virtual void Setup();
|
||||
|
||||
/** @brief Assemble the diagonal of the gradient into diag
|
||||
|
||||
For adaptively refined meshes, this returns P^T d_e, where d_e is the
|
||||
locally assembled diagonal on each element and P^T is the transpose of
|
||||
the conforming prolongation. In general this is not the correct diagonal
|
||||
for an AMR mesh. */
|
||||
void AssembleGradientDiagonal(Vector &diag) const;
|
||||
|
||||
|
||||
/// Get the finite element space prolongation matrix
|
||||
virtual const Operator *GetProlongation() const { return P; }
|
||||
/// Get the finite element space restriction matrix
|
||||
|
||||
+38
-77
@@ -13,101 +13,62 @@
|
||||
// PABilinearFormExtension and MFBilinearFormExtension.
|
||||
|
||||
#include "nonlinearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
NonlinearFormExtension::NonlinearFormExtension(const NonlinearForm *nlf)
|
||||
: Operator(nlf->FESpace()->GetTrueVSize()), nlf(nlf) { }
|
||||
|
||||
PANonlinearForm::PANonlinearForm(NonlinearForm *nlf):
|
||||
NonlinearFormExtension(nlf),
|
||||
x_grad(NULL),
|
||||
fes(*nlf->FESpace()),
|
||||
dnfi(*nlf->GetDNFI()),
|
||||
R(fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC))
|
||||
NonlinearFormExtension::NonlinearFormExtension(NonlinearForm *form)
|
||||
: Operator(form->FESpace()->GetTrueVSize()), n(form)
|
||||
{
|
||||
MFEM_VERIFY(R, "Not yet implemented!");
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
ye.SetSize(R->Height(), Device::GetMemoryType());
|
||||
ye.UseDevice(true);
|
||||
// empty
|
||||
}
|
||||
|
||||
double PANonlinearForm::GetGridFunctionEnergy(const Vector &x) const
|
||||
PANonlinearFormExtension::PANonlinearFormExtension(NonlinearForm *form):
|
||||
NonlinearFormExtension(form), fes(*form->FESpace())
|
||||
{
|
||||
double energy = 0.0;
|
||||
|
||||
R->Mult(x, xe);
|
||||
for (int i = 0; i < dnfi.Size(); i++)
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
elem_restrict_lex = fes.GetElementRestriction(ordering);
|
||||
if (elem_restrict_lex)
|
||||
{
|
||||
energy += dnfi[i]->GetGridFunctionEnergyPA(xe);
|
||||
localX.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
|
||||
localY.SetSize(elem_restrict_lex->Height(), Device::GetMemoryType());
|
||||
localY.UseDevice(true); // ensure 'localY = 0.0' is done on device
|
||||
}
|
||||
return energy;
|
||||
}
|
||||
|
||||
void PANonlinearForm::Setup()
|
||||
void PANonlinearFormExtension::AssemblePA()
|
||||
{
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(fes); }
|
||||
}
|
||||
|
||||
void PANonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
ye = 0.0;
|
||||
R->Mult(x, xe);
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultPA(xe, ye); }
|
||||
R->MultTranspose(ye, y);
|
||||
}
|
||||
|
||||
void PANonlinearForm::AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
MFEM_VERIFY(x_grad, "GetGradient() has not been called");
|
||||
R->Mult(*x_grad, xe);
|
||||
|
||||
ye = 0.0;
|
||||
for (int i = 0; i < dnfi.Size(); ++i)
|
||||
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
|
||||
const int Ni = integrators.Size();
|
||||
for (int i = 0; i < Ni; ++i)
|
||||
{
|
||||
dnfi[i]->AssembleGradientDiagonalPA(xe, ye);
|
||||
integrators[i]->AssemblePA(*n->FESpace());
|
||||
}
|
||||
R->MultTranspose(ye, diag);
|
||||
}
|
||||
|
||||
Operator &PANonlinearForm::GetGradient(const Vector &x) const
|
||||
void PANonlinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
// Store the last x that was used to compute the gradient.
|
||||
x_grad = &x;
|
||||
|
||||
Grad.Reset(new PANonlinearForm::Gradient(x, *this));
|
||||
return *Grad.Ptr();
|
||||
Array<NonlinearFormIntegrator*> &integrators = *n->GetDNFI();
|
||||
const int iSz = integrators.Size();
|
||||
if (elem_restrict_lex)
|
||||
{
|
||||
elem_restrict_lex->Mult(x, localX);
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AddMultPA(localX, localY);
|
||||
}
|
||||
elem_restrict_lex->MultTranspose(localY, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
y.UseDevice(true); // typically this is a large vector, so store on device
|
||||
y = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AddMultPA(x, y);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
PANonlinearForm::Gradient::Gradient(const Vector &x, const PANonlinearForm &e):
|
||||
Operator(e.fes.GetVSize()), R(e.R), dnfi(e.dnfi)
|
||||
{
|
||||
ge.UseDevice(true);
|
||||
ge.SetSize(R->Height(), Device::GetMemoryType());
|
||||
R->Mult(x, ge);
|
||||
|
||||
xe.UseDevice(true);
|
||||
xe.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
ye.UseDevice(true);
|
||||
ye.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
ze.UseDevice(true);
|
||||
ze.SetSize(R->Height(), Device::GetMemoryType());
|
||||
|
||||
// Do we still need to do this?
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AssemblePA(e.fes); }
|
||||
}
|
||||
|
||||
void PANonlinearForm::Gradient::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
ze = x;
|
||||
ye = 0.0;
|
||||
R->Mult(ze, xe);
|
||||
for (int i = 0; i < dnfi.Size(); ++i) { dnfi[i]->AddMultGradPA(ge, xe, ye); }
|
||||
R->MultTranspose(ye, y);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -17,60 +17,28 @@
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class NonlinearForm;
|
||||
class NonlinearFormIntegrator;
|
||||
|
||||
/** @brief Class extending the NonlinearForm class to support the different
|
||||
AssemblyLevel%s. */
|
||||
class NonlinearFormExtension : public Operator
|
||||
{
|
||||
protected:
|
||||
const NonlinearForm *nlf;
|
||||
NonlinearForm *n; ///< Not owned
|
||||
public:
|
||||
NonlinearFormExtension(const NonlinearForm*);
|
||||
virtual void Setup() = 0;
|
||||
virtual Operator &GetGradient(const Vector&) const = 0;
|
||||
virtual double GetGridFunctionEnergy(const Vector &x) const = 0;
|
||||
virtual void AssembleGradientDiagonal(Vector &diag) const
|
||||
{
|
||||
MFEM_ABORT("Not implemented for this assembly level!");
|
||||
}
|
||||
NonlinearFormExtension(NonlinearForm *form);
|
||||
virtual void AssemblePA() = 0;
|
||||
};
|
||||
|
||||
class PANonlinearForm;
|
||||
|
||||
|
||||
/// Data and methods for partially-assembled nonlinear forms
|
||||
class PANonlinearForm : public NonlinearFormExtension
|
||||
class PANonlinearFormExtension : public NonlinearFormExtension
|
||||
{
|
||||
private:
|
||||
class Gradient : public Operator
|
||||
{
|
||||
protected:
|
||||
const Operator *R;
|
||||
mutable Vector ge, xe, ye, ze;
|
||||
const Array<NonlinearFormIntegrator*> &dnfi;
|
||||
public:
|
||||
Gradient(const Vector &x, const PANonlinearForm &ext);
|
||||
virtual void Mult(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
protected:
|
||||
mutable Vector xe, ye;
|
||||
mutable const Vector *x_grad;
|
||||
mutable OperatorHandle Grad;
|
||||
const FiniteElementSpace &fes;
|
||||
const Array<NonlinearFormIntegrator*> &dnfi;
|
||||
const Operator *R;
|
||||
|
||||
const FiniteElementSpace &fes; // Not owned
|
||||
mutable Vector localX, localY;
|
||||
const Operator *elem_restrict_lex; // Not owned
|
||||
public:
|
||||
PANonlinearForm(NonlinearForm *nlf);
|
||||
void Setup();
|
||||
PANonlinearFormExtension(NonlinearForm*);
|
||||
void AssemblePA();
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
Operator &GetGradient(const Vector &x) const;
|
||||
double GetGridFunctionEnergy(const Vector &x) const;
|
||||
void AssembleGradientDiagonal(Vector &diag) const;
|
||||
};
|
||||
}
|
||||
#endif // NONLINEARFORM_EXT_HPP
|
||||
|
||||
@@ -15,13 +15,6 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
double NonlinearFormIntegrator::GetGridFunctionEnergyPA(const Vector &x) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::GetGridFunctionEnergyPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AssemblePA(...)\n"
|
||||
@@ -41,20 +34,6 @@ void NonlinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AddMultGradPA(const Vector&,
|
||||
const Vector&, Vector&) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AddMultGradPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssembleGradientDiagonalPA(const mfem::Vector &x,
|
||||
mfem::Vector &diag) const
|
||||
{
|
||||
mfem_error ("NonlinearFormIntegrator::AssembleDiagonalPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void NonlinearFormIntegrator::AssembleElementVector(
|
||||
const FiniteElement &el, ElementTransformation &Tr,
|
||||
const Vector &elfun, Vector &elvect)
|
||||
|
||||
@@ -68,9 +68,6 @@ public:
|
||||
ElementTransformation &Tr,
|
||||
const Vector &elfun);
|
||||
|
||||
/// Compute the local energy with partial assembly.
|
||||
virtual double GetGridFunctionEnergyPA(const Vector &x) const;
|
||||
|
||||
/// Method defining partial assembly.
|
||||
/** The result of the partial assembly is stored internally so that it can be
|
||||
used later in the methods AddMultPA(). */
|
||||
@@ -91,12 +88,6 @@ public:
|
||||
called. */
|
||||
virtual void AddMultPA(const Vector &x, Vector &y) const;
|
||||
|
||||
/// Method for partially assembled gradient action.
|
||||
virtual void AddMultGradPA(const Vector &g,
|
||||
const Vector &x, Vector &y) const;
|
||||
|
||||
virtual void AssembleGradientDiagonalPA(const Vector &x, Vector &diag) const;
|
||||
|
||||
virtual ~NonlinearFormIntegrator() { }
|
||||
};
|
||||
|
||||
|
||||
@@ -241,7 +241,7 @@ void ParBilinearForm::Assemble(int skip_zeros)
|
||||
|
||||
BilinearForm::Assemble(skip_zeros);
|
||||
|
||||
if (!ext && fbfi.Size() > 0)
|
||||
if (fbfi.Size() > 0)
|
||||
{
|
||||
AssembleSharedFaces(skip_zeros);
|
||||
}
|
||||
|
||||
+2
-2
@@ -3147,7 +3147,7 @@ static void SetSubVector(const int N,
|
||||
const Array<int> &indices,
|
||||
const Vector &in, Vector &out)
|
||||
{
|
||||
auto y = out.ReadWrite();
|
||||
auto y = out.Write();
|
||||
const auto x = in.Read();
|
||||
const auto I = indices.Read();
|
||||
MFEM_FORALL(i, N, y[I[i]] = x[i];);
|
||||
@@ -3234,7 +3234,7 @@ static void AddSubVector(const int num_unique_dst_indices,
|
||||
const Vector &src,
|
||||
Vector &dst)
|
||||
{
|
||||
auto y = dst.ReadWrite();
|
||||
auto y = dst.Write();
|
||||
const auto x = src.Read();
|
||||
const auto DST_I = unique_dst_indices.Read();
|
||||
const auto SRC_O = unique_to_src_offsets.Read();
|
||||
|
||||
@@ -711,9 +711,7 @@ void ParGridFunction::SaveAsOne(std::ostream &out)
|
||||
int *nfdofs = new int[NRanks];
|
||||
int *nrdofs = new int[NRanks];
|
||||
|
||||
HostReadWrite();
|
||||
values[0] = data;
|
||||
|
||||
nv[0] = pfes -> GetVSize();
|
||||
nvdofs[0] = pfes -> GetNVDofs();
|
||||
nedofs[0] = pfes -> GetNEDofs();
|
||||
|
||||
+9
-16
@@ -14,7 +14,6 @@
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
#include "fem.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -50,7 +49,6 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
|
||||
if (fnfi.Size())
|
||||
{
|
||||
MFEM_VERIFY(!NonlinearForm::ext,"");
|
||||
// Terms over shared interior faces in parallel.
|
||||
ParFiniteElementSpace *pfes = ParFESpace();
|
||||
ParMesh *pmesh = pfes->GetParMesh();
|
||||
@@ -88,16 +86,15 @@ void ParNonlinearForm::Mult(const Vector &x, Vector &y) const
|
||||
|
||||
P->MultTranspose(aux2, y);
|
||||
|
||||
const int N = ess_tdof_list.Size();
|
||||
const auto idx = ess_tdof_list.Read();
|
||||
auto Y = y.ReadWrite();
|
||||
MFEM_FORALL(i, N, Y[idx[i]] = 0.0; );
|
||||
y.HostReadWrite();
|
||||
for (int i = 0; i < ess_tdof_list.Size(); i++)
|
||||
{
|
||||
y(ess_tdof_list[i]) = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
const SparseMatrix &ParNonlinearForm::GetLocalGradient(const Vector &x) const
|
||||
{
|
||||
if (NonlinearForm::ext) { MFEM_ABORT("Not yet implemented!"); }
|
||||
|
||||
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
|
||||
return *Grad;
|
||||
@@ -107,20 +104,16 @@ Operator &ParNonlinearForm::GetGradient(const Vector &x) const
|
||||
{
|
||||
ParFiniteElementSpace *pfes = ParFESpace();
|
||||
|
||||
Operator &grad = NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
|
||||
pGrad.Clear();
|
||||
|
||||
NonlinearForm::GetGradient(x); // (re)assemble Grad, no b.c.
|
||||
|
||||
OperatorHandle dA(pGrad.Type()), Ph(pGrad.Type());
|
||||
|
||||
if (fnfi.Size() == 0)
|
||||
{
|
||||
if (NonlinearForm::ext) { dA.Reset(&grad, false); }
|
||||
else
|
||||
{
|
||||
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
|
||||
pfes->GetDofOffsets(), Grad);
|
||||
}
|
||||
dA.MakeSquareBlockDiag(pfes->GetComm(), pfes->GlobalVSize(),
|
||||
pfes->GetDofOffsets(), Grad);
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
+57
-136
@@ -27,24 +27,35 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
ElementDofOrdering e_ordering,
|
||||
FaceType type,
|
||||
L2FaceValues m)
|
||||
: L2FaceRestriction(fes, type, m)
|
||||
: fes(fes),
|
||||
nf(fes.GetNFbyType(type)),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndofs(fes.GetNDofs()),
|
||||
dof(nf>0 ?
|
||||
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
|
||||
: 0),
|
||||
m(m),
|
||||
nfdofs(nf*dof),
|
||||
scatter_indices1(nf*dof),
|
||||
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
|
||||
offsets(ndofs+1),
|
||||
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
|
||||
{
|
||||
if (nf==0) { return; }
|
||||
// If fespace == L2
|
||||
const ParFiniteElementSpace &pfes =
|
||||
static_cast<const ParFiniteElementSpace&>(this->fes);
|
||||
const FiniteElement *fe = pfes.GetFE(0);
|
||||
const FiniteElement *fe = fes.GetFE(0);
|
||||
const TensorBasisElement *tfe = dynamic_cast<const TensorBasisElement*>(fe);
|
||||
MFEM_VERIFY(tfe != NULL &&
|
||||
(tfe->GetBasisType()==BasisType::GaussLobatto ||
|
||||
tfe->GetBasisType()==BasisType::Positive),
|
||||
"Only Gauss-Lobatto and Bernstein basis are supported in "
|
||||
"ParL2FaceRestriction.");
|
||||
MFEM_VERIFY(pfes.GetMesh()->Conforming(),
|
||||
MFEM_VERIFY(fes.GetMesh()->Conforming(),
|
||||
"Non-conforming meshes not yet supported with partial assembly.");
|
||||
// Assuming all finite elements are using Gauss-Lobatto dofs
|
||||
height = (m==L2FaceValues::DoubleValued? 2 : 1)*vdim*nf*dof;
|
||||
width = pfes.GetVSize();
|
||||
width = fes.GetVSize();
|
||||
const bool dof_reorder = (e_ordering == ElementDofOrdering::LEXICOGRAPHIC);
|
||||
if (!dof_reorder)
|
||||
{
|
||||
@@ -52,32 +63,32 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
}
|
||||
if (dof_reorder && nf > 0)
|
||||
{
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
{
|
||||
const FiniteElement *fe =
|
||||
pfes.GetTraceElement(f, pfes.GetMesh()->GetFaceBaseGeometry(f));
|
||||
fes.GetTraceElement(f, fes.GetMesh()->GetFaceBaseGeometry(f));
|
||||
const TensorBasisElement* el =
|
||||
dynamic_cast<const TensorBasisElement*>(fe);
|
||||
if (el) { continue; }
|
||||
mfem_error("Finite element not suitable for lexicographic ordering");
|
||||
}
|
||||
}
|
||||
const Table& e2dTable = pfes.GetElementToDofTable();
|
||||
const Table& e2dTable = fes.GetElementToDofTable();
|
||||
const int* elementMap = e2dTable.GetJ();
|
||||
Array<int> faceMap1(dof), faceMap2(dof);
|
||||
int e1, e2;
|
||||
int inf1, inf2;
|
||||
int face_id1, face_id2;
|
||||
int orientation;
|
||||
const int dof1d = pfes.GetFE(0)->GetOrder()+1;
|
||||
const int elem_dofs = pfes.GetFE(0)->GetDof();
|
||||
const int dim = pfes.GetMesh()->SpaceDimension();
|
||||
const int dof1d = fes.GetFE(0)->GetOrder()+1;
|
||||
const int elem_dofs = fes.GetFE(0)->GetDof();
|
||||
const int dim = fes.GetMesh()->SpaceDimension();
|
||||
// Computation of scatter indices
|
||||
int f_ind=0;
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
{
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if (dof_reorder)
|
||||
{
|
||||
orientation = inf1 % 64;
|
||||
@@ -125,7 +136,7 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
{
|
||||
const int se2 = -1 - e2;
|
||||
Array<int> sharedDofs;
|
||||
pfes.GetFaceNbrElementVDofs(se2, sharedDofs);
|
||||
fes.GetFaceNbrElementVDofs(se2, sharedDofs);
|
||||
for (int d = 0; d < dof; ++d)
|
||||
{
|
||||
const int pd = PermuteFaceL2(dim, face_id1, face_id2,
|
||||
@@ -169,10 +180,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
offsets[i] = 0;
|
||||
}
|
||||
f_ind = 0;
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
{
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
|
||||
(type==FaceType::Boundary && e2<0 && inf2<0) )
|
||||
{
|
||||
@@ -211,10 +222,10 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
f_ind = 0;
|
||||
for (int f = 0; f < pfes.GetNF(); ++f)
|
||||
for (int f = 0; f < fes.GetNF(); ++f)
|
||||
{
|
||||
pfes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
pfes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
fes.GetMesh()->GetFaceElements(f, &e1, &e2);
|
||||
fes.GetMesh()->GetFaceInfos(f, &inf1, &inf2);
|
||||
if ((type==FaceType::Interior && (e2>=0 || (e2<0 && inf2>=0))) ||
|
||||
(type==FaceType::Boundary && e2<0 && inf2<0) )
|
||||
{
|
||||
@@ -261,10 +272,8 @@ ParL2FaceRestriction::ParL2FaceRestriction(const ParFiniteElementSpace &fes,
|
||||
|
||||
void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
|
||||
{
|
||||
const ParFiniteElementSpace &pfes =
|
||||
static_cast<const ParFiniteElementSpace&>(this->fes);
|
||||
ParGridFunction x_gf;
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&pfes),
|
||||
x_gf.MakeRef(const_cast<ParFiniteElementSpace*>(&fes),
|
||||
const_cast<Vector&>(x), 0);
|
||||
x_gf.ExchangeFaceNbrData();
|
||||
|
||||
@@ -328,122 +337,34 @@ void ParL2FaceRestriction::Mult(const Vector& x, Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
|
||||
void ParL2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
|
||||
{
|
||||
int val = AtomicAdd(I[iE],dofs);
|
||||
return val;
|
||||
}
|
||||
|
||||
void ParL2FaceRestriction::FillI(SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int Ndofs = ndofs;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
auto I_face = face_mat.ReadWriteI();
|
||||
MFEM_FORALL(i, ne*elemDofs*vdim+1,
|
||||
// Assumes all elements have the same number of dofs
|
||||
const int nd = dof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
const int dofs = nfdofs;
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = gather_indices.Read();
|
||||
auto d_x = Reshape(x.Read(), nd, vd, 2, nf);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
I_face[i] = 0;
|
||||
});
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
if (iE1 < Ndofs)
|
||||
const int offset = d_offsets[i];
|
||||
const int nextOffset = d_offsets[i + 1];
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
double dofValue = 0;
|
||||
for (int j = offset; j < nextOffset; ++j)
|
||||
{
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
if (jE2 < Ndofs)
|
||||
{
|
||||
AddNnz(iE1,I,1);
|
||||
}
|
||||
else
|
||||
{
|
||||
AddNnz(iE1,I_face,1);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
if (iE2 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
if (jE1 < Ndofs)
|
||||
{
|
||||
AddNnz(iE2,I,1);
|
||||
}
|
||||
else
|
||||
{
|
||||
AddNnz(iE2,I_face,1);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void ParL2FaceRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int Ndofs = ndofs;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto I = mat.ReadWriteI();
|
||||
auto I_face = face_mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto J_face = face_mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto Data_face = face_mat.WriteData();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
if (iE1 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
if (jE2 < Ndofs)
|
||||
{
|
||||
const int offset = AddNnz(iE1,I,1);
|
||||
J[offset] = jE2;
|
||||
Data[offset] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
else
|
||||
{
|
||||
const int offset = AddNnz(iE1,I_face,1);
|
||||
J_face[offset] = jE2-Ndofs;
|
||||
Data_face[offset] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
if (iE2 < Ndofs)
|
||||
{
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
if (jE1 < Ndofs)
|
||||
{
|
||||
const int offset = AddNnz(iE2,I,1);
|
||||
J[offset] = jE1;
|
||||
Data[offset] = mat_fea(jF,iF,0,f);
|
||||
}
|
||||
else
|
||||
{
|
||||
const int offset = AddNnz(iE2,I_face,1);
|
||||
J_face[offset] = jE1-Ndofs;
|
||||
Data_face[offset] = mat_fea(jF,iF,0,f);
|
||||
}
|
||||
int idx_j = d_indices[j];
|
||||
bool isE1 = idx_j < dofs;
|
||||
idx_j = isE1 ? idx_j : idx_j - dofs;
|
||||
dofValue += isE1 ?
|
||||
d_x(idx_j % nd, c, 0, idx_j / nd)
|
||||
:d_x(idx_j % nd, c, 1, idx_j / nd);
|
||||
}
|
||||
d_y(t?c:i,t?i:c) += dofValue;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
+16
-9
@@ -26,21 +26,28 @@ class ParFiniteElementSpace;
|
||||
/// Operator that extracts Face degrees of freedom in parallel.
|
||||
/** Objects of this type are typically created and owned by FiniteElementSpace
|
||||
objects, see FiniteElementSpace::GetFaceRestriction(). */
|
||||
class ParL2FaceRestriction : public L2FaceRestriction
|
||||
class ParL2FaceRestriction : public Operator
|
||||
{
|
||||
protected:
|
||||
const ParFiniteElementSpace &fes;
|
||||
const int nf;
|
||||
const int vdim;
|
||||
const bool byvdim;
|
||||
const int ndofs;
|
||||
const int dof;
|
||||
const L2FaceValues m;
|
||||
const int nfdofs;
|
||||
Array<int> scatter_indices1;
|
||||
Array<int> scatter_indices2;
|
||||
Array<int> offsets;
|
||||
Array<int> gather_indices;
|
||||
|
||||
public:
|
||||
ParL2FaceRestriction(const ParFiniteElementSpace&, ElementDofOrdering,
|
||||
FaceType type,
|
||||
L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ParL2FaceRestriction. */
|
||||
void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this ParL2FaceRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
+1121
-396
File diff suppressed because it is too large
Load Diff
+15
-33
@@ -41,11 +41,10 @@ protected:
|
||||
const FiniteElementSpace *fespace; ///< Not owned
|
||||
const QuadratureSpace *qspace; ///< Not owned
|
||||
const IntegrationRule *IntRule; ///< Not owned
|
||||
|
||||
mutable QVectorLayout q_layout; ///< Output Q-vector layout
|
||||
mutable bool use_tensor_products; ///< Tensor product evaluation mmode
|
||||
|
||||
public:
|
||||
mutable bool use_tensor_products;
|
||||
|
||||
static const int MAX_NQ2D = 100;
|
||||
static const int MAX_ND2D = 100;
|
||||
static const int MAX_VDIM2D = 3;
|
||||
@@ -54,6 +53,7 @@ public:
|
||||
static const int MAX_ND3D = 1000;
|
||||
static const int MAX_VDIM3D = 3;
|
||||
|
||||
public:
|
||||
enum EvalFlags
|
||||
{
|
||||
VALUES = 1 << 0, ///< Evaluate the values at quadrature points
|
||||
@@ -61,28 +61,21 @@ public:
|
||||
/** @brief Assuming the derivative at quadrature points form a matrix,
|
||||
this flag can be used to compute and store their determinants. This
|
||||
flag can only be used in Mult(). */
|
||||
DETERMINANTS = 1 << 2,
|
||||
PHYSICAL_DERIVATIVES = 1 << 3 ///< Evaluate the physical derivatives
|
||||
DETERMINANTS = 1 << 2
|
||||
};
|
||||
|
||||
QuadratureInterpolator(const FiniteElementSpace &fes,
|
||||
const IntegrationRule &ir,
|
||||
const bool use_tensor_products = false);
|
||||
const IntegrationRule &ir);
|
||||
|
||||
QuadratureInterpolator(const FiniteElementSpace &fes,
|
||||
const QuadratureSpace &qs,
|
||||
const bool use_tensor_products = false);
|
||||
const QuadratureSpace &qs);
|
||||
|
||||
/** @brief Disable the use of tensor product evaluations, for tensor-product
|
||||
elements, e.g. quads and hexes. */
|
||||
void DisableTensorProducts() const { use_tensor_products = false; }
|
||||
|
||||
/** @brief Enable the use of tensor product evaluations, for tensor-product
|
||||
elements, e.g. quads and hexes. */
|
||||
void EnableTensorProducts() const { use_tensor_products = true; }
|
||||
|
||||
/** @brief Query the current evaluation mode. */
|
||||
bool UseTensorProducts() const { return use_tensor_products; }
|
||||
/** Currently, tensor product evaluations are not implemented and this method
|
||||
has no effect. */
|
||||
void DisableTensorProducts(bool disable = true) const
|
||||
{ use_tensor_products = !disable; }
|
||||
|
||||
/** @brief Query the current output Q-vector layout. The default value is
|
||||
QVectorLayout::byNODES. */
|
||||
@@ -90,7 +83,8 @@ public:
|
||||
|
||||
/** @brief Set the desired output Q-vector layout. The default value is
|
||||
QVectorLayout::byNODES. */
|
||||
void SetOutputLayout(QVectorLayout layout) const { q_layout = layout; }
|
||||
void SetOutputLayout(QVectorLayout out_layout) const
|
||||
{ q_layout = out_layout; }
|
||||
|
||||
/// Interpolate the E-vector @a e_vec to quadrature points.
|
||||
/** The @a eval_flags are a bitwise mask of constants from the EvalFlags
|
||||
@@ -105,36 +99,26 @@ public:
|
||||
Vector &q_val, Vector &q_der, Vector &q_det) const;
|
||||
|
||||
/// Interpolate the values of the E-vector @a e_vec at quadrature points.
|
||||
template <QVectorLayout>
|
||||
void Values(const Vector &e_vec, Vector &q_val) const;
|
||||
void Values(const Vector &e_vec, Vector &q_val) const;
|
||||
|
||||
/** @brief Interpolate the derivatives of the E-vector @a e_vec at quadrature
|
||||
points. */
|
||||
template <QVectorLayout>
|
||||
void Derivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
void Derivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
|
||||
/** @brief Interpolate the derivatives in physical space of the E-vector
|
||||
@a e_vec at quadrature points. */
|
||||
template <QVectorLayout>
|
||||
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
void PhysDerivatives(const Vector &e_vec, Vector &q_der) const;
|
||||
|
||||
/// Compute the determinant of the E-vector @a e_vec at quadrature points.
|
||||
void Determinants(const Vector &e_vec, Vector &q_det) const;
|
||||
|
||||
/// Perform the transpose operation of Mult(). (TODO)
|
||||
void MultTranspose(unsigned eval_flags, const Vector &q_val,
|
||||
const Vector &q_der, Vector &e_vec) const;
|
||||
|
||||
// Compute kernels follow (cannot be private or protected with nvcc)
|
||||
|
||||
/// Template compute kernel for 2D.
|
||||
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
|
||||
static void Mult2D(const int NE,
|
||||
static void Eval2D(const int NE,
|
||||
const int vdim,
|
||||
const QVectorLayout q_layout,
|
||||
const GeometricFactors *geom,
|
||||
const DofToQuad &maps,
|
||||
const Vector &e_vec,
|
||||
Vector &q_val,
|
||||
@@ -144,10 +128,8 @@ public:
|
||||
|
||||
/// Template compute kernel for 3D.
|
||||
template<const int T_VDIM = 0, const int T_ND = 0, const int T_NQ = 0>
|
||||
static void Mult3D(const int NE,
|
||||
static void Eval3D(const int NE,
|
||||
const int vdim,
|
||||
const QVectorLayout q_layout,
|
||||
const GeometricFactors *geom,
|
||||
const DofToQuad &maps,
|
||||
const Vector &e_vec,
|
||||
Vector &q_val,
|
||||
|
||||
@@ -1,208 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop_pa.hpp"
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../fem/kernels.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Det2D(const int NE,
|
||||
const double *b,
|
||||
const double *g,
|
||||
const double *x,
|
||||
double *y,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b, Q1D, D1D);
|
||||
const auto G = Reshape(g, Q1D, D1D);
|
||||
const auto X = Reshape(x, D1D, D1D, DIM, NE);
|
||||
auto Y = Reshape(y, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double J[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,J);
|
||||
Y(qx,qy,e) = kernels::Det<2>(J);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Det3D(const int NE,
|
||||
const double *b,
|
||||
const double *g,
|
||||
const double *x,
|
||||
double *y,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b, Q1D, D1D);
|
||||
const auto G = Reshape(g, Q1D, D1D);
|
||||
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
|
||||
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double sm0[9][MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[9][MDQ*MDQ*MDQ];
|
||||
|
||||
double (*DDD)[MD1*MD1*MD1] = (double (*)[MD1*MD1*MD1]) (sm0);
|
||||
double (*DDQ)[MD1*MD1*MQ1] = (double (*)[MD1*MD1*MQ1]) (sm1);
|
||||
double (*DQQ)[MD1*MQ1*MQ1] = (double (*)[MD1*MQ1*MQ1]) (sm0);
|
||||
double (*QQQ)[MQ1*MQ1*MQ1] = (double (*)[MQ1*MQ1*MQ1]) (sm1);
|
||||
|
||||
kernels::LoadX<MD1>(e,D1D,X,DDD);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
|
||||
kernels::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
|
||||
kernels::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double J[9];
|
||||
kernels::PullGrad<MQ1>(qx,qy,qz, QQQ, J);
|
||||
Y(qx,qy,qz,e) = kernels::Det<3>(J);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void QuadratureInterpolator::Determinants(const Vector &e_vec,
|
||||
Vector &q_det) const
|
||||
{
|
||||
if (use_tensor_products)
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_det.Write();
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2222: return Det2D<2,2>(NE,B,G,X,Y);
|
||||
case 0x2223: return Det2D<2,3>(NE,B,G,X,Y);
|
||||
case 0x2224: return Det2D<2,4>(NE,B,G,X,Y);
|
||||
case 0x2226: return Det2D<2,6>(NE,B,G,X,Y);
|
||||
case 0x2234: return Det2D<3,4>(NE,B,G,X,Y);
|
||||
case 0x2236: return Det2D<3,6>(NE,B,G,X,Y);
|
||||
case 0x2244: return Det2D<4,4>(NE,B,G,X,Y);
|
||||
case 0x2246: return Det2D<4,6>(NE,B,G,X,Y);
|
||||
case 0x2256: return Det2D<5,6>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3324: return Det3D<2,4>(NE,B,G,X,Y);
|
||||
case 0x3333: return Det3D<3,3>(NE,B,G,X,Y);
|
||||
case 0x3335: return Det3D<3,5>(NE,B,G,X,Y);
|
||||
case 0x3336: return Det3D<3,6>(NE,B,G,X,Y);
|
||||
//case 0x3348: return Det3D<4,8>(NE,B,G,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
return Det2D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
constexpr int MD1 = 6;
|
||||
constexpr int MQ1 = 6;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
return Det3D<0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
MFEM_ABORT("Kernel " << std::hex << id << std::dec << " not supported yet");
|
||||
}
|
||||
else
|
||||
{
|
||||
Vector empty;
|
||||
Mult(e_vec, DETERMINANTS, empty, empty, q_det);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,233 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Eval2D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, NE):
|
||||
Reshape(y_, VDIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_DD[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> DD((double*)(s_DD+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double s_DQ[NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ((double*)(s_DQ+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
DD(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
u += B(qx,dx) * DD(dx,dy);
|
||||
}
|
||||
DQ(dy,qx) = u;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ(dy,qx) * B(qy,dy);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,e) = u; }
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,c,e) = u; }
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Eval3D(const int NE,
|
||||
const double *b_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT == QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, NE):
|
||||
Reshape(y_, VDIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MDQ = (MQ1 > MD1) ? MQ1 : MD1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[MDQ*MDQ*MDQ];
|
||||
MFEM_SHARED double sm1[MDQ*MDQ*MDQ];
|
||||
DeviceTensor<3,double> DDD(sm0, MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ(sm1, MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ(sm0, MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; c++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
DDD(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
u += B(qx,dx) * DDD(dx,dy,dz);
|
||||
}
|
||||
DDQ(dz,dy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ(dz,qy,qx) = u;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ(dz,qy,qx) * B(qz,dz);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM) { y(c,qx,qy,qz,e) = u; }
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES) { y(qx,qy,qz,c,e) = u; }
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,110 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_eval.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Values<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_val.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2133: return Eval2D<L,1,3,3>(NE,B,X,Y);
|
||||
case 0x2124: return Eval2D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x2132: return Eval2D<L,1,3,2>(NE,B,X,Y);
|
||||
case 0x2134: return Eval2D<L,1,3,4>(NE,B,X,Y);
|
||||
case 0x2143: return Eval2D<L,1,4,3>(NE,B,X,Y);
|
||||
case 0x2144: return Eval2D<L,1,4,4>(NE,B,X,Y);
|
||||
|
||||
case 0x2222: return Eval2D<L,2,2,2>(NE,B,X,Y);
|
||||
case 0x2223: return Eval2D<L,2,2,3>(NE,B,X,Y);
|
||||
case 0x2224: return Eval2D<L,2,2,4>(NE,B,X,Y);
|
||||
case 0x2225: return Eval2D<L,2,2,5>(NE,B,X,Y);
|
||||
case 0x2226: return Eval2D<L,2,2,6>(NE,B,X,Y);
|
||||
case 0x2233: return Eval2D<L,2,3,3>(NE,B,X,Y);
|
||||
case 0x2234: return Eval2D<L,2,3,4>(NE,B,X,Y);
|
||||
case 0x2236: return Eval2D<L,2,3,6>(NE,B,X,Y);
|
||||
case 0x2243: return Eval2D<L,2,4,3>(NE,B,X,Y);
|
||||
case 0x2244: return Eval2D<L,2,4,4>(NE,B,X,Y);
|
||||
case 0x2245: return Eval2D<L,2,4,5>(NE,B,X,Y);
|
||||
case 0x2246: return Eval2D<L,2,4,6>(NE,B,X,Y);
|
||||
case 0x2247: return Eval2D<L,2,4,7>(NE,B,X,Y);
|
||||
case 0x2256: return Eval2D<L,2,5,6>(NE,B,X,Y);
|
||||
|
||||
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x3133: return Eval3D<L,1,3,3>(NE,B,X,Y);
|
||||
case 0x3134: return Eval3D<L,1,3,4>(NE,B,X,Y);
|
||||
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
|
||||
case 0x3143: return Eval3D<L,1,4,3>(NE,B,X,Y);
|
||||
case 0x3144: return Eval3D<L,1,4,4>(NE,B,X,Y);
|
||||
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
|
||||
|
||||
case 0x3222: return Eval3D<L,2,2,2>(NE,B,X,Y);
|
||||
case 0x3223: return Eval3D<L,2,2,3>(NE,B,X,Y);
|
||||
case 0x3234: return Eval3D<L,2,3,4>(NE,B,X,Y);
|
||||
|
||||
case 0x3323: return Eval3D<L,3,2,3>(NE,B,X,Y);
|
||||
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
|
||||
case 0x3325: return Eval3D<L,3,2,5>(NE,B,X,Y);
|
||||
case 0x3326: return Eval3D<L,3,2,6>(NE,B,X,Y);
|
||||
case 0x3333: return Eval3D<L,3,3,3>(NE,B,X,Y);
|
||||
case 0x3334: return Eval3D<L,3,3,4>(NE,B,X,Y);
|
||||
case 0x3335: return Eval3D<L,3,3,5>(NE,B,X,Y);
|
||||
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
|
||||
case 0x3343: return Eval3D<L,3,4,3>(NE,B,X,Y);
|
||||
case 0x3344: return Eval3D<L,3,4,4>(NE,B,X,Y);
|
||||
case 0x3346: return Eval3D<L,3,4,6>(NE,B,X,Y);
|
||||
case 0x3347: return Eval3D<L,3,4,7>(NE,B,X,Y);
|
||||
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,79 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_eval.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Values<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_val.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2124: return Eval2D<L,1,2,4,8>(NE,B,X,Y);
|
||||
case 0x2136: return Eval2D<L,1,3,6,4>(NE,B,X,Y);
|
||||
case 0x2148: return Eval2D<L,1,4,8,2>(NE,B,X,Y);
|
||||
|
||||
case 0x2224: return Eval2D<L,2,2,4,8>(NE,B,X,Y);
|
||||
case 0x2234: return Eval2D<L,2,3,4,8>(NE,B,X,Y);
|
||||
case 0x2236: return Eval2D<L,2,3,6,4>(NE,B,X,Y);
|
||||
case 0x2248: return Eval2D<L,2,4,8,2>(NE,B,X,Y);
|
||||
|
||||
case 0x3124: return Eval3D<L,1,2,4>(NE,B,X,Y);
|
||||
case 0x3136: return Eval3D<L,1,3,6>(NE,B,X,Y);
|
||||
case 0x3148: return Eval3D<L,1,4,8>(NE,B,X,Y);
|
||||
|
||||
case 0x3324: return Eval3D<L,3,2,4>(NE,B,X,Y);
|
||||
case 0x3336: return Eval3D<L,3,3,6>(NE,B,X,Y);
|
||||
case 0x3348: return Eval3D<L,3,4,8>(NE,B,X,Y);
|
||||
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Eval2D<L,0,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Eval3D<L,0,0,0,MD1,MQ1>(NE,B,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -495,9 +495,8 @@ void FaceQuadratureInterpolator::Mult(
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void FaceQuadratureInterpolator::Values(const Vector &e_vec,
|
||||
Vector &q_val) const
|
||||
void FaceQuadratureInterpolator::Values(
|
||||
const Vector &e_vec, Vector &q_val) const
|
||||
{
|
||||
Vector q_der, q_det, q_nor;
|
||||
Mult(e_vec, VALUES, q_val, q_der, q_det, q_nor);
|
||||
|
||||
@@ -1,282 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Grad2D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
|
||||
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
MFEM_SHARED double s_G[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
DeviceTensor<2,double> G(s_G, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_X[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double s_DQ[2][NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ0((double*)(s_DQ[0]+tidz), MD1, MQ1);
|
||||
DeviceTensor<2,double> DQ1((double*)(s_DQ[1]+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
X(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DQ0(dy,qx) = u;
|
||||
DQ1(dy,qx) = v;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ1(dy,qx) * B(qy,dy);
|
||||
v += DQ0(dy,qx) * G(qy,dy);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,c,0,e) = u;
|
||||
y(qx,qy,c,1,e) = v;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,e) = u;
|
||||
y(c,1,qx,qy,e) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void Grad3D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
|
||||
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1*MD1];
|
||||
MFEM_SHARED double s_G[MQ1*MD1];
|
||||
DeviceTensor<2,double> B(s_B, Q1D, D1D);
|
||||
DeviceTensor<2,double> G(s_G, Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
|
||||
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
X(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy,dz);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DDQ0(dz,dy,qx) = u;
|
||||
DDQ1(dz,dy,qx) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ1(dz,dy,qx) * B(qy,dy);
|
||||
v += DDQ0(dz,dy,qx) * G(qy,dy);
|
||||
w += DDQ0(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ0(dz,qy,qx) = u;
|
||||
DQQ1(dz,qy,qx) = v;
|
||||
DQQ2(dz,qy,qx) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ0(dz,qy,qx) * B(qz,dz);
|
||||
v += DQQ1(dz,qy,qx) * B(qz,dz);
|
||||
w += DQQ2(dz,qy,qx) * G(qz,dz);
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,qz,c,0,e) = u;
|
||||
y(qx,qy,qz,c,1,e) = v;
|
||||
y(qx,qy,qz,c,2,e) = w;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,qz,e) = u;
|
||||
y(c,1,qx,qy,qz,e) = v;
|
||||
y(c,2,qx,qy,qz,e) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,109 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Derivatives<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2133: return Grad2D<L,1,3,3,16>(NE,B,G,X,Y);
|
||||
case 0x2134: return Grad2D<L,1,3,4,16>(NE,B,G,X,Y);
|
||||
case 0x2143: return Grad2D<L,1,4,3,16>(NE,B,G,X,Y);
|
||||
case 0x2144: return Grad2D<L,1,4,4,16>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2222: return Grad2D<L,2,2,2,16>(NE,B,G,X,Y);
|
||||
case 0x2223: return Grad2D<L,2,2,3,8>(NE,B,G,X,Y);
|
||||
case 0x2224: return Grad2D<L,2,2,4,4>(NE,B,G,X,Y);
|
||||
case 0x2225: return Grad2D<L,2,2,5,4>(NE,B,G,X,Y);
|
||||
case 0x2226: return Grad2D<L,2,2,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2233: return Grad2D<L,2,3,3,2>(NE,B,G,X,Y);
|
||||
case 0x2234: return Grad2D<L,2,3,4,4>(NE,B,G,X,Y);
|
||||
case 0x2243: return Grad2D<L,2,4,3,4>(NE,B,G,X,Y);
|
||||
case 0x2236: return Grad2D<L,2,3,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2244: return Grad2D<L,2,4,4,2>(NE,B,G,X,Y);
|
||||
case 0x2245: return Grad2D<L,2,4,5,2>(NE,B,G,X,Y);
|
||||
case 0x2246: return Grad2D<L,2,4,6,2>(NE,B,G,X,Y);
|
||||
case 0x2247: return Grad2D<L,2,4,7,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2256: return Grad2D<L,2,5,6,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3124: return Grad3D<L,1,2,4>(NE,B,G,X,Y);
|
||||
case 0x3133: return Grad3D<L,1,3,3>(NE,B,G,X,Y);
|
||||
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
|
||||
case 0x3136: return Grad3D<L,1,3,6>(NE,B,G,X,Y);
|
||||
case 0x3144: return Grad3D<L,1,4,4>(NE,B,G,X,Y);
|
||||
case 0x3148: return Grad3D<L,1,4,8>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3323: return Grad3D<L,3,2,3>(NE,B,G,X,Y);
|
||||
case 0x3324: return Grad3D<L,3,2,4>(NE,B,G,X,Y);
|
||||
case 0x3325: return Grad3D<L,3,2,5>(NE,B,G,X,Y);
|
||||
case 0x3326: return Grad3D<L,3,2,6>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3333: return Grad3D<L,3,3,3>(NE,B,G,X,Y);
|
||||
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
|
||||
case 0x3335: return Grad3D<L,3,3,5>(NE,B,G,X,Y);
|
||||
case 0x3336: return Grad3D<L,3,3,6>(NE,B,G,X,Y);
|
||||
case 0x3344: return Grad3D<L,3,4,4>(NE,B,G,X,Y);
|
||||
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
|
||||
case 0x3347: return Grad3D<L,3,4,7>(NE,B,G,X,Y);
|
||||
case 0x3348: return Grad3D<L,3,4,8>(NE,B,G,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2)
|
||||
{
|
||||
return Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
return Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,76 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::Derivatives<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (dim<<12) | (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
switch (id)
|
||||
{
|
||||
case 0x2134: return Grad2D<L,1,3,4,8>(NE,B,G,X,Y);
|
||||
case 0x2146: return Grad2D<L,1,4,6,4>(NE,B,G,X,Y);
|
||||
case 0x2158: return Grad2D<L,1,5,8,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x2234: return Grad2D<L,2,3,4,8>(NE,B,G,X,Y);
|
||||
case 0x2246: return Grad2D<L,2,4,6,4>(NE,B,G,X,Y);
|
||||
case 0x2258: return Grad2D<L,2,5,8,2>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3134: return Grad3D<L,1,3,4>(NE,B,G,X,Y);
|
||||
case 0x3146: return Grad3D<L,1,4,6>(NE,B,G,X,Y);
|
||||
case 0x3158: return Grad3D<L,1,5,8>(NE,B,G,X,Y);
|
||||
|
||||
case 0x3334: return Grad3D<L,3,3,4>(NE,B,G,X,Y);
|
||||
case 0x3346: return Grad3D<L,3,4,6>(NE,B,G,X,Y);
|
||||
case 0x3358: return Grad3D<L,3,5,8>(NE,B,G,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD1 = 8;
|
||||
constexpr int MQ1 = 8;
|
||||
MFEM_VERIFY(D1D <= MD1, "Orders higher than " << MD1-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "Quadrature rules with more than "
|
||||
<< MQ1 << " 1D points are not supported!");
|
||||
if (dim == 2) { Grad2D<L,0,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
|
||||
if (dim == 3) { Grad3D<L,0,0,0,MD1,MQ1>(NE,B,G,X,Y,vdim,D1D,Q1D); }
|
||||
return;
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Kernel not supported yet");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,303 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int T_NBZ = 1, int MAX_D1D = 0, int MAX_Q1D = 0>
|
||||
static void PhysGrad2D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *j_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 0,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, 2, 2, NE);
|
||||
const auto x = Reshape(x_, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, VDIM, 2, NE):
|
||||
Reshape(y_, VDIM, 2, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1][MD1];
|
||||
MFEM_SHARED double s_G[MQ1][MD1];
|
||||
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
|
||||
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double s_X[NBZ][MD1*MD1];
|
||||
DeviceTensor<2,double> X((double*)(s_X+tidz), MD1, MD1);
|
||||
|
||||
MFEM_SHARED double sm[2][NBZ][MD1*MQ1];
|
||||
DeviceTensor<2,double> DQ0((double*)(sm[0]+tidz), MD1, MQ1);
|
||||
DeviceTensor<2,double> DQ1((double*)(sm[1]+tidz), MD1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
X(dx,dy) = x(dx,dy,c,e);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double input = X(dx,dy);
|
||||
u += input * B(qx,dx);
|
||||
v += input * G(qx,dx);
|
||||
}
|
||||
DQ0(dy,qx) = u;
|
||||
DQ1(dy,qx) = v;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DQ1(dy,qx) * B(qy,dy);
|
||||
v += DQ0(dy,qx) * G(qy,dy);
|
||||
}
|
||||
double Jloc[4], Jinv[4];
|
||||
Jloc[0] = j(qx,qy,0,0,e);
|
||||
Jloc[1] = j(qx,qy,1,0,e);
|
||||
Jloc[2] = j(qx,qy,0,1,e);
|
||||
Jloc[3] = j(qx,qy,1,1,e);
|
||||
kernels::CalcInverse<2>(Jloc, Jinv);
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,e) = Jinv[0]*u + Jinv[1]*v;
|
||||
y(c,1,qx,qy,e) = Jinv[2]*u + Jinv[3]*v;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,c,0,e) = Jinv[0]*u + Jinv[1]*v;
|
||||
y(qx,qy,c,1,e) = Jinv[2]*u + Jinv[3]*v;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<QVectorLayout Q_LAYOUT,
|
||||
int T_VDIM = 0, int T_D1D = 0, int T_Q1D = 0,
|
||||
int MAX_D = 0, int MAX_Q = 0>
|
||||
static void PhysGrad3D(const int NE,
|
||||
const double *b_,
|
||||
const double *g_,
|
||||
const double *j_,
|
||||
const double *x_,
|
||||
double *y_,
|
||||
const int vdim = 1,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
|
||||
const auto b = Reshape(b_, Q1D, D1D);
|
||||
const auto g = Reshape(g_, Q1D, D1D);
|
||||
const auto j = Reshape(j_, Q1D, Q1D, Q1D, 3, 3, NE);
|
||||
const auto x = Reshape(x_, D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Q_LAYOUT ==QVectorLayout:: byNODES ?
|
||||
Reshape(y_, Q1D, Q1D, Q1D, VDIM, 3, NE):
|
||||
Reshape(y_, VDIM, 3, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
const int VDIM = T_VDIM ? T_VDIM : vdim;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
MFEM_SHARED double s_B[MQ1][MD1];
|
||||
MFEM_SHARED double s_G[MQ1][MD1];
|
||||
DeviceTensor<2,double> B((double*)(s_B), Q1D, D1D);
|
||||
DeviceTensor<2,double> G((double*)(s_G), Q1D, D1D);
|
||||
|
||||
MFEM_SHARED double sm0[3][MQ1*MQ1*MQ1];
|
||||
MFEM_SHARED double sm1[3][MQ1*MQ1*MQ1];
|
||||
DeviceTensor<3,double> X((double*)(sm0+2), MD1, MD1, MD1);
|
||||
DeviceTensor<3,double> DDQ0((double*)(sm0+0), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DDQ1((double*)(sm0+1), MD1, MD1, MQ1);
|
||||
DeviceTensor<3,double> DQQ0((double*)(sm1+0), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ1((double*)(sm1+1), MD1, MQ1, MQ1);
|
||||
DeviceTensor<3,double> DQQ2((double*)(sm1+2), MD1, MQ1, MQ1);
|
||||
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B(q,d) = b(q,d);
|
||||
G(q,d) = g(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int c = 0; c < VDIM; ++c)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
X(dx,dy,dz) = x(dx,dy,dz,c,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
const double coords = X(dx,dy,dz);
|
||||
u += coords * B(qx,dx);
|
||||
v += coords * G(qx,dx);
|
||||
}
|
||||
DDQ0(dz,dy,qx) = u;
|
||||
DDQ1(dz,dy,qx) = v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
u += DDQ1(dz,dy,qx) * B(qy,dy);
|
||||
v += DDQ0(dz,dy,qx) * G(qy,dy);
|
||||
w += DDQ0(dz,dy,qx) * B(qy,dy);
|
||||
}
|
||||
DQQ0(dz,qy,qx) = u;
|
||||
DQQ1(dz,qy,qx) = v;
|
||||
DQQ2(dz,qy,qx) = w;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
double u = 0.0;
|
||||
double v = 0.0;
|
||||
double w = 0.0;
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
u += DQQ0(dz,qy,qx) * B(qz,dz);
|
||||
v += DQQ1(dz,qy,qx) * B(qz,dz);
|
||||
w += DQQ2(dz,qy,qx) * G(qz,dz);
|
||||
}
|
||||
double Jloc[9], Jinv[9];
|
||||
for (int col = 0; col < 3; col++)
|
||||
{
|
||||
for (int row = 0; row < 3; row++)
|
||||
{
|
||||
Jloc[row+3*col] = j(qx,qy,qz,row,col,e);
|
||||
}
|
||||
}
|
||||
kernels::CalcInverse<3>(Jloc, Jinv);
|
||||
if (Q_LAYOUT == QVectorLayout::byNODES)
|
||||
{
|
||||
y(qx,qy,qz,c,0,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
|
||||
y(qx,qy,qz,c,1,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
|
||||
y(qx,qy,qz,c,2,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
|
||||
}
|
||||
if (Q_LAYOUT == QVectorLayout::byVDIM)
|
||||
{
|
||||
y(c,0,qx,qy,qz,e) = Jinv[0]*u + Jinv[1]*v + Jinv[2]*w;
|
||||
y(c,1,qx,qy,qz,e) = Jinv[3]*u + Jinv[4]*v + Jinv[5]*w;
|
||||
y(c,2,qx,qy,qz,e) = Jinv[6]*u + Jinv[7]*v + Jinv[8]*w;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,110 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad_phys.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byNODES>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
Mesh *mesh = fespace->GetMesh();
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
|
||||
const GeometricFactors *geom =
|
||||
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *J = geom->J.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byNODES;
|
||||
|
||||
const int id = (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x133: return PhysGrad2D<L,1,3,3,8>(NE,B,G,J,X,Y);
|
||||
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE,B,G,J,X,Y);
|
||||
case 0x143: return PhysGrad2D<L,1,4,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x144: return PhysGrad2D<L,1,4,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE,B,G,J,X,Y);
|
||||
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE,B,G,J,X,Y);
|
||||
|
||||
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE,B,G,J,X,Y);
|
||||
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE,B,G,J,X,Y);
|
||||
case 0x243: return PhysGrad2D<L,2,4,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x244: return PhysGrad2D<L,2,4,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE,B,G,J,X,Y);
|
||||
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE,B,G,J,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = MAX_D1D;
|
||||
constexpr int MQ = MAX_Q1D;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x133: return PhysGrad3D<L,1,3,3>(NE,B,G,J,X,Y);
|
||||
case 0x134: return PhysGrad3D<L,1,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x144: return PhysGrad3D<L,1,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x146: return PhysGrad3D<L,1,4,6>(NE,B,G,J,X,Y);
|
||||
case 0x158: return PhysGrad3D<L,1,5,8>(NE,B,G,J,X,Y);
|
||||
|
||||
case 0x333: return PhysGrad3D<L,3,3,3>(NE,B,G,J,X,Y);
|
||||
case 0x334: return PhysGrad3D<L,3,3,4>(NE,B,G,J,X,Y);
|
||||
case 0x344: return PhysGrad3D<L,3,4,4>(NE,B,G,J,X,Y);
|
||||
case 0x346: return PhysGrad3D<L,3,4,6>(NE,B,G,J,X,Y);
|
||||
case 0x358: return PhysGrad3D<L,3,5,8>(NE,B,G,J,X,Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = 8;
|
||||
constexpr int MQ = 8;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,101 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "quadinterpolator_grad_phys.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template<>
|
||||
void QuadratureInterpolator::PhysDerivatives<QVectorLayout::byVDIM>(
|
||||
const Vector &e_vec, Vector &q_der) const
|
||||
{
|
||||
const int NE = fespace->GetNE();
|
||||
if (NE == 0) { return; }
|
||||
Mesh *mesh = fespace->GetMesh();
|
||||
const int vdim = fespace->GetVDim();
|
||||
const int dim = fespace->GetMesh()->Dimension();
|
||||
const FiniteElement *fe = fespace->GetFE(0);
|
||||
const IntegrationRule *ir =
|
||||
IntRule ? IntRule : &qspace->GetElementIntRule(0);
|
||||
constexpr DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
const DofToQuad &maps = fe->GetDofToQuad(*ir, mode);
|
||||
const GeometricFactors *geom =
|
||||
mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
const int D1D = maps.ndof;
|
||||
const int Q1D = maps.nqpt;
|
||||
|
||||
const double *B = maps.B.Read();
|
||||
const double *G = maps.G.Read();
|
||||
const double *J = geom->J.Read();
|
||||
const double *X = e_vec.Read();
|
||||
double *Y = q_der.Write();
|
||||
|
||||
constexpr QVectorLayout L = QVectorLayout::byVDIM;
|
||||
|
||||
const int id = (vdim<<8) | (D1D<<4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x134: return PhysGrad2D<L,1,3,4,8>(NE, B, G, J, X, Y);
|
||||
case 0x146: return PhysGrad2D<L,1,4,6,4>(NE, B, G, J, X, Y);
|
||||
case 0x158: return PhysGrad2D<L,1,5,8,2>(NE, B, G, J, X, Y);
|
||||
|
||||
case 0x233: return PhysGrad2D<L,2,3,3,8>(NE, B, G, J, X, Y);
|
||||
case 0x234: return PhysGrad2D<L,2,3,4,8>(NE, B, G, J, X, Y);
|
||||
case 0x246: return PhysGrad2D<L,2,4,6,4>(NE, B, G, J, X, Y);
|
||||
case 0x258: return PhysGrad2D<L,2,5,8,2>(NE, B, G, J, X, Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = MAX_D1D;
|
||||
constexpr int MQ = MAX_Q1D;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad2D<L,0,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x134: return PhysGrad3D<L,1,3,4>(NE, B, G, J, X, Y);
|
||||
case 0x146: return PhysGrad3D<L,1,4,6>(NE, B, G, J, X, Y);
|
||||
case 0x158: return PhysGrad3D<L,1,5,8>(NE, B, G, J, X, Y);
|
||||
|
||||
case 0x334: return PhysGrad3D<L,3,3,4>(NE, B, G, J, X, Y);
|
||||
case 0x346: return PhysGrad3D<L,3,4,6>(NE, B, G, J, X, Y);
|
||||
case 0x358: return PhysGrad3D<L,3,5,8>(NE, B, G, J, X, Y);
|
||||
default:
|
||||
{
|
||||
constexpr int MD = 8;
|
||||
constexpr int MQ = 8;
|
||||
MFEM_VERIFY(D1D <= MD, "Orders higher than " << MD-1
|
||||
<< " are not supported!");
|
||||
MFEM_VERIFY(Q1D <= MQ, "Quadrature rules with more than " << MQ
|
||||
<< " 1D points are not supported!");
|
||||
PhysGrad3D<L,0,0,0,MD,MQ>(NE, B, G, J, X, Y, vdim, D1D, Q1D);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel");
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
+52
-421
@@ -17,6 +17,55 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
|
||||
: ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
|
||||
ndofs(fes.GetNDofs())
|
||||
{
|
||||
height = vdim*ne*ndof;
|
||||
width = vdim*ne*ndof;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_y = Reshape(y.Write(), nd, vd, ne);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), nd, vd, ne);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
ElementRestriction::ElementRestriction(const FiniteElementSpace &f,
|
||||
ElementDofOrdering e_ordering)
|
||||
: fes(f),
|
||||
@@ -232,309 +281,7 @@ void ElementRestriction::BooleanMask(Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
void ElementRestriction::FillSparseMatrix(const Vector &mat_ea,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
mat.GetMemoryI().New(mat.Height()+1, mat.GetMemoryI().GetMemoryType());
|
||||
const int nnz = FillI(mat);
|
||||
mat.GetMemoryJ().New(nnz, mat.GetMemoryJ().GetMemoryType());
|
||||
mat.GetMemoryData().New(nnz, mat.GetMemoryData().GetMemoryType());
|
||||
FillJAndData(mat_ea, mat);
|
||||
}
|
||||
|
||||
template <int MaxNbNbr>
|
||||
static MFEM_HOST_DEVICE int GetMinElt(const int *my_elts, const int nbElts,
|
||||
const int *nbr_elts, const int nbrNbElts)
|
||||
{
|
||||
// Building the intersection
|
||||
int inter[MaxNbNbr];
|
||||
int cpt = 0;
|
||||
for (int i = 0; i < nbElts; i++)
|
||||
{
|
||||
const int e_i = my_elts[i];
|
||||
for (int j = 0; j < nbrNbElts; j++)
|
||||
{
|
||||
if (e_i==nbr_elts[j])
|
||||
{
|
||||
inter[cpt] = e_i;
|
||||
cpt++;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finding the minimum
|
||||
int min = inter[0];
|
||||
for (int i = 1; i < cpt; i++)
|
||||
{
|
||||
if (inter[i] < min)
|
||||
{
|
||||
min = inter[i];
|
||||
}
|
||||
}
|
||||
return min;
|
||||
}
|
||||
|
||||
/** Returns the index where a non-zero entry should be added and increment the
|
||||
number of non-zeros for the row i_L. */
|
||||
static MFEM_HOST_DEVICE int GetAndIncrementNnzIndex(const int i_L, int* I)
|
||||
{
|
||||
int ind = AtomicAdd(I[i_L],1);
|
||||
return ind;
|
||||
}
|
||||
|
||||
int ElementRestriction::FillI(SparseMatrix &mat) const
|
||||
{
|
||||
static constexpr int Max = MaxNbNbr;
|
||||
const int all_dofs = ndofs;
|
||||
const int vd = vdim;
|
||||
const int elt_dofs = dof;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = indices.Read();
|
||||
auto d_gatherMap = gatherMap.Read();
|
||||
MFEM_FORALL(i_L, vd*all_dofs+1,
|
||||
{
|
||||
I[i_L] = 0;
|
||||
});
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < elt_dofs; i++)
|
||||
{
|
||||
int i_elts[Max];
|
||||
const int i_E = e*elt_dofs + i;
|
||||
const int i_L = d_gatherMap[i_E];
|
||||
const int i_offset = d_offsets[i_L];
|
||||
const int i_nextOffset = d_offsets[i_L+1];
|
||||
const int i_nbElts = i_nextOffset - i_offset;
|
||||
for (int e_i = 0; e_i < i_nbElts; ++e_i)
|
||||
{
|
||||
const int i_E = d_indices[i_offset+e_i];
|
||||
i_elts[e_i] = i_E/elt_dofs;
|
||||
}
|
||||
for (int j = 0; j < elt_dofs; j++)
|
||||
{
|
||||
const int j_E = e*elt_dofs + j;
|
||||
const int j_L = d_gatherMap[j_E];
|
||||
const int j_offset = d_offsets[j_L];
|
||||
const int j_nextOffset = d_offsets[j_L+1];
|
||||
const int j_nbElts = j_nextOffset - j_offset;
|
||||
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
|
||||
{
|
||||
GetAndIncrementNnzIndex(i_L, I);
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
for (int e_j = 0; e_j < j_nbElts; ++e_j)
|
||||
{
|
||||
const int j_E = d_indices[j_offset+e_j];
|
||||
const int elt = j_E/elt_dofs;
|
||||
j_elts[e_j] = elt;
|
||||
}
|
||||
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
|
||||
if (e == min_e) // add the nnz only once
|
||||
{
|
||||
GetAndIncrementNnzIndex(i_L, I);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
// We need to sum the entries of I, we do it on CPU as it is very sequential.
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
const int nTdofs = vd*all_dofs;
|
||||
int sum = 0;
|
||||
for (int i = 0; i < nTdofs; i++)
|
||||
{
|
||||
const int nnz = h_I[i];
|
||||
h_I[i] = sum;
|
||||
sum+=nnz;
|
||||
}
|
||||
h_I[nTdofs] = sum;
|
||||
// We return the number of nnz
|
||||
return h_I[nTdofs];
|
||||
}
|
||||
|
||||
void ElementRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
static constexpr int Max = MaxNbNbr;
|
||||
const int all_dofs = ndofs;
|
||||
const int vd = vdim;
|
||||
const int elt_dofs = dof;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto d_offsets = offsets.Read();
|
||||
auto d_indices = indices.Read();
|
||||
auto d_gatherMap = gatherMap.Read();
|
||||
auto mat_ea = Reshape(ea_data.Read(), elt_dofs, elt_dofs, ne);
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < elt_dofs; i++)
|
||||
{
|
||||
int i_elts[Max];
|
||||
int i_B[Max];
|
||||
const int i_E = e*elt_dofs + i;
|
||||
const int i_L = d_gatherMap[i_E];
|
||||
const int i_offset = d_offsets[i_L];
|
||||
const int i_nextOffset = d_offsets[i_L+1];
|
||||
const int i_nbElts = i_nextOffset - i_offset;
|
||||
for (int e_i = 0; e_i < i_nbElts; ++e_i)
|
||||
{
|
||||
const int i_E = d_indices[i_offset+e_i];
|
||||
i_elts[e_i] = i_E/elt_dofs;
|
||||
i_B[e_i] = i_E%elt_dofs;
|
||||
}
|
||||
for (int j = 0; j < elt_dofs; j++)
|
||||
{
|
||||
const int j_E = e*elt_dofs + j;
|
||||
const int j_L = d_gatherMap[j_E];
|
||||
const int j_offset = d_offsets[j_L];
|
||||
const int j_nextOffset = d_offsets[j_L+1];
|
||||
const int j_nbElts = j_nextOffset - j_offset;
|
||||
if (i_nbElts == 1 || j_nbElts == 1) // no assembly required
|
||||
{
|
||||
const int nnz = GetAndIncrementNnzIndex(i_L, I);
|
||||
J[nnz] = j_L;
|
||||
Data[nnz] = mat_ea(j,i,e);
|
||||
}
|
||||
else // assembly required
|
||||
{
|
||||
int j_elts[Max];
|
||||
int j_B[Max];
|
||||
for (int e_j = 0; e_j < j_nbElts; ++e_j)
|
||||
{
|
||||
const int j_E = d_indices[j_offset+e_j];
|
||||
const int elt = j_E/elt_dofs;
|
||||
j_elts[e_j] = elt;
|
||||
j_B[e_j] = j_E%elt_dofs;
|
||||
}
|
||||
int min_e = GetMinElt<Max>(i_elts, i_nbElts, j_elts, j_nbElts);
|
||||
if (e == min_e) // add the nnz only once
|
||||
{
|
||||
double val = 0.0;
|
||||
for (int i = 0; i < i_nbElts; i++)
|
||||
{
|
||||
const int e_i = i_elts[i];
|
||||
const int i_Bloc = i_B[i];
|
||||
for (int j = 0; j < j_nbElts; j++)
|
||||
{
|
||||
const int e_j = j_elts[j];
|
||||
const int j_Bloc = j_B[j];
|
||||
if (e_i == e_j)
|
||||
{
|
||||
val += mat_ea(j_Bloc, i_Bloc, e_i);
|
||||
}
|
||||
}
|
||||
}
|
||||
const int nnz = GetAndIncrementNnzIndex(i_L, I);
|
||||
J[nnz] = j_L;
|
||||
Data[nnz] = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
// We need to shift again the entries of I, we do it on CPU as it is very
|
||||
// sequential.
|
||||
auto h_I = mat.HostReadWriteI();
|
||||
const int size = vd*all_dofs;
|
||||
for (int i = 0; i < size; i++)
|
||||
{
|
||||
h_I[size-i] = h_I[size-(i+1)];
|
||||
}
|
||||
h_I[0] = 0;
|
||||
}
|
||||
|
||||
L2ElementRestriction::L2ElementRestriction(const FiniteElementSpace &fes)
|
||||
: ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndof(ne > 0 ? fes.GetFE(0)->GetDof() : 0),
|
||||
ndofs(fes.GetNDofs())
|
||||
{
|
||||
height = vdim*ne*ndof;
|
||||
width = vdim*ne*ndof;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), t?vd:ndofs, t?ndofs:vd);
|
||||
auto d_y = Reshape(y.Write(), nd, vd, ne);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(dof, c, e) = d_x(t?c:idx, t?idx:c);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::MultTranspose(const Vector &x, Vector &y) const
|
||||
{
|
||||
const int nd = ndof;
|
||||
const int vd = vdim;
|
||||
const bool t = byvdim;
|
||||
auto d_x = Reshape(x.Read(), nd, vd, ne);
|
||||
auto d_y = Reshape(y.Write(), t?vd:ndofs, t?ndofs:vd);
|
||||
MFEM_FORALL(i, ndofs,
|
||||
{
|
||||
const int idx = i;
|
||||
const int dof = idx % nd;
|
||||
const int e = idx / nd;
|
||||
for (int c = 0; c < vd; ++c)
|
||||
{
|
||||
d_y(t?c:idx,t?idx:c) = d_x(dof, c, e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2ElementRestriction::FillI(SparseMatrix &mat) const
|
||||
{
|
||||
const int elem_dofs = ndof;
|
||||
const int vd = vdim;
|
||||
auto I = mat.WriteI();
|
||||
MFEM_FORALL(dof, ne*elem_dofs*vd,
|
||||
{
|
||||
I[dof] = elem_dofs;
|
||||
});
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE int AddNnz(const int iE, int *I, const int dofs)
|
||||
{
|
||||
int val = AtomicAdd(I[iE],dofs);
|
||||
return val;
|
||||
}
|
||||
|
||||
void L2ElementRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat) const
|
||||
{
|
||||
const int elem_dofs = ndof;
|
||||
const int vd = vdim;
|
||||
auto I = mat.ReadWriteI();
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
auto mat_ea = Reshape(ea_data.Read(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(iE, ne*elem_dofs*vd,
|
||||
{
|
||||
const int offset = AddNnz(iE,I,elem_dofs);
|
||||
const int e = iE/elem_dofs;
|
||||
const int i = iE%elem_dofs;
|
||||
for (int j = 0; j < elem_dofs; j++)
|
||||
{
|
||||
J[offset+j] = e*elem_dofs+j;
|
||||
Data[offset+j] = mat_ea(j,i,e);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Return the face degrees of freedom returned in Lexicographic order.
|
||||
/// Return the face degrees of freedom returned in Lexicographic order.
|
||||
void GetFaceDofs(const int dim, const int face_id,
|
||||
const int dof1d, Array<int> &faceMap)
|
||||
{
|
||||
@@ -953,7 +700,7 @@ static int PermuteFace3D(const int face_id1, const int face_id2,
|
||||
return ToLexOrdering3D(face_id2, size1d, new_i, new_j);
|
||||
}
|
||||
|
||||
// Permute dofs or quads on a face for e2 to match with the ordering of e1
|
||||
/// Permute dofs or quads on a face for e2 to match with the ordering of e1
|
||||
int PermuteFaceL2(const int dim, const int face_id1,
|
||||
const int face_id2, const int orientation,
|
||||
const int size1d, const int index)
|
||||
@@ -973,32 +720,23 @@ int PermuteFaceL2(const int dim, const int face_id1,
|
||||
}
|
||||
|
||||
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
|
||||
const ElementDofOrdering e_ordering,
|
||||
const FaceType type,
|
||||
const L2FaceValues m)
|
||||
: fes(fes),
|
||||
nf(fes.GetNFbyType(type)),
|
||||
ne(fes.GetNE()),
|
||||
vdim(fes.GetVDim()),
|
||||
byvdim(fes.GetOrdering() == Ordering::byVDIM),
|
||||
ndofs(fes.GetNDofs()),
|
||||
dof(nf > 0 ?
|
||||
fes.GetTraceElement(0, fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof()
|
||||
: 0),
|
||||
elemDofs(fes.GetFE(0)->GetDof()),
|
||||
m(m),
|
||||
nfdofs(nf*dof),
|
||||
scatter_indices1(nf*dof),
|
||||
scatter_indices2(m==L2FaceValues::DoubleValued?nf*dof:0),
|
||||
offsets(ndofs+1),
|
||||
gather_indices((m==L2FaceValues::DoubleValued? 2 : 1)*nf*dof)
|
||||
{
|
||||
}
|
||||
|
||||
L2FaceRestriction::L2FaceRestriction(const FiniteElementSpace &fes,
|
||||
const ElementDofOrdering e_ordering,
|
||||
const FaceType type,
|
||||
const L2FaceValues m)
|
||||
: L2FaceRestriction(fes, type, m)
|
||||
{
|
||||
// If fespace == L2
|
||||
const FiniteElement *fe = fes.GetFE(0);
|
||||
@@ -1296,113 +1034,6 @@ void L2FaceRestriction::MultTranspose(const Vector& x, Vector& y) const
|
||||
}
|
||||
}
|
||||
|
||||
void L2FaceRestriction::FillI(SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int iE1 = d_indices1[fdof];
|
||||
const int iE2 = d_indices2[fdof];
|
||||
AddNnz(iE1,I,face_dofs);
|
||||
AddNnz(iE2,I,face_dofs);
|
||||
});
|
||||
}
|
||||
|
||||
void L2FaceRestriction::FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto I = mat.ReadWriteI();
|
||||
auto mat_fea = Reshape(ea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto J = mat.WriteJ();
|
||||
auto Data = mat.WriteData();
|
||||
MFEM_FORALL(fdof, nf*face_dofs,
|
||||
{
|
||||
const int f = fdof/face_dofs;
|
||||
const int iF = fdof%face_dofs;
|
||||
const int iE1 = d_indices1[f*face_dofs+iF];
|
||||
const int iE2 = d_indices2[f*face_dofs+iF];
|
||||
const int offset1 = AddNnz(iE1,I,face_dofs);
|
||||
const int offset2 = AddNnz(iE2,I,face_dofs);
|
||||
for (int jF = 0; jF < face_dofs; jF++)
|
||||
{
|
||||
const int jE1 = d_indices1[f*face_dofs+jF];
|
||||
const int jE2 = d_indices2[f*face_dofs+jF];
|
||||
J[offset2+jF] = jE1;
|
||||
J[offset1+jF] = jE2;
|
||||
Data[offset2+jF] = mat_fea(jF,iF,0,f);
|
||||
Data[offset1+jF] = mat_fea(jF,iF,1,f);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void L2FaceRestriction::AddFaceMatricesToElementMatrices(Vector &fea_data,
|
||||
Vector &ea_data) const
|
||||
{
|
||||
const int face_dofs = dof;
|
||||
const int elem_dofs = elemDofs;
|
||||
const int NE = ne;
|
||||
if (m==L2FaceValues::DoubleValued)
|
||||
{
|
||||
auto d_indices1 = scatter_indices1.Read();
|
||||
auto d_indices2 = scatter_indices2.Read();
|
||||
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, 2, nf);
|
||||
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
const int e1 = d_indices1[f*face_dofs]/elem_dofs;
|
||||
const int e2 = d_indices2[f*face_dofs]/elem_dofs;
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jB1 = d_indices1[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iB1 = d_indices1[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iB1,jB1,e1), mat_fea(i,j,0,f));
|
||||
}
|
||||
}
|
||||
if (e2 < NE)
|
||||
{
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jB2 = d_indices2[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iB2 = d_indices2[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iB2,jB2,e2), mat_fea(i,j,1,f));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
auto d_indices = scatter_indices1.Read();
|
||||
auto mat_fea = Reshape(fea_data.Read(), face_dofs, face_dofs, nf);
|
||||
auto mat_ea = Reshape(ea_data.ReadWrite(), elem_dofs, elem_dofs, ne);
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
const int e = d_indices[f*face_dofs]/elem_dofs;
|
||||
for (int j = 0; j < face_dofs; j++)
|
||||
{
|
||||
const int jE = d_indices[f*face_dofs+j]%elem_dofs;
|
||||
for (int i = 0; i < face_dofs; i++)
|
||||
{
|
||||
const int iE = d_indices[f*face_dofs+i]%elem_dofs;
|
||||
AtomicAdd(mat_ea(iE,jE,e), mat_fea(i,j,f));
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
int ToLexOrdering(const int dim, const int face_id, const int size1d,
|
||||
const int index)
|
||||
{
|
||||
|
||||
+1
-39
@@ -30,11 +30,6 @@ enum class L2FaceValues : bool {SingleValued, DoubleValued};
|
||||
objects, see FiniteElementSpace::GetElementRestriction(). */
|
||||
class ElementRestriction : public Operator
|
||||
{
|
||||
private:
|
||||
/** This number defines the maximum number of elements any dof can belong to
|
||||
for the FillSparseMatrix method. */
|
||||
static const int MaxNbNbr = 16;
|
||||
|
||||
protected:
|
||||
const FiniteElementSpace &fes;
|
||||
const int ne;
|
||||
@@ -64,16 +59,6 @@ public:
|
||||
emulate SetSubVector and its transpose on GPUs. This method is running on
|
||||
the host, since the `processed` array requires a large shared memory. */
|
||||
void BooleanMask(Vector& y) const;
|
||||
|
||||
/// Fill a Sparse Matrix with Element Matrices.
|
||||
void FillSparseMatrix(const Vector &mat_ea, SparseMatrix &mat) const;
|
||||
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ElementRestriction. */
|
||||
int FillI(SparseMatrix &mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this ElementRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
|
||||
};
|
||||
|
||||
/// Operator that converts L2 FiniteElementSpace L-vectors to E-vectors.
|
||||
@@ -92,12 +77,6 @@ public:
|
||||
L2ElementRestriction(const FiniteElementSpace&);
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this ElementRestriction. */
|
||||
void FillI(SparseMatrix &mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this L2FaceRestriction, and the values of ea_data. */
|
||||
void FillJAndData(const Vector &ea_data, SparseMatrix &mat) const;
|
||||
};
|
||||
|
||||
/// Operator that extracts Face degrees of freedom.
|
||||
@@ -132,12 +111,10 @@ class L2FaceRestriction : public Operator
|
||||
protected:
|
||||
const FiniteElementSpace &fes;
|
||||
const int nf;
|
||||
const int ne;
|
||||
const int vdim;
|
||||
const bool byvdim;
|
||||
const int ndofs;
|
||||
const int dof;
|
||||
const int elemDofs;
|
||||
const L2FaceValues m;
|
||||
const int nfdofs;
|
||||
Array<int> scatter_indices1;
|
||||
@@ -145,27 +122,12 @@ protected:
|
||||
Array<int> offsets;
|
||||
Array<int> gather_indices;
|
||||
|
||||
L2FaceRestriction(const FiniteElementSpace&,
|
||||
const FaceType,
|
||||
const L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
|
||||
public:
|
||||
L2FaceRestriction(const FiniteElementSpace&, const ElementDofOrdering,
|
||||
const FaceType,
|
||||
const L2FaceValues m = L2FaceValues::DoubleValued);
|
||||
virtual void Mult(const Vector &x, Vector &y) const;
|
||||
void Mult(const Vector &x, Vector &y) const;
|
||||
void MultTranspose(const Vector &x, Vector &y) const;
|
||||
/** Fill the I array of SparseMatrix corresponding to the sparsity pattern
|
||||
given by this L2FaceRestriction. */
|
||||
virtual void FillI(SparseMatrix &mat, SparseMatrix &face_mat) const;
|
||||
/** Fill the J and Data arrays of SparseMatrix corresponding to the sparsity
|
||||
pattern given by this L2FaceRestriction, and the values of ea_data. */
|
||||
virtual void FillJAndData(const Vector &ea_data,
|
||||
SparseMatrix &mat,
|
||||
SparseMatrix &face_mat) const;
|
||||
/// This methods adds the DG face matrices to the element matrices.
|
||||
void AddFaceMatricesToElementMatrices(Vector &fea_data,
|
||||
Vector &ea_data) const;
|
||||
};
|
||||
|
||||
// Return the face degrees of freedom returned in Lexicographic order.
|
||||
|
||||
+75
-573
@@ -13,7 +13,6 @@
|
||||
#include "linearform.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
#include "tmop_tools.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -442,8 +441,8 @@ void TMOP_Metric_058::AssembleH(const DenseMatrix &Jpt,
|
||||
double TMOP_Metric_077::EvalW(const DenseMatrix &Jpt) const
|
||||
{
|
||||
ie.SetJacobian(Jpt.GetData());
|
||||
const double I2b = ie.Get_I2b();
|
||||
return 0.5*(I2b*I2b + 1./(I2b*I2b) - 2.);
|
||||
const double I2 = ie.Get_I2b();
|
||||
return 0.5*(I2*I2 + 1./(I2*I2) - 2.);
|
||||
}
|
||||
|
||||
void TMOP_Metric_077::EvalP(const DenseMatrix &Jpt, DenseMatrix &P) const
|
||||
@@ -928,20 +927,9 @@ void TargetConstructor::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void TargetConstructor::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
|
||||
|
||||
// TODO: Compute derivative for targets with GIVEN_SHAPE or/and GIVEN_SIZE
|
||||
for (int i = 0; i < Tpr.GetFE()->GetDim()*ir.GetNPoints(); i++) { dJtr(i) = 0.; }
|
||||
}
|
||||
|
||||
void AnalyticAdaptTC::SetAnalyticTargetSpec(Coefficient *sspec,
|
||||
VectorCoefficient *vspec,
|
||||
TMOPMatrixCoefficient *mspec)
|
||||
MatrixCoefficient *mspec)
|
||||
{
|
||||
scalar_tspec = sspec;
|
||||
vector_tspec = vspec;
|
||||
@@ -982,39 +970,6 @@ void AnalyticAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void AnalyticAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
const FiniteElement *fe = Tpr.GetFE();
|
||||
DenseMatrix point_mat;
|
||||
point_mat.UseExternalData(elfun.GetData(), fe->GetDof(), fe->GetDim());
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
case GIVEN_FULL:
|
||||
{
|
||||
MFEM_VERIFY(matrix_tspec != NULL,
|
||||
"Target type GIVEN_FULL requires a TMOPMatrixCoefficient.");
|
||||
|
||||
for (int d = 0; d < fe->GetDim(); d++)
|
||||
{
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
Tpr.SetIntPoint(&ip);
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
matrix_tspec->EvalGrad(dJtr_i, Tpr, ip, d);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MFEM_ABORT("Incompatible target type for analytic adaptation!");
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
void DiscreteAdaptTC::FinalizeParDiscreteTargetSpec(const ParGridFunction
|
||||
&tspec_)
|
||||
@@ -1039,10 +994,11 @@ void DiscreteAdaptTC::SetTspecAtIndex(int idx, const ParGridFunction &tspec_)
|
||||
{
|
||||
const int vdim = tspec_.FESpace()->GetVDim(),
|
||||
dof_cnt = tspec_.Size()/vdim;
|
||||
const auto tspec__d = tspec_.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
const int offset = idx*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+idx*dof_cnt) = tspec_(i);
|
||||
}
|
||||
|
||||
FinalizeParDiscreteTargetSpec(tspec_);
|
||||
}
|
||||
|
||||
@@ -1102,33 +1058,34 @@ void DiscreteAdaptTC::SetDiscreteTargetBase(const GridFunction &tspec_)
|
||||
// make a copy of tspec->tspec_temp, increase its size, and
|
||||
// copy data from tspec_temp -> tspec, then add new entries
|
||||
Vector tspec_temp = tspec;
|
||||
tspec.UseDevice(true);
|
||||
tspec_sav.UseDevice(true);
|
||||
tspec.SetSize(ncomp*dof_cnt);
|
||||
|
||||
const auto tspec_temp_d = tspec_temp.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
MFEM_FORALL(i, tspec_temp.Size(), tspec_d[i] = tspec_temp_d[i];);
|
||||
for (int i = 0; i < tspec_temp.Size(); i++)
|
||||
{
|
||||
tspec(i) = tspec_temp(i);
|
||||
}
|
||||
|
||||
const auto tspec__d = tspec_.Read();
|
||||
const int offset = (ncomp-vdim)*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+(ncomp-vdim)*dof_cnt) = tspec_(i);
|
||||
}
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::SetTspecAtIndex(int idx, const GridFunction &tspec_)
|
||||
{
|
||||
const int vdim = tspec_.FESpace()->GetVDim(),
|
||||
dof_cnt = tspec_.Size()/vdim;
|
||||
for (int i = 0; i < dof_cnt*vdim; i++)
|
||||
{
|
||||
tspec(i+idx*dof_cnt) = tspec_(i);
|
||||
}
|
||||
|
||||
const auto tspec__d = tspec_.Read();
|
||||
auto tspec_d = tspec.ReadWrite();
|
||||
const int offset = idx*dof_cnt;
|
||||
MFEM_FORALL(i, dof_cnt*vdim, tspec_d[i+offset] = tspec__d[i];);
|
||||
FinalizeSerialDiscreteTargetSpec();
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::SetSerialDiscreteTargetSize(const GridFunction &tspec_)
|
||||
{
|
||||
|
||||
if (sizeidx > -1) { SetTspecAtIndex(sizeidx, tspec_); return; }
|
||||
sizeidx = ncomp;
|
||||
SetDiscreteTargetBase(tspec_);
|
||||
@@ -1211,7 +1168,7 @@ void DiscreteAdaptTC::UpdateTargetSpecificationAtNode(const FiniteElement &el,
|
||||
|
||||
Array<int> dofs;
|
||||
tspec_fes->GetElementDofs(T.ElementNo, dofs);
|
||||
const int cnt = tspec.Size()/ncomp; // dofs per scalar-field
|
||||
const int cnt = tspec.Size()/ncomp; //dofs per scalar-field
|
||||
|
||||
for (int i = 0; i < ncomp; i++)
|
||||
{
|
||||
@@ -1239,9 +1196,6 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
DenseTensor &Jtr) const
|
||||
{
|
||||
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
|
||||
const int dim = fe.GetDim(),
|
||||
nqp = ir.GetNPoints();
|
||||
Jtrcomp.SetSize(dim, dim, 4*nqp);
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
@@ -1251,44 +1205,36 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
const DenseMatrix &Wideal =
|
||||
Geometries.GetGeomToPerfGeomJac(fe.GetGeomType());
|
||||
const int dim = Wideal.Height(),
|
||||
ndofs = tspec_fes->GetFE(e_id)->GetDof(),
|
||||
ndofs = tspec_fes->GetFE(0)->GetDof(),
|
||||
ntspec_dofs = ndofs*ncomp;
|
||||
|
||||
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
|
||||
par_vals_c1, par_vals_c2, par_vals_c3;
|
||||
|
||||
Array<int> dofs;
|
||||
DenseMatrix D_rho(dim), Q_phi(dim), R_theta(dim);
|
||||
tspec_fesv->GetElementVDofs(e_id, dofs);
|
||||
tspec.UseDevice(true);
|
||||
tspec.GetSubVector(dofs, tspec_vals);
|
||||
|
||||
for (int q = 0; q < nqp; q++)
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(q);
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
Jtr(i) = Wideal; //Initialize to identity
|
||||
|
||||
Jtr(q) = Wideal; // Initialize to identity
|
||||
for (int d = 0; d < 4; d++)
|
||||
{
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(d + 4*q), dim, dim);
|
||||
Jtrcomp_q = Wideal; // Initialize to identity
|
||||
}
|
||||
|
||||
if (sizeidx != -1) // Set size
|
||||
if (sizeidx != -1) //Set size
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
|
||||
const double min_size = par_vals.Min();
|
||||
MFEM_VERIFY(min_size > 0.0,
|
||||
"Non-positive size propagated in the target definition.");
|
||||
const double size = std::max(shape * par_vals, min_size);
|
||||
Jtr(q).Set(std::pow(size, 1.0/dim), Jtr(q));
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(0 + 4*q), dim, dim);
|
||||
Jtrcomp_q = Jtr(q);
|
||||
} // Done size
|
||||
Jtr(i).Set(std::pow(size, 1.0/dim), Jtr(i));
|
||||
} //Done size
|
||||
|
||||
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
|
||||
|
||||
if (aspectratioidx != -1) // Set aspect ratio
|
||||
if (aspectratioidx != -1) //Set aspect ratio
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1316,13 +1262,12 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
D_rho(1,1) = pow(rho2,2./3.);
|
||||
D_rho(2,2) = pow(rho3,2./3.);
|
||||
}
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(1 + 4*q), dim, dim);
|
||||
Jtrcomp_q = D_rho;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(D_rho, Temp, Jtr(q));
|
||||
} // Done aspect ratio
|
||||
|
||||
if (skewidx != -1) // Set skew
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(D_rho, Temp, Jtr(i));
|
||||
} //Done aspect ratio
|
||||
|
||||
if (skewidx != -1) //Set skew
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1358,13 +1303,12 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
|
||||
Q_phi(2,2) = sin(phi13)*sin(chi);
|
||||
}
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*q), dim, dim);
|
||||
Jtrcomp_q = Q_phi;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(Q_phi, Temp, Jtr(q));
|
||||
} // Done skew
|
||||
|
||||
if (orientationidx != -1) // Set orientation
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(Q_phi, Temp, Jtr(i));
|
||||
} // done skew
|
||||
|
||||
if (orientationidx != -1) //Set orientation
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
@@ -1389,28 +1333,33 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
const double psi = shape * par_vals_c2;
|
||||
const double beta = shape * par_vals_c3;
|
||||
|
||||
DenseMatrix R_tp(dim), R_beta(dim), R_theta(dim);
|
||||
double ct = cos(theta), st = sin(theta),
|
||||
cp = cos(psi), sp = sin(psi),
|
||||
cb = cos(beta), sb = sin(beta);
|
||||
cp = cos(psi), sp = sin(psi);
|
||||
R_tp(0,0) = ct*sp;
|
||||
R_tp(1,0) = st*sp;
|
||||
R_tp(2,0) = cp;
|
||||
|
||||
R_theta = 0.;
|
||||
R_theta(0,0) = ct*sp;
|
||||
R_theta(1,0) = st*sp;
|
||||
R_theta(2,0) = cp;
|
||||
R_tp(0,1) = -(ct*st*sp*sp)/(1+cp);
|
||||
R_tp(1,1) = cp+(pow(ct,2.)*pow(sp,2.))/(1+cp);
|
||||
R_tp(2,1) = -st*sp;
|
||||
|
||||
R_theta(0,1) = -st*cb + ct*cp*sb;
|
||||
R_theta(1,1) = ct*cb + st*cp*sb;
|
||||
R_theta(2,1) = -sp*sb;
|
||||
R_tp(0,2) = -cp-(pow(st,2.)*pow(sp,2.))/(1+cp);
|
||||
R_tp(1,2) = -R_tp(0,1);
|
||||
R_tp(2,2) = ct*sp;
|
||||
|
||||
R_theta(0,0) = -st*sb - ct*cp*cb;
|
||||
R_theta(1,0) = ct*sb - st*cp*cb;
|
||||
R_theta(2,0) = sp*cb;
|
||||
R_beta = 0.;
|
||||
R_beta(0,0) = 1.;
|
||||
R_beta(1,1) = cos(beta);
|
||||
R_beta(1,2) = -sin(beta);
|
||||
R_beta(2,1) = sin(beta);
|
||||
R_beta(2,2) = cos(beta);
|
||||
|
||||
Mult(R_tp, R_beta, R_theta);
|
||||
}
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(3 + 4*q), dim, dim);
|
||||
Jtrcomp_q = R_theta;
|
||||
DenseMatrix Temp = Jtr(q);
|
||||
Mult(R_theta, Temp, Jtr(q));
|
||||
} // Done orientation
|
||||
DenseMatrix Temp = Jtr(i);
|
||||
Mult(R_theta, Temp, Jtr(i));
|
||||
} // done orientation
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1419,353 +1368,6 @@ void DiscreteAdaptTC::ComputeElementTargets(int e_id, const FiniteElement &fe,
|
||||
}
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const
|
||||
{
|
||||
MFEM_ASSERT(target_type == IDEAL_SHAPE_UNIT_SIZE || nodes != NULL, "");
|
||||
|
||||
MFEM_VERIFY(tspec_fesv, "No target specifications have been set.");
|
||||
|
||||
dJtr = 0.;
|
||||
const int e_id = Tpr.ElementNo;
|
||||
const FiniteElement *fe = Tpr.GetFE();
|
||||
|
||||
switch (target_type)
|
||||
{
|
||||
case IDEAL_SHAPE_GIVEN_SIZE:
|
||||
case GIVEN_SHAPE_AND_SIZE:
|
||||
{
|
||||
const DenseMatrix &Wideal =
|
||||
Geometries.GetGeomToPerfGeomJac(fe->GetGeomType());
|
||||
const int dim = Wideal.Height(),
|
||||
ndofs = fe->GetDof(),
|
||||
ntspec_dofs = ndofs*ncomp;
|
||||
|
||||
Vector shape(ndofs), tspec_vals(ntspec_dofs), par_vals,
|
||||
par_vals_c1(ndofs), par_vals_c2(ndofs), par_vals_c3(ndofs);
|
||||
|
||||
Array<int> dofs;
|
||||
DenseMatrix dD_rho(dim), dQ_phi(dim), dR_theta(dim);
|
||||
DenseMatrix dQ_phi13(dim), dQ_phichi(dim); // dQ_phi is used for dQ/dphi12 in 3D
|
||||
DenseMatrix dR_psi(dim), dR_beta(dim);
|
||||
tspec_fesv->GetElementVDofs(e_id, dofs);
|
||||
tspec.GetSubVector(dofs, tspec_vals);
|
||||
|
||||
DenseMatrix grad_e_c1(ndofs, dim),
|
||||
grad_e_c2(ndofs, dim),
|
||||
grad_e_c3(ndofs, dim);
|
||||
Vector grad_ptr_c1(grad_e_c1.GetData(), ndofs*dim),
|
||||
grad_ptr_c2(grad_e_c2.GetData(), ndofs*dim),
|
||||
grad_ptr_c3(grad_e_c3.GetData(), ndofs*dim);
|
||||
|
||||
DenseMatrix grad_phys; // This will be (dof x dim, dof).
|
||||
fe->ProjectGrad(*fe, Tpr, grad_phys);
|
||||
|
||||
for (int i = 0; i < ir.GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(i);
|
||||
DenseMatrix Jtrcomp_s(Jtrcomp.GetData(0 + 4*i), dim, dim); // size
|
||||
DenseMatrix Jtrcomp_d(Jtrcomp.GetData(1 + 4*i), dim, dim); // aspect-ratio
|
||||
DenseMatrix Jtrcomp_q(Jtrcomp.GetData(2 + 4*i), dim, dim); // skew
|
||||
DenseMatrix Jtrcomp_r(Jtrcomp.GetData(3 + 4*i), dim, dim); // orientation
|
||||
DenseMatrix work1(dim), work2(dim), work3(dim);
|
||||
|
||||
if (sizeidx != -1) // Set size
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+sizeidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double min_size = par_vals.Min();
|
||||
MFEM_VERIFY(min_size > 0.0,
|
||||
"Non-positive size propagated in the target definition.");
|
||||
const double size = std::max(shape * par_vals, min_size);
|
||||
double dz_dsize = (1./dim)*pow(size, 1./dim - 1.);
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_r, work1, work2); // R*Q*D
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = Wideal;
|
||||
work1.Set(dz_dsize, work1); // dz/dsize
|
||||
work1 *= grad_q(d); // dz/dsize*dsize/dx
|
||||
AddMult(work1, work2, dJtr_i); // dz/dx*R*Q*D
|
||||
}
|
||||
} // Done size
|
||||
|
||||
if (target_type == IDEAL_SHAPE_GIVEN_SIZE) { continue; }
|
||||
|
||||
if (aspectratioidx != -1) // Set aspect ratio
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
aspectratioidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double aspectratio = shape * par_vals;
|
||||
dD_rho = 0.;
|
||||
dD_rho(0,0) = -0.5*pow(aspectratio,-1.5);
|
||||
dD_rho(1,1) = 0.5*pow(aspectratio,-0.5);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
|
||||
Mult(work1, Jtrcomp_q, work2); // z*R*Q
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dD_rho;
|
||||
work1 *= grad_q(d); // work1 = dD/drho*drho/dx
|
||||
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
|
||||
}
|
||||
}
|
||||
else // 3D
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
aspectratioidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double rho1 = shape * par_vals_c1;
|
||||
const double rho2 = shape * par_vals_c2;
|
||||
const double rho3 = shape * par_vals_c3;
|
||||
dD_rho = 0.;
|
||||
dD_rho(0,0) = (2./3.)*pow(rho1,-1./3.);
|
||||
dD_rho(1,1) = (2./3.)*pow(rho2,-1./3.);
|
||||
dD_rho(2,2) = (2./3.)*pow(rho3,-1./3.);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work1); // z*R
|
||||
Mult(work1, Jtrcomp_q, work2); // z*R*Q
|
||||
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dD_rho;
|
||||
work1(0,0) *= grad_q1(d);
|
||||
work1(1,2) *= grad_q2(d);
|
||||
work1(2,2) *= grad_q3(d);
|
||||
// work1 = dD/dx = dD/drho1*drho1/dx + dD/drho2*drho2/dx
|
||||
AddMult(work2, work1, dJtr_i); // z*R*Q*dD/dx
|
||||
}
|
||||
}
|
||||
} // Done aspect ratio
|
||||
|
||||
if (skewidx != -1) // Set skew
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
skewidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double skew = shape * par_vals;
|
||||
|
||||
dQ_phi = 0.;
|
||||
dQ_phi(0,0) = 1.;
|
||||
dQ_phi(0,1) = -sin(skew);
|
||||
dQ_phi(1,1) = cos(skew);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dQ_phi;
|
||||
work1 *= grad_q(d); // work1 = dQ/dphi*dphi/dx
|
||||
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
|
||||
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
skewidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double phi12 = shape * par_vals_c1;
|
||||
const double phi13 = shape * par_vals_c2;
|
||||
const double chi = shape * par_vals_c3;
|
||||
|
||||
dQ_phi = 0.;
|
||||
dQ_phi(0,0) = 1.;
|
||||
dQ_phi(0,1) = -sin(phi12);
|
||||
dQ_phi(1,1) = cos(phi12);
|
||||
|
||||
dQ_phi13 = 0.;
|
||||
dQ_phi13(0,2) = -sin(phi13);
|
||||
dQ_phi13(1,2) = cos(phi13)*cos(chi);
|
||||
dQ_phi13(2,2) = cos(phi13)*sin(chi);
|
||||
|
||||
dQ_phichi = 0.;
|
||||
dQ_phichi(1,2) = -sin(phi13)*sin(chi);
|
||||
dQ_phichi(2,2) = sin(phi13)*cos(chi);
|
||||
|
||||
Mult(Jtrcomp_s, Jtrcomp_r, work2); // z*R
|
||||
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dQ_phi;
|
||||
work1 *= grad_q1(d); // work1 = dQ/dphi12*dphi12/dx
|
||||
work1.Add(grad_q2(d), dQ_phi13); // + dQ/dphi13*dphi13/dx
|
||||
work1.Add(grad_q3(d), dQ_phichi); // + dQ/dchi*dchi/dx
|
||||
Mult(work1, Jtrcomp_d, work3); // dQ/dx*D
|
||||
AddMult(work2, work3, dJtr_i); // z*R*dQ/dx*D
|
||||
}
|
||||
}
|
||||
} // Done skew
|
||||
|
||||
if (orientationidx != -1) // Set orientation
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
orientationidx*ndofs, ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals, grad_ptr_c1);
|
||||
Vector grad_q(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q);
|
||||
|
||||
const double theta = shape * par_vals;
|
||||
dR_theta(0,0) = -sin(theta);
|
||||
dR_theta(0,1) = -cos(theta);
|
||||
dR_theta(1,0) = cos(theta);
|
||||
dR_theta(1,1) = -sin(theta);
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_s, work1, work2); // z*Q*D
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dR_theta;
|
||||
work1 *= grad_q(d); // work1 = dR/dtheta*dtheta/dx
|
||||
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
par_vals.SetDataAndSize(tspec_vals.GetData()+
|
||||
orientationidx*ndofs, ndofs*3);
|
||||
par_vals_c1.SetData(par_vals.GetData());
|
||||
par_vals_c2.SetData(par_vals.GetData()+ndofs);
|
||||
par_vals_c3.SetData(par_vals.GetData()+2*ndofs);
|
||||
|
||||
grad_phys.Mult(par_vals_c1, grad_ptr_c1);
|
||||
grad_phys.Mult(par_vals_c2, grad_ptr_c2);
|
||||
grad_phys.Mult(par_vals_c3, grad_ptr_c3);
|
||||
Vector grad_q1(dim), grad_q2(dim), grad_q3(dim);
|
||||
tspec_fes->GetFE(e_id)->CalcShape(ip, shape);
|
||||
grad_e_c1.MultTranspose(shape, grad_q1);
|
||||
grad_e_c2.MultTranspose(shape, grad_q2);
|
||||
grad_e_c3.MultTranspose(shape, grad_q3);
|
||||
|
||||
const double theta = shape * par_vals_c1;
|
||||
const double psi = shape * par_vals_c2;
|
||||
const double beta = shape * par_vals_c3;
|
||||
|
||||
const double ct = cos(theta), st = sin(theta),
|
||||
cp = cos(psi), sp = sin(psi),
|
||||
cb = cos(beta), sb = sin(beta);
|
||||
|
||||
dR_theta = 0.;
|
||||
dR_theta(0,0) = -st*sp;
|
||||
dR_theta(1,0) = ct*sp;
|
||||
dR_theta(2,0) = 0;
|
||||
|
||||
dR_theta(0,1) = -ct*cb - st*cp*sb;
|
||||
dR_theta(1,1) = -st*cb + ct*cp*sb;
|
||||
dR_theta(2,1) = 0.;
|
||||
|
||||
dR_theta(0,0) = -ct*sb + st*cp*cb;
|
||||
dR_theta(1,0) = -st*sb - ct*cp*cb;
|
||||
dR_theta(2,0) = 0.;
|
||||
|
||||
dR_beta = 0.;
|
||||
dR_beta(0,0) = 0.;
|
||||
dR_beta(1,0) = 0.;
|
||||
dR_beta(2,0) = 0.;
|
||||
|
||||
dR_beta(0,1) = st*sb + ct*cp*cb;
|
||||
dR_beta(1,1) = -ct*sb + st*cp*cb;
|
||||
dR_beta(2,1) = -sp*cb;
|
||||
|
||||
dR_beta(0,0) = -st*cb + ct*cp*sb;
|
||||
dR_beta(1,0) = ct*cb + st*cp*sb;
|
||||
dR_beta(2,0) = 0.;
|
||||
|
||||
dR_psi = 0.;
|
||||
dR_psi(0,0) = ct*cp;
|
||||
dR_psi(1,0) = st*cp;
|
||||
dR_psi(2,0) = -sp;
|
||||
|
||||
dR_psi(0,1) = 0. - ct*sp*sb;
|
||||
dR_psi(1,1) = 0. + st*sp*sb;
|
||||
dR_psi(2,1) = -cp*sb;
|
||||
|
||||
dR_psi(0,0) = 0. + ct*sp*cb;
|
||||
dR_psi(1,0) = 0. + st*sp*cb;
|
||||
dR_psi(2,0) = cp*cb;
|
||||
|
||||
Mult(Jtrcomp_q, Jtrcomp_d, work1); // Q*D
|
||||
Mult(Jtrcomp_s, work1, work2); // z*Q*D
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
DenseMatrix &dJtr_i = dJtr(i + d*ir.GetNPoints());
|
||||
work1 = dR_theta;
|
||||
work1 *= grad_q1(d); // work1 = dR/dtheta*dtheta/dx
|
||||
work1.Add(grad_q2(d), dR_psi); // +dR/dpsi*dpsi/dx
|
||||
work1.Add(grad_q3(d), dR_beta); // +dR/dbeta*dbeta/dx
|
||||
AddMult(work1, work2, dJtr_i); // z*dR/dx*Q*D
|
||||
}
|
||||
}
|
||||
} // Done orientation
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
MFEM_ABORT("Incompatible target type for discrete adaptation!");
|
||||
}
|
||||
Jtrcomp.Clear();
|
||||
}
|
||||
|
||||
void DiscreteAdaptTC::UpdateGradientTargetSpecification(const Vector &x,
|
||||
const double dx,
|
||||
bool use_flag)
|
||||
@@ -1872,17 +1474,6 @@ void AdaptivityEvaluator::SetParMetaInfo(const ParMesh &m,
|
||||
}
|
||||
#endif
|
||||
|
||||
void AdaptivityEvaluator::ClearGeometricFactors()
|
||||
{
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pmesh) pmesh->DeleteGeometricFactors();
|
||||
if (pfes) pfes->GetParMesh()->DeleteGeometricFactors();
|
||||
#else
|
||||
if (mesh) mesh->DeleteGeometricFactors();
|
||||
if (fes) fes->GetMesh()->DeleteGeometricFactors();
|
||||
#endif
|
||||
}
|
||||
|
||||
AdaptivityEvaluator::~AdaptivityEvaluator()
|
||||
{
|
||||
delete fes;
|
||||
@@ -1910,7 +1501,6 @@ void TMOP_Integrator::EnableLimiting(const GridFunction &n0,
|
||||
{
|
||||
EnableLimiting(n0, w0, lfunc);
|
||||
lim_dist = &dist;
|
||||
if (PA.enabled) { EnableLimitingPA(n0); }
|
||||
}
|
||||
void TMOP_Integrator::EnableLimiting(const GridFunction &n0, Coefficient &w0,
|
||||
TMOP_LimiterFunction *lfunc)
|
||||
@@ -2056,8 +1646,7 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
|
||||
PMatI.MultTranspose(shape, p);
|
||||
pos0.MultTranspose(shape, p0);
|
||||
val += lim_normal *
|
||||
lim_func->Eval(p, p0, d_vals(i)) *
|
||||
coeff0->Eval(*Tpr, ip);
|
||||
lim_func->Eval(p, p0, d_vals(i)) * coeff0->Eval(*Tpr, ip);
|
||||
}
|
||||
|
||||
if (adaptive_limiting)
|
||||
@@ -2108,7 +1697,6 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
{
|
||||
const int dof = el.GetDof(), dim = el.GetDim();
|
||||
|
||||
DenseMatrix Amat(dim), work1(dim), work2(dim);
|
||||
DSh.SetSize(dof, dim);
|
||||
DS.SetSize(dof, dim);
|
||||
Jrt.SetSize(dim);
|
||||
@@ -2124,15 +1712,14 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
elvect = 0.0;
|
||||
Vector weights(nqp);
|
||||
DenseTensor Jtr(dim, dim, nqp);
|
||||
DenseTensor dJtr(dim, dim, dim*nqp);
|
||||
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
|
||||
|
||||
// Limited case.
|
||||
DenseMatrix pos0;
|
||||
Vector shape, p, p0, d_vals, grad;
|
||||
shape.SetSize(dof);
|
||||
if (coeff0)
|
||||
{
|
||||
shape.SetSize(dof);
|
||||
p.SetSize(dim);
|
||||
p0.SetSize(dim);
|
||||
pos0.SetSize(dof, dim);
|
||||
@@ -2152,7 +1739,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
|
||||
// Define ref->physical transformation, when a Coefficient is specified.
|
||||
IsoparametricTransformation *Tpr = NULL;
|
||||
if (coeff1 || coeff0 || zeta || exact_action)
|
||||
if (coeff1 || coeff0 || zeta)
|
||||
{
|
||||
Tpr = new IsoparametricTransformation;
|
||||
Tpr->SetFE(&el);
|
||||
@@ -2160,16 +1747,8 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
Tpr->ElementType = ElementTransformation::ELEMENT;
|
||||
Tpr->Attribute = T.Attribute;
|
||||
Tpr->GetPointMat().Transpose(PMatI); // PointMat = PMatI^T
|
||||
if (exact_action)
|
||||
{
|
||||
targetC->ComputeElementTargetsGradient(*ir, elfun, *Tpr, dJtr);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Vector d_detW_dx(dim);
|
||||
Vector d_Winv_dx(dim);
|
||||
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(q);
|
||||
@@ -2188,44 +1767,13 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
|
||||
if (coeff1) { weight_m *= coeff1->Eval(*Tpr, ip); }
|
||||
|
||||
P *= weight_m;
|
||||
AddMultABt(DS, P, PMatO); // w_q det(W) dmu/dx : dA/dx Winv
|
||||
AddMultABt(DS, P, PMatO);
|
||||
|
||||
if (exact_action)
|
||||
{
|
||||
el.CalcShape(ip, shape);
|
||||
// Derivatives of adaptivity-based targets.
|
||||
// First term: w_q d*(Det W)/dx * mu(T)
|
||||
// d(Det W)/dx = det(W)*Tr[Winv*dW/dx]
|
||||
DenseMatrix dwdx(dim);
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
const DenseMatrix &dJtr_q = dJtr(q + d*ir->GetNPoints());
|
||||
Mult(Jrt, dJtr_q, dwdx );
|
||||
d_detW_dx(d) = dwdx.Trace();
|
||||
}
|
||||
d_detW_dx *= weight_m*metric->EvalW(Jpt); // *[w_q*det(W)]*mu(T)
|
||||
|
||||
// Second term: w_q det(W) dmu/dx : AdWinv/dx
|
||||
// dWinv/dx = -Winv*dW/dx*Winv
|
||||
MultAtB(PMatI, DSh, Amat);
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
const DenseMatrix &dJtr_q = dJtr(q + d*nqp);
|
||||
Mult(Jrt, dJtr_q, work1); // Winv*dw/dx
|
||||
Mult(work1, Jrt, work2); // Winv*dw/dx*Winv
|
||||
Mult(Amat, work2, work1); // A*Winv*dw/dx*Winv
|
||||
MultAtB(P, work1, work2); // dmu/dT^T*A*Winv*dw/dx*Winv
|
||||
d_Winv_dx(d) = work2.Trace(); // Tr[dmu/dT : AWinv*dw/dx*Winv]
|
||||
}
|
||||
d_Winv_dx *= -weight_m; // Include (-) factor as well
|
||||
|
||||
d_detW_dx += d_Winv_dx;
|
||||
AddMultVWt(shape, d_detW_dx, PMatO);
|
||||
}
|
||||
// TODO: derivatives of adaptivity-based targets.
|
||||
|
||||
if (coeff0)
|
||||
{
|
||||
if (!exact_action) { el.CalcShape(ip, shape); }
|
||||
el.CalcShape(ip, shape);
|
||||
PMatI.MultTranspose(shape, p);
|
||||
pos0.MultTranspose(shape, p0);
|
||||
lim_func->Eval_d1(p, p0, d_vals(q), grad);
|
||||
@@ -2666,8 +2214,7 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
|
||||
Jpt.SetSize(dim);
|
||||
|
||||
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
|
||||
const int nqp = ir->GetNPoints();
|
||||
DenseTensor Jtr(dim, dim, nqp);
|
||||
DenseTensor Jtr(dim, dim, ir->GetNPoints());
|
||||
|
||||
metric_energy = 0.0;
|
||||
lim_energy = 0.0;
|
||||
@@ -2680,12 +2227,12 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
|
||||
|
||||
targetC->ComputeElementTargets(i, *fe, *ir, x_vals, Jtr);
|
||||
|
||||
for (int q = 0; q < nqp; q++)
|
||||
for (int i = 0; i < ir->GetNPoints(); i++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(q);
|
||||
metric->SetTargetJacobian(Jtr(q));
|
||||
CalcInverse(Jtr(q), Jrt);
|
||||
const double weight = ip.weight * Jtr(q).Det();
|
||||
const IntegrationPoint &ip = ir->IntPoint(i);
|
||||
metric->SetTargetJacobian(Jtr(i));
|
||||
CalcInverse(Jtr(i), Jrt);
|
||||
const double weight = ip.weight * Jtr(i).Det();
|
||||
|
||||
fe->CalcDShape(ip, DSh);
|
||||
MultAtB(PMatI, DSh, Jpr);
|
||||
@@ -2737,8 +2284,6 @@ void TMOP_Integrator::ComputeMinJac(const Vector &x,
|
||||
|
||||
void TMOP_Integrator::UpdateAfterMeshChange(const Vector &new_x)
|
||||
{
|
||||
PA.setup_Jtr = false;
|
||||
PA.setup_Grad = false;
|
||||
// Update zeta if adaptive limiting is enabled.
|
||||
if (zeta) { adapt_eval->ComputeAtNewPosition(new_x, *zeta); }
|
||||
}
|
||||
@@ -2894,49 +2439,6 @@ void TMOPComboIntegrator::ParEnableNormalization(const ParGridFunction &x)
|
||||
}
|
||||
#endif
|
||||
|
||||
void TMOPComboIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AssemblePA(fes);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AssembleGradientDiagonalPA(const Vector &xe,
|
||||
Vector &de) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AssembleGradientDiagonalPA(xe, de);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AddMultPA(const Vector &xe, Vector &ye) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AddMultPA(xe, ye);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOPComboIntegrator::AddMultGradPA(const Vector &xe, const Vector &re,
|
||||
Vector &ce) const
|
||||
{
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
tmopi[i]->AddMultGradPA(xe, re, ce);
|
||||
}
|
||||
}
|
||||
|
||||
double TMOPComboIntegrator::GetGridFunctionEnergyPA(const Vector &xe) const
|
||||
{
|
||||
double energy = 0.0;
|
||||
for (int i = 0; i < tmopi.Size(); i++)
|
||||
{
|
||||
energy += tmopi[i]->GetGridFunctionEnergyPA(xe);
|
||||
}
|
||||
return energy;
|
||||
}
|
||||
|
||||
void InterpolateTMOP_QualityMetric(TMOP_QualityMetric &metric,
|
||||
const TargetConstructor &tc,
|
||||
|
||||
+11
-178
@@ -68,10 +68,6 @@ public:
|
||||
*/
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const = 0;
|
||||
|
||||
/** @brief Return the metric ID.
|
||||
*/
|
||||
virtual int Id() const { return 0; }
|
||||
};
|
||||
|
||||
|
||||
@@ -89,8 +85,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 1; }
|
||||
};
|
||||
|
||||
/// Skew metric, 2D.
|
||||
@@ -182,8 +176,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 2; }
|
||||
};
|
||||
|
||||
/// Shape & area, ideal barrier metric, 2D
|
||||
@@ -200,8 +192,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 7; }
|
||||
};
|
||||
|
||||
/// Shape & area metric, 2D
|
||||
@@ -288,6 +278,7 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
};
|
||||
|
||||
/// Shape, ideal barrier metric, 2D
|
||||
@@ -305,6 +296,7 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
};
|
||||
|
||||
/// Area, ideal barrier metric, 2D
|
||||
@@ -322,7 +314,6 @@ public:
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 77; }
|
||||
};
|
||||
|
||||
/// Shape & orientation metric, 2D.
|
||||
@@ -409,8 +400,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 302; }
|
||||
};
|
||||
|
||||
/// Shape, ideal barrier metric, 3D
|
||||
@@ -427,8 +416,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 303; }
|
||||
};
|
||||
|
||||
/// Volume metric, 3D
|
||||
@@ -445,8 +432,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 315; }
|
||||
};
|
||||
|
||||
/// Volume, ideal barrier metric, 3D
|
||||
@@ -481,8 +466,6 @@ public:
|
||||
|
||||
virtual void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
|
||||
const double weight, DenseMatrix &A) const;
|
||||
|
||||
virtual int Id() const { return 321; }
|
||||
};
|
||||
|
||||
/// Shifted barrier form of 3D metric 16 (volume, ideal barrier metric), 3D
|
||||
@@ -606,8 +589,6 @@ public:
|
||||
|
||||
virtual void ComputeAtNewPosition(const Vector &new_nodes,
|
||||
Vector &new_field) = 0;
|
||||
|
||||
void ClearGeometricFactors();
|
||||
};
|
||||
|
||||
/** @brief Base class representing target-matrix construction algorithms for
|
||||
@@ -683,14 +664,9 @@ public:
|
||||
nodes are used by all target types except IDEAL_SHAPE_UNIT_SIZE. */
|
||||
void SetNodes(const GridFunction &n) { nodes = &n; avg_volume = 0.0; }
|
||||
|
||||
/** @brief Get the nodes to be used in the target-matrix construction. */
|
||||
const GridFunction *GetNodes() const { return nodes; }
|
||||
|
||||
/// Used by target type IDEAL_SHAPE_EQUAL_SIZE. The default volume scale is 1.
|
||||
void SetVolumeScale(double vol_scale) { volume_scale = vol_scale; }
|
||||
|
||||
const TargetType &Type() const { return target_type; }
|
||||
|
||||
/// Checks if the target matrices contain non-trivial size specification.
|
||||
virtual bool ContainsVolumeInfo() const;
|
||||
|
||||
@@ -701,35 +677,6 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
template<int DIM>
|
||||
bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
class TMOPMatrixCoefficient : public MatrixCoefficient
|
||||
{
|
||||
public:
|
||||
explicit TMOPMatrixCoefficient(int dim) : MatrixCoefficient(dim, dim) { }
|
||||
|
||||
/** @brief Evaluate the derivative of the matrix coefficient with respect to
|
||||
@a comp in the element described by @a T at the point @a ip, storing the
|
||||
result in @a K. */
|
||||
virtual void EvalGrad(DenseMatrix &K, ElementTransformation &T,
|
||||
const IntegrationPoint &ip, int comp) = 0;
|
||||
|
||||
virtual ~TMOPMatrixCoefficient() { }
|
||||
};
|
||||
|
||||
class AnalyticAdaptTC : public TargetConstructor
|
||||
@@ -738,7 +685,7 @@ protected:
|
||||
// Analytic target specification.
|
||||
Coefficient *scalar_tspec;
|
||||
VectorCoefficient *vector_tspec;
|
||||
TMOPMatrixCoefficient *matrix_tspec;
|
||||
MatrixCoefficient *matrix_tspec;
|
||||
|
||||
public:
|
||||
AnalyticAdaptTC(TargetType ttype)
|
||||
@@ -747,7 +694,7 @@ public:
|
||||
|
||||
virtual void SetAnalyticTargetSpec(Coefficient *sspec,
|
||||
VectorCoefficient *vspec,
|
||||
TMOPMatrixCoefficient *mspec);
|
||||
MatrixCoefficient *mspec);
|
||||
|
||||
/** @brief Given an element and quadrature rule, computes ref->target
|
||||
transformation Jacobians for each quadrature point in the element.
|
||||
@@ -756,16 +703,6 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
@@ -787,36 +724,25 @@ protected:
|
||||
// eta1(x+h,y), eta2(x+h,y) ... etan(x+h,y), eta1(x,y+h), eta2(x,y+h) ...
|
||||
// same for tspec_pert2h and tspec_pertmix.
|
||||
|
||||
// Components of Target Jacobian at each quadrature point of an element. This
|
||||
// is required for computation of the derivative using chain rule.
|
||||
mutable DenseTensor Jtrcomp;
|
||||
|
||||
// Note: do not use the Nodes of this space as they may not be on the
|
||||
// positions corresponding to the values of tspec.
|
||||
const FiniteElementSpace *tspec_fes;
|
||||
const FiniteElementSpace *tspec_fesv;
|
||||
|
||||
// These flags can be used by outside functions to avoid recomputing the
|
||||
// tspec and tspec_perth fields again on the same mesh.
|
||||
// These flags can be used by outside functions to avoid recomputing
|
||||
// the tspec and tspec_perth fields again on the same mesh.
|
||||
bool good_tspec, good_tspec_grad, good_tspec_hess;
|
||||
|
||||
// Evaluation of the discrete target specification on different meshes.
|
||||
// Owned.
|
||||
AdaptivityEvaluator *adapt_eval;
|
||||
|
||||
// PA extension
|
||||
struct { mutable Vector tspec_e; } PA;
|
||||
|
||||
void FinalizeSerialDiscreteTargetSpec();
|
||||
#ifdef MFEM_USE_MPI
|
||||
void FinalizeParDiscreteTargetSpec(const ParGridFunction &tspec_);
|
||||
#endif
|
||||
|
||||
public: // MFEM_FORALL nvcc restriction that it must be public
|
||||
void SetDiscreteTargetBase(const GridFunction &tspec_);
|
||||
void SetTspecAtIndex(int idx, const GridFunction &tspec_);
|
||||
void FinalizeSerialDiscreteTargetSpec();
|
||||
#ifdef MFEM_USE_MPI
|
||||
void SetTspecAtIndex(int idx, const ParGridFunction &tspec_);
|
||||
void FinalizeParDiscreteTargetSpec(const ParGridFunction &tspec_);
|
||||
#endif
|
||||
|
||||
public:
|
||||
@@ -901,7 +827,6 @@ public:
|
||||
const Vector &GetTspecPert1H() { return tspec_pert1h; }
|
||||
const Vector &GetTspecPert2H() { return tspec_pert2h; }
|
||||
const Vector &GetTspecPertMixH() { return tspec_pertmix; }
|
||||
const FiniteElementSpace *GetTspecFesv() const { return tspec_fesv; }
|
||||
|
||||
/** @brief Given an element and quadrature rule, computes ref->target
|
||||
transformation Jacobians for each quadrature point in the element.
|
||||
@@ -912,16 +837,6 @@ public:
|
||||
const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
DenseTensor &Jtr) const;
|
||||
|
||||
virtual bool ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe = Vector()) const;
|
||||
|
||||
virtual void ComputeElementTargetsGradient(const IntegrationRule &ir,
|
||||
const Vector &elfun,
|
||||
IsoparametricTransformation &Tpr,
|
||||
DenseTensor &dJtr) const;
|
||||
};
|
||||
|
||||
class TMOPNewtonSolver;
|
||||
@@ -974,9 +889,6 @@ protected:
|
||||
// Specifies that ComputeElementTargets is being called by a FD function.
|
||||
// It's used to skip terms that have exact derivative calculations.
|
||||
bool fd_call_flag;
|
||||
// Compute the exact action of the Integrator (includes derivative of the
|
||||
// target with respect to spatial position)
|
||||
bool exact_action;
|
||||
|
||||
Array <Vector *> ElemDer; //f'(x)
|
||||
Array <Vector *> ElemPertEnergy; //f(x+h)
|
||||
@@ -993,25 +905,10 @@ protected:
|
||||
// output - the result of AssembleElementVector() (dof x dim).
|
||||
DenseMatrix DSh, DS, Jrt, Jpr, Jpt, P, PMatI, PMatO;
|
||||
|
||||
// PA extension
|
||||
struct
|
||||
{
|
||||
bool enabled;
|
||||
int dim, ne, nq;
|
||||
mutable DenseTensor Jtr;
|
||||
mutable bool setup_Grad, setup_Jtr;
|
||||
mutable Vector E, O, W, X0, H, C0, LD, H0;
|
||||
const DofToQuad *maps;
|
||||
const DofToQuad *maps_lim = nullptr;
|
||||
const GeometricFactors *geom;
|
||||
const FiniteElementSpace *fes;
|
||||
const Operator *R;
|
||||
const IntegrationRule *ir;
|
||||
} PA;
|
||||
|
||||
void ComputeNormalizationEnergies(const GridFunction &x,
|
||||
double &metric_energy, double &lim_energy);
|
||||
|
||||
|
||||
void AssembleElementVectorExact(const FiniteElement &el,
|
||||
ElementTransformation &T,
|
||||
const Vector &elfun, Vector &elvect);
|
||||
@@ -1081,8 +978,8 @@ public:
|
||||
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
|
||||
zeta_0(NULL), zeta(NULL), coeff_zeta(NULL), adapt_eval(NULL),
|
||||
discr_tc(dynamic_cast<DiscreteAdaptTC *>(tc)),
|
||||
fdflag(false), dxscale(1.0e3), fd_call_flag(false), exact_action(false)
|
||||
{ PA.enabled = false; }
|
||||
fdflag(false), dxscale(1.0e3), fd_call_flag(false)
|
||||
{ }
|
||||
|
||||
~TMOP_Integrator();
|
||||
|
||||
@@ -1150,45 +1047,6 @@ public:
|
||||
virtual void AssembleElementGrad(const FiniteElement &el,
|
||||
ElementTransformation &T,
|
||||
const Vector &elfun, DenseMatrix &elmat);
|
||||
/// PA extension
|
||||
void SetupGradPA(const Vector &xe) const;
|
||||
void EnableLimitingPA(const GridFunction &n0);
|
||||
void ComputeElementTargetsPA(const Vector &xe = Vector()) const;
|
||||
|
||||
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
|
||||
double GetGridFunctionEnergyPA_2D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_C0_2D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_3D(const Vector&) const;
|
||||
double GetGridFunctionEnergyPA_C0_3D(const Vector&) const;
|
||||
virtual double GetGridFunctionEnergyPA(const Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace&);
|
||||
|
||||
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
|
||||
void AssembleDiagonalPA_2D(Vector&) const;
|
||||
void AssembleDiagonalPA_3D(Vector&) const;
|
||||
void AssembleDiagonalPA_C0_2D(Vector&) const;
|
||||
void AssembleDiagonalPA_C0_3D(Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AddMultPA;
|
||||
void AddMultPA_2D(const Vector&, Vector&) const;
|
||||
void AddMultPA_3D(const Vector&, Vector&) const;
|
||||
void AddMultPA_C0_2D(const Vector&, Vector&) const;
|
||||
void AddMultPA_C0_3D(const Vector&, Vector&) const;
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
|
||||
using NonlinearFormIntegrator::AddMultGradPA;
|
||||
void AddMultGradPA_2D(const Vector&, Vector&) const;
|
||||
void AddMultGradPA_3D(const Vector&, const Vector&, Vector&) const;
|
||||
void AddMultGradPA_C0_2D(const Vector&, const Vector&, Vector&) const;
|
||||
void AddMultGradPA_C0_3D(const Vector&, const Vector&, Vector&) const;
|
||||
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
|
||||
|
||||
void AssembleGradPA_2D(const Vector&) const;
|
||||
void AssembleGradPA_3D(const Vector&) const;
|
||||
void AssembleGradPA_C0_2D(const Vector&) const;
|
||||
void AssembleGradPA_C0_3D(const Vector&) const;
|
||||
|
||||
DiscreteAdaptTC *GetDiscreteAdaptTC() const { return discr_tc; }
|
||||
|
||||
@@ -1208,20 +1066,6 @@ public:
|
||||
void SetFDhScale(double _dxscale) { dxscale = _dxscale; }
|
||||
bool GetFDFlag() const { return fdflag; }
|
||||
double GetFDh() const { return dx; }
|
||||
|
||||
/** @brief Flag to control if exact action of Integration is effected. */
|
||||
void SetExactActionFlag(bool flag_) { exact_action = flag_; }
|
||||
|
||||
void ReleaseTemporaryMemory()
|
||||
{
|
||||
if (PA.enabled)
|
||||
{
|
||||
PA.H.GetMemory().DeleteDevice();
|
||||
PA.H0.GetMemory().DeleteDevice();
|
||||
//PA.Jtr.GetMemory().DeleteDevice();
|
||||
//PA.setup_Jtr = false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
class TMOPComboIntegrator : public NonlinearFormIntegrator
|
||||
@@ -1270,17 +1114,6 @@ public:
|
||||
#ifdef MFEM_USE_MPI
|
||||
void ParEnableNormalization(const ParGridFunction &x);
|
||||
#endif
|
||||
|
||||
/// PA extension
|
||||
using NonlinearFormIntegrator::AssemblePA;
|
||||
virtual void AssemblePA(const FiniteElementSpace&);
|
||||
virtual void AssembleGradientDiagonalPA(const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::AddMultPA;
|
||||
virtual void AddMultPA(const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::AddMultGradPA;
|
||||
virtual void AddMultGradPA(const Vector&, const Vector&, Vector&) const;
|
||||
using NonlinearFormIntegrator::GetGridFunctionEnergyPA;
|
||||
virtual double GetGridFunctionEnergyPA(const Vector&) const;
|
||||
};
|
||||
|
||||
/// Interpolates the @a metric's values at the nodes of @a metric_gf.
|
||||
|
||||
-363
@@ -1,363 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "pgridfunc.hpp"
|
||||
#include "tmop_tools.hpp"
|
||||
#include "quadinterpolator.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
|
||||
void TMOP_Integrator::SetupGradPA(const Vector &xe) const
|
||||
{
|
||||
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
|
||||
PA.setup_Grad = true;
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AssembleGradPA_2D(xe);
|
||||
if (coeff0) { AssembleGradPA_C0_2D(xe); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AssembleGradPA_3D(xe);
|
||||
if (coeff0) { AssembleGradPA_C0_3D(xe); }
|
||||
}
|
||||
}
|
||||
|
||||
// We might come here w/o knowing that PA will be used.
|
||||
// It is the case when EnableLimiting is called before the Setup => AssemblePA.
|
||||
void TMOP_Integrator::EnableLimitingPA(const GridFunction &n0)
|
||||
{
|
||||
MFEM_VERIFY(PA.enabled, "EnableLimitingPA but PA is not enabled!");
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
|
||||
// Nodes0
|
||||
const FiniteElementSpace *n0_fes = n0.FESpace();
|
||||
const Operator *n0_R = n0_fes->GetElementRestriction(ordering);
|
||||
PA.X0.SetSize(n0_R->Height(), Device::GetMemoryType());
|
||||
PA.X0.UseDevice(true);
|
||||
n0_R->Mult(n0, PA.X0);
|
||||
|
||||
// Get the 1D maps for the distance FE space.
|
||||
const IntegrationRule &ir = *EnergyIntegrationRule(*n0.FESpace()->GetFE(0));
|
||||
PA.maps_lim =
|
||||
&lim_dist->FESpace()->GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
|
||||
// lim_dist & lim_func checks
|
||||
MFEM_VERIFY(lim_dist, "No lim_dist!")
|
||||
const FiniteElementSpace *ld_fes = lim_dist->FESpace();
|
||||
const Operator *ld_R = ld_fes->GetElementRestriction(ordering);
|
||||
MFEM_VERIFY(ld_R, "No lim_dist restriction operator found!");
|
||||
PA.LD.SetSize(ld_R->Height(), Device::GetMemoryType());
|
||||
PA.LD.UseDevice(true);
|
||||
ld_R->Mult(*lim_dist, PA.LD);
|
||||
|
||||
// Only TMOP_QuadraticLimiter is supported
|
||||
MFEM_VERIFY(lim_func, "No lim_func!")
|
||||
MFEM_VERIFY(dynamic_cast<TMOP_QuadraticLimiter*>(lim_func),
|
||||
"Only TMOP_QuadraticLimiter is supported");
|
||||
}
|
||||
|
||||
bool TargetConstructor::ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe) const
|
||||
{
|
||||
MFEM_VERIFY(Jtr.SizeI() == Jtr.SizeJ() && Jtr.SizeI() > 1, "");
|
||||
const int dim = Jtr.SizeI();
|
||||
if (dim == 2) { return ComputeElementTargetsPA<2>(fes, ir, Jtr, xe); }
|
||||
if (dim == 3) { return ComputeElementTargetsPA<3>(fes, ir, Jtr, xe); }
|
||||
return false;
|
||||
}
|
||||
|
||||
bool AnalyticAdaptTC::ComputeElementTargetsPA(const FiniteElementSpace *fes,
|
||||
const IntegrationRule *ir,
|
||||
DenseTensor &Jtr,
|
||||
const Vector &xe) const
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Code paths leading to ComputeElementTargets:
|
||||
// - GetElementEnergy(elfun) which is done through GetGridFunctionEnergyPA(x)
|
||||
// - AssembleElementVectorExact(elfun)
|
||||
// - AssembleElementGradExact(elfun)
|
||||
// - EnableNormalization(x) -> ComputeNormalizationEnergies(x)
|
||||
// - (AssembleElementVectorFD(elfun))
|
||||
// - (AssembleElementGradFD(elfun))
|
||||
// ============================================================================
|
||||
// - TargetConstructor():
|
||||
// - IDEAL_SHAPE_UNIT_SIZE: Wideal
|
||||
// - IDEAL_SHAPE_EQUAL_SIZE: α * Wideal
|
||||
// - IDEAL_SHAPE_GIVEN_SIZE: β * Wideal
|
||||
// - GIVEN_SHAPE_AND_SIZE: β * Wideal
|
||||
// - AnalyticAdaptTC(elfun):
|
||||
// - GIVEN_FULL: matrix_tspec->Eval(Jtr(elfun))
|
||||
// - DiscreteAdaptTC():
|
||||
// - IDEAL_SHAPE_GIVEN_SIZE: size^{1.0/dim} * Jtr(i) (size)
|
||||
// - GIVEN_SHAPE_AND_SIZE: Jtr(i) *= D_rho (ratio)
|
||||
// Jtr(i) *= Q_phi (skew)
|
||||
// Jtr(i) *= R_theta (orientation)
|
||||
void TMOP_Integrator::ComputeElementTargetsPA(const Vector &xe) const
|
||||
{
|
||||
PA.setup_Jtr = false;
|
||||
const FiniteElementSpace *fes = PA.fes;
|
||||
const IntegrationRule *ir = EnergyIntegrationRule(*fes->GetFE(0));
|
||||
const TargetConstructor::TargetType &target_type = targetC->Type();
|
||||
const DiscreteAdaptTC *discr_tc = GetDiscreteAdaptTC();
|
||||
|
||||
// Skip when TargetConstructor needs the nodes but have not been set
|
||||
const bool use_nodes =
|
||||
target_type == TargetConstructor::IDEAL_SHAPE_EQUAL_SIZE ||
|
||||
target_type == TargetConstructor::IDEAL_SHAPE_GIVEN_SIZE ||
|
||||
target_type == TargetConstructor::GIVEN_SHAPE_AND_SIZE;
|
||||
if (targetC && !discr_tc && use_nodes && !targetC->GetNodes()) { return; }
|
||||
|
||||
// Try to use the TargetConstructor ComputeElementTargetsPA
|
||||
PA.setup_Jtr = targetC->ComputeElementTargetsPA(fes, ir, PA.Jtr);
|
||||
if (PA.setup_Jtr) { return; }
|
||||
|
||||
// Defaulting to host version
|
||||
PA.Jtr.HostWrite();
|
||||
|
||||
const int NE = PA.ne;
|
||||
const int NQ = PA.nq;
|
||||
const int dim = PA.dim;
|
||||
DenseTensor &Jtr = PA.Jtr;
|
||||
|
||||
Vector x;
|
||||
const bool useable_input_vector = xe.Size() > 0;
|
||||
const bool use_input_vector = target_type == TargetConstructor::GIVEN_FULL;
|
||||
|
||||
if (use_input_vector && !useable_input_vector) { return; }
|
||||
|
||||
if (discr_tc && !discr_tc->GetTspecFesv()) { return; }
|
||||
|
||||
if (use_input_vector)
|
||||
{
|
||||
x.SetSize(PA.R->Width(), Device::GetMemoryType());
|
||||
x.UseDevice(true);
|
||||
PA.R->MultTranspose(xe, x);
|
||||
// Scale by weights
|
||||
const int N = PA.W.Size();
|
||||
const auto W = Reshape(PA.W.Read(), N);
|
||||
auto X = Reshape(x.ReadWrite(), N);
|
||||
MFEM_FORALL(i, N, X(i) /= W(i););
|
||||
}
|
||||
|
||||
// Use TargetConstructor::ComputeElementTargets to fill the PA.Jtr
|
||||
Vector elfun;
|
||||
Array<int> vdofs;
|
||||
DenseTensor J;
|
||||
for (int e = 0; e < NE; e++)
|
||||
{
|
||||
const FiniteElement &fe = *fes->GetFE(e);
|
||||
if (use_input_vector)
|
||||
{
|
||||
fes->GetElementVDofs(e, vdofs);
|
||||
x.GetSubVector(vdofs, elfun);
|
||||
}
|
||||
J.UseExternalData(Jtr(e*NQ).Data(), dim, dim, NQ);
|
||||
targetC->ComputeElementTargets(e, fe, *ir, elfun, J);
|
||||
}
|
||||
PA.setup_Jtr = true;
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
PA.enabled = true;
|
||||
MFEM_ASSERT(fes.GetMesh()->GetNE() > 0, "");
|
||||
PA.ir = EnergyIntegrationRule(*fes.GetFE(0));
|
||||
const IntegrationRule *ir = PA.ir;
|
||||
MFEM_ASSERT(fes.GetOrdering() == Ordering::byNODES,
|
||||
"PA Only supports Ordering::byNODES!");
|
||||
|
||||
PA.fes = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const int nq = PA.nq = ir->GetNPoints();
|
||||
const int ne = PA.ne = fes.GetMesh()->GetNE();
|
||||
const int dim = PA.dim = mesh->Dimension();
|
||||
MFEM_VERIFY(PA.dim == 2 || PA.dim == 3, "Not yet implemented!");
|
||||
|
||||
const DofToQuad::Mode mode = DofToQuad::TENSOR;
|
||||
PA.maps = &fes.GetFE(0)->GetDofToQuad(*ir, mode);
|
||||
PA.geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mode);
|
||||
|
||||
// Energy vector
|
||||
PA.E.UseDevice(true);
|
||||
PA.E.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
|
||||
// Setup initialization
|
||||
PA.setup_Jtr = false;
|
||||
PA.setup_Grad = false;
|
||||
|
||||
|
||||
#ifdef MFEM_USE_UMPIRE
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType() == MemoryType::DEVICE_UMPIRE
|
||||
? MemoryType::DEVICE_UMPIRE_2 : Device::GetDeviceMemoryType();
|
||||
#else
|
||||
const MemoryType temp_type = Device::GetDeviceMemoryType();
|
||||
#endif
|
||||
|
||||
// H for Grad
|
||||
PA.H.SetSize(dim*dim * dim*dim * nq*ne, temp_type);
|
||||
// H0 for coeff0
|
||||
PA.H0.SetSize(dim * dim * nq*ne, temp_type);
|
||||
|
||||
// Restriction setup
|
||||
const ElementDofOrdering ordering = ElementDofOrdering::LEXICOGRAPHIC;
|
||||
PA.R = fes.GetElementRestriction(ordering);
|
||||
MFEM_VERIFY(PA.R, "Not yet implemented!");
|
||||
|
||||
// Weight of the R^t
|
||||
PA.W.SetSize(PA.R->Width(), Device::GetDeviceMemoryType());
|
||||
PA.W.UseDevice(true);
|
||||
PA.O.SetSize(dim*ne*nq, Device::GetDeviceMemoryType());
|
||||
PA.O.UseDevice(true);
|
||||
PA.O = 1.0;
|
||||
PA.R->MultTranspose(PA.O, PA.W);
|
||||
|
||||
// Scalar vector of '1'
|
||||
PA.O.SetSize(ne*nq, Device::GetDeviceMemoryType());
|
||||
PA.O = 1.0;
|
||||
|
||||
// TargetConstructor TargetType setup
|
||||
PA.Jtr.SetSize(dim, dim, PA.ne*PA.nq);//, temp_type);
|
||||
ComputeElementTargetsPA();
|
||||
|
||||
// Coeff0 PA.C0
|
||||
PA.C0.UseDevice(true);
|
||||
if (coeff0 == nullptr)
|
||||
{
|
||||
PA.C0.SetSize(1, Device::GetMemoryType());
|
||||
PA.C0.HostWrite();
|
||||
PA.C0(0) = 0.0;
|
||||
}
|
||||
else if (ConstantCoefficient* cQ =
|
||||
dynamic_cast<ConstantCoefficient*>(coeff0))
|
||||
{
|
||||
PA.C0.SetSize(1, Device::GetMemoryType());
|
||||
PA.C0.HostWrite();
|
||||
PA.C0(0) = cQ->constant;
|
||||
}
|
||||
else
|
||||
{
|
||||
PA.C0.SetSize(PA.nq * PA.ne, Device::GetMemoryType());
|
||||
auto C0 = Reshape(PA.C0.HostWrite(), PA.nq, PA.ne);
|
||||
for (int e = 0; e < ne; ++e)
|
||||
{
|
||||
ElementTransformation& T = *fes.GetElementTransformation(e);
|
||||
for (int q = 0; q < nq; ++q)
|
||||
{
|
||||
C0(q,e) = coeff0->Eval(T, ir->IntPoint(q));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (coeff0)
|
||||
{
|
||||
MFEM_VERIFY(nodes0, "nodes0 has not been set!");
|
||||
EnableLimitingPA(*nodes0);
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradientDiagonalPA(const Vector &xe,
|
||||
Vector &de) const
|
||||
{
|
||||
MFEM_VERIFY(PA.R, "PA extension setup has not been done!");
|
||||
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
|
||||
|
||||
if (!PA.setup_Grad) { SetupGradPA(xe); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AssembleDiagonalPA_2D(de);
|
||||
if (coeff0) { AssembleDiagonalPA_C0_2D(de); }
|
||||
}
|
||||
else if (PA.dim == 3)
|
||||
{
|
||||
AssembleDiagonalPA_3D(de);
|
||||
if (coeff0) { AssembleDiagonalPA_C0_3D(de); }
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("3D diagonal computation is WIP.");
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultPA(const Vector &xe, Vector &ye) const
|
||||
{
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AddMultPA_2D(xe,ye);
|
||||
if (coeff0) { AddMultPA_C0_2D(xe,ye); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AddMultPA_3D(xe,ye);
|
||||
if (coeff0) { AddMultPA_C0_3D(xe,ye); }
|
||||
}
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA(const Vector &xe,
|
||||
const Vector &re, Vector &ce) const
|
||||
{
|
||||
if (!PA.setup_Jtr) { ComputeElementTargetsPA(xe); }
|
||||
|
||||
if (!PA.setup_Grad) { SetupGradPA(xe); }
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
AddMultGradPA_2D(re,ce);
|
||||
if (coeff0) { AddMultGradPA_C0_2D(xe,re,ce); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
AddMultGradPA_3D(xe,re,ce);
|
||||
if (coeff0) { AddMultGradPA_C0_3D(xe,re,ce); }
|
||||
}
|
||||
}
|
||||
|
||||
double TMOP_Integrator::GetGridFunctionEnergyPA(const Vector &xe) const
|
||||
{
|
||||
double energy = 0.0;
|
||||
|
||||
ComputeElementTargetsPA(xe);
|
||||
|
||||
if (PA.dim == 2)
|
||||
{
|
||||
energy = GetGridFunctionEnergyPA_2D(xe);
|
||||
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_2D(xe); }
|
||||
}
|
||||
|
||||
if (PA.dim == 3)
|
||||
{
|
||||
energy = GetGridFunctionEnergyPA_3D(xe);
|
||||
if (coeff0) { energy += GetGridFunctionEnergyPA_C0_3D(xe); }
|
||||
}
|
||||
|
||||
return energy;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
-143
@@ -1,143 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_TMOP_PA_HPP
|
||||
#define MFEM_TMOP_PA_HPP
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
#include "../fem/kernels.hpp"
|
||||
|
||||
#include <unordered_map>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace kernels
|
||||
{
|
||||
|
||||
/// Generic emplace
|
||||
template<typename K, const int N,
|
||||
typename Key_t = typename K::Key_t,
|
||||
typename Kernel_t = typename K::Kernel_t>
|
||||
void emplace(std::unordered_map<Key_t, Kernel_t> &map)
|
||||
{
|
||||
constexpr Key_t key = K::template GetKey<N>();
|
||||
constexpr Kernel_t value = K::template GetValue<key>();
|
||||
map.emplace(key, value);
|
||||
}
|
||||
|
||||
/// Instances
|
||||
template<class K, typename T, T... idx>
|
||||
struct instances
|
||||
{
|
||||
static void Fill(std::unordered_map<typename K::Key_t,
|
||||
typename K::Kernel_t> &map)
|
||||
{
|
||||
using unused = int[];
|
||||
(void) unused {0, (emplace<K,idx>(map), 0)... };
|
||||
}
|
||||
};
|
||||
|
||||
/// Cat instances
|
||||
template<class K, typename Offset, typename Lhs, typename Rhs> struct cat;
|
||||
template<class K, typename T, T Offset, T... Lhs, T... Rhs>
|
||||
struct cat<K, std::integral_constant<T, Offset>,
|
||||
instances<K, T, Lhs...>,
|
||||
instances<K, T, Rhs...> >
|
||||
{ using type = instances<K, T, Lhs..., (Offset + Rhs)...>; };
|
||||
|
||||
/// Sequence, empty and one element terminal cases
|
||||
template<class K, typename T, typename N>
|
||||
struct sequence
|
||||
{
|
||||
using Lhs = std::integral_constant<T, N::value/2>;
|
||||
using Rhs = std::integral_constant<T, N::value-Lhs::value>;
|
||||
using type = typename cat<K, Lhs,
|
||||
typename sequence<K, T, Lhs>::type,
|
||||
typename sequence<K, T, Rhs>::type>::type;
|
||||
};
|
||||
|
||||
template<class K, typename T>
|
||||
struct sequence<K, T, std::integral_constant<T,0> >
|
||||
{ using type = instances<K,T>; };
|
||||
|
||||
template<class K, typename T>
|
||||
struct sequence<K, T, std::integral_constant<T,1> >
|
||||
{ using type = instances<K,T,0>; };
|
||||
|
||||
/// Make_sequence
|
||||
template<class Instance, typename T = typename Instance::Key_t>
|
||||
using make_sequence =
|
||||
typename sequence<Instance, T, std::integral_constant<T,Instance::N> >::type;
|
||||
|
||||
/// Instantiator class
|
||||
template<class Instance,
|
||||
typename Key_t = typename Instance::Key_t,
|
||||
typename Return_t = typename Instance::Return_t,
|
||||
typename Kernel_t = typename Instance::Kernel_t>
|
||||
class Instantiator
|
||||
{
|
||||
private:
|
||||
using map_t = std::unordered_map<Key_t, Kernel_t>;
|
||||
map_t map;
|
||||
|
||||
public:
|
||||
Instantiator() { make_sequence<Instance>().Fill(map); }
|
||||
|
||||
bool Find(const Key_t id)
|
||||
{
|
||||
return (map.find(id) != map.end()) ? true : false;
|
||||
}
|
||||
|
||||
Kernel_t At(const Key_t id) { return map.at(id); }
|
||||
};
|
||||
|
||||
/// MFEM_REGISTER_TMOP_KERNELS macro:
|
||||
/// - forward declaration of the kernel
|
||||
/// - kernel pointer declaration
|
||||
/// - struct K##name##_T definition
|
||||
/// - Instantiator definition
|
||||
/// - re-use kernel return type and name before its body
|
||||
#define MFEM_REGISTER_TMOP_KERNELS(return_t, kernel, ...) \
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_MAX = 0> \
|
||||
return_t kernel(__VA_ARGS__);\
|
||||
typedef return_t (*kernel##_p)(__VA_ARGS__);\
|
||||
struct K##kernel##_T {\
|
||||
static const int N = 14;\
|
||||
using Key_t = std::size_t;\
|
||||
using Kernel_t = kernel##_p;\
|
||||
using Return_t = return_t;\
|
||||
template<Key_t I> static constexpr Key_t GetKey() noexcept { return \
|
||||
I==0 ? 0x22 : I==1 ? 0x23 : I==2 ? 0x24 : I==3 ? 0x25 : I==4 ? 0x26 :\
|
||||
I==5 ? 0x33 : I==6 ? 0x34 : I==7 ? 0x35 : I==8 ? 0x36 :\
|
||||
I==9 ? 0x44 : I==10 ? 0x45 : I==11 ? 0x46 :\
|
||||
I==12 ? 0x55 : I==13 ? 0x56 : 0; }\
|
||||
template<Key_t ID> static constexpr Kernel_t GetValue() noexcept\
|
||||
{ return &kernel<(ID>>4)&0xF, ID&0xF>; }\
|
||||
};\
|
||||
static kernels::Instantiator<K##kernel##_T> K##kernel;\
|
||||
template<int T_D1D, int T_Q1D, int T_MAX> return_t kernel(__VA_ARGS__)
|
||||
|
||||
/// MFEM_LAUNCH_TMOP_KERNEL macro
|
||||
#define MFEM_LAUNCH_TMOP_KERNEL(kernel, id, ...)\
|
||||
if (K##kernel.Find(id)) { return K##kernel.At(id)(__VA_ARGS__,0,0); }\
|
||||
else {\
|
||||
constexpr int T_MAX = 4;\
|
||||
MFEM_VERIFY(D1D <= MAX_D1D && Q1D <= MAX_Q1D, "Max size error!");\
|
||||
return kernel<0,0,T_MAX>(__VA_ARGS__,D1D,Q1D); }
|
||||
|
||||
} // namespace kernels
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_TMOP_PA_HPP
|
||||
@@ -1,161 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/* // Original i-j assembly (old invariants code).
|
||||
for (int e = 0; e < NE; e++)
|
||||
{
|
||||
for (int q = 0; q < nqp; q++)
|
||||
{
|
||||
el.CalcDShape(ip, DSh);
|
||||
Mult(DSh, Jrt, DS);
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
for (int r = 0; r < dim; r++)
|
||||
{
|
||||
for (int c = 0; c < dim; c++)
|
||||
{
|
||||
for (int rr = 0; rr < dim; rr++)
|
||||
{
|
||||
for (int cc = 0; cc < dim; cc++)
|
||||
{
|
||||
const double H = h(r, c, rr, cc);
|
||||
A(e, i + r*dof, j + rr*dof) +=
|
||||
weight_q * DS(i, c) * DS(j, cc) * H;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}*/
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_2D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const DenseTensor &j,
|
||||
const Vector &h,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qd[DIM*DIM*MQ1*MD1];
|
||||
DeviceTensor<4,double> QD(qd, DIM, DIM, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD(0,0,qx,dy) = 0.0;
|
||||
QD(0,1,qx,dy) = 0.0;
|
||||
QD(1,0,qx,dy) = 0.0;
|
||||
QD(1,1,qx,dy) = 0.0;
|
||||
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double j[4];
|
||||
ConstDeviceMatrix Jrt(j,2,2);
|
||||
kernels::CalcInverse<2>(Jtr, j);
|
||||
|
||||
const double gg = G(qy,dy) * G(qy,dy);
|
||||
const double gb = G(qy,dy) * B(qy,dy);
|
||||
const double bb = B(qy,dy) * B(qy,dy);
|
||||
const double bgb[4] = { bb, gb, gb, gg };
|
||||
ConstDeviceMatrix BG(bgb,2,2);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
const double Jij = Jrt(i,i) * Jrt(j,j);
|
||||
const double alpha = Jij * BG(i,j);
|
||||
QD(i,j,qx,dy) += alpha * H(v,i,v,j,qx,qy,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double gg = G(qx,dx) * G(qx,dx);
|
||||
const double gb = G(qx,dx) * B(qx,dx);
|
||||
const double bb = B(qx,dx) * B(qx,dx);
|
||||
d += gg * QD(0,0,qx,dy);
|
||||
d += gb * QD(0,1,qx,dy);
|
||||
d += gb * QD(1,0,qx,dy);
|
||||
d += bb * QD(1,1,qx,dy);
|
||||
}
|
||||
D(dx,dy,v,e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_2D,id,N,B,G,J,H,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,96 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_2D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &h0,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qd[MQ1*MD1];
|
||||
DeviceTensor<2,double> QD(qd, MQ1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; v++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD(qx,dy) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double bb = B(qy,dy) * B(qy,dy);
|
||||
QD(qx,dy) += bb * H0(v,v,qx,qy,e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double bb = B(qx,dx) * B(qx,dx);
|
||||
d += bb * QD(qx,dy);
|
||||
}
|
||||
D(dx,dy,v,e) += d;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_2D,id,N,B,H0,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,128 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_2D,
|
||||
const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
const DenseTensor &j_,
|
||||
const Vector &h_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
|
||||
const auto H = Reshape(h_.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,XY);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^T.DSh
|
||||
double Jpr[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,QQ,Jpr);
|
||||
|
||||
// Jpt = Jpr . Jrt
|
||||
double Jpt[4];
|
||||
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
|
||||
|
||||
// B = Jpt : H
|
||||
double B[4];
|
||||
DeviceMatrix M(B,2,2);
|
||||
ConstDeviceMatrix J(Jpt,2,2);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
M(i,j) = 0.0;
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
M(i,j) += H(r,c,i,j,qx,qy,e) * J(r,c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// C = Jrt . B
|
||||
double C[4];
|
||||
kernels::MultABt(2,2,2, Jrt, B, C);
|
||||
|
||||
// Overwrite QQ = Jrt . (Jpt : H)^t
|
||||
kernels::PushGrad<MQ1,NBZ>(qx,qy, C, QQ);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::LoadBGt<MD1,MQ1>(D1D,Q1D,b,g,BG);
|
||||
kernels::GradYt<MD1,MQ1,NBZ>(D1D,Q1D,BG,QQ,DQ);
|
||||
kernels::GradXt<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,Y,e);
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_2D(const Vector &R, Vector &C) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_2D,id,N,B,G,J,H,R,C);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,107 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AddMultGradPA_Kernel_C0_2D,
|
||||
const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &h0_,
|
||||
const Vector &r_,
|
||||
Vector &c_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto H0 = Reshape(h0_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto R = Reshape(r_.Read(), D1D, D1D, DIM, NE);
|
||||
|
||||
auto Y = Reshape(c_.ReadWrite(), D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
|
||||
MFEM_SHARED double XY[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[2][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[2][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,R,XY);
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
|
||||
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,B,XY,DQ);
|
||||
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
// Xh = X^T . Sh
|
||||
double Xh[2];
|
||||
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,Xh);
|
||||
|
||||
double B[4];
|
||||
DeviceMatrix H(B,2,2);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
H(i,j) = H0(i,j,qx,qy,e);
|
||||
}
|
||||
}
|
||||
|
||||
// p2 = B . Xh
|
||||
double p2[2];
|
||||
kernels::Mult(2,2,B,Xh,p2);
|
||||
kernels::PushEval<MQ1,NBZ>(qx,qy,p2,QQ);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
kernels::LoadBt<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
kernels::EvalXt<MD1,MQ1,NBZ>(D1D,Q1D,B,QQ,DQ);
|
||||
kernels::EvalYt<MD1,MQ1,NBZ>(D1D,Q1D,B,DQ,Y,e);
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AddMultGradPA_C0_2D(const Vector &X, const Vector &R,
|
||||
Vector &C) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AddMultGradPA_Kernel_C0_2D,id,N,B,H0,R,C);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,247 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
#include "../linalg/dinvariants.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
using Args = kernels::InvariantsEvaluator2D::Buffers;
|
||||
|
||||
// weight * ddI1
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_001(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const double h = ddi1(r,c);
|
||||
H(r,c,i,j,qx,qy,e) = weight * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 0.5 * weight * dI1b
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_002(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4], ddI1b[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.ddI1(ddI1)
|
||||
.ddI1b(ddI1b)
|
||||
.dI2b(dI2b));
|
||||
const double w = 0.5 * weight;
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
const double h = ddi1b(r,c);
|
||||
H(r,c,i,j,qx,qy,e) = w * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_007(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.ddI1(ddI1)
|
||||
.ddI2(ddI2)
|
||||
.dI1(dI1)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b));
|
||||
const double c1 = 1./ie.Get_I2();
|
||||
const double c2 = weight*c1*c1;
|
||||
const double c3 = ie.Get_I1()*c2;
|
||||
ConstDeviceMatrix di1(ie.Get_dI1(),DIM,DIM);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi1(ie.Get_ddI1(i,j),DIM,DIM);
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r,c,i,j,qx,qy,e) =
|
||||
weight * (1.0 + c1) * ddi1(r,c)
|
||||
- c3 * ddi2(r,c)
|
||||
- c2 * ( di1(i,j) * di2(r,c) + di2(i,j) * di1(r,c) )
|
||||
+ 2.0 * c1 * c3 * di2(r,c) * di2(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline
|
||||
void EvalH_077(const int e, const int qx, const int qy,
|
||||
const double weight, const double *Jpt,
|
||||
DeviceTensor<7,double> H)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
double dI2[4], dI2b[4], ddI2[4];
|
||||
kernels::InvariantsEvaluator2D ie(Args()
|
||||
.J(Jpt)
|
||||
.dI2(dI2)
|
||||
.dI2b(dI2b)
|
||||
.ddI2(ddI2));
|
||||
const double I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
|
||||
ConstDeviceMatrix di2(ie.Get_dI2(),DIM,DIM);
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
ConstDeviceMatrix ddi2(ie.Get_ddI2(i,j),DIM,DIM);
|
||||
for (int r = 0; r < DIM; r++)
|
||||
{
|
||||
for (int c = 0; c < DIM; c++)
|
||||
{
|
||||
H(r,c,i,j,qx,qy,e) =
|
||||
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r,c)
|
||||
+ weight * (I2inv_sq / I2) * di2(r,c) * di2(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_2D,
|
||||
const Vector &x_,
|
||||
const double metric_normal,
|
||||
const int mid,
|
||||
const int NE,
|
||||
const Array<double> &w_,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
const DenseTensor &j_,
|
||||
Vector &h_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
MFEM_VERIFY(mid == 1 || mid == 2 || mid == 7 || mid == 77,
|
||||
"Metric not yet implemented!");
|
||||
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto W = Reshape(w_.Read(), Q1D, Q1D);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto X = Reshape(x_.Read(), D1D, D1D, DIM, NE);
|
||||
auto H = Reshape(h_.Write(), DIM, DIM, DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double s_BG[2][MQ1*MD1];
|
||||
MFEM_SHARED double s_X[2][NBZ][MD1*MD1];
|
||||
MFEM_SHARED double s_DQ[4][NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double s_QQ[4][NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,X,s_X);
|
||||
kernels::LoadBG<MD1,MQ1>(D1D, Q1D, b, g, s_BG);
|
||||
|
||||
kernels::GradX<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_X, s_DQ);
|
||||
kernels::GradY<MD1,MQ1,NBZ>(D1D, Q1D, s_BG, s_DQ, s_QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
const double detJtr = kernels::Det<2>(Jtr);
|
||||
const double weight = metric_normal * W(qx,qy) * detJtr;
|
||||
|
||||
// Jrt = Jtr^{-1}
|
||||
double Jrt[4];
|
||||
kernels::CalcInverse<2>(Jtr, Jrt);
|
||||
|
||||
// Jpr = X^t.DSh
|
||||
double Jpr[4];
|
||||
kernels::PullGrad<MQ1,NBZ>(qx,qy,s_QQ,Jpr);
|
||||
|
||||
// Jpt = Jpr.Jrt
|
||||
double Jpt[4];
|
||||
kernels::Mult(2,2,2, Jpr, Jrt, Jpt);
|
||||
|
||||
// metric->AssembleH
|
||||
if (mid == 1) { EvalH_001(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 2) { EvalH_002(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 7) { EvalH_007(e,qx,qy,weight,Jpt,H); }
|
||||
if (mid == 77) { EvalH_077(e,qx,qy,weight,Jpt,H); }
|
||||
} // qx
|
||||
} // qy
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_2D(const Vector &X) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int M = metric->Id();
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const double mn = metric_normal;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &W = PA.ir->GetWeights();
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_2D,id,X,mn,M,N,W,B,G,J,H);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,125 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "linearform.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, SetupGradPA_C0_2D,
|
||||
const double lim_normal,
|
||||
const Vector &lim_dist,
|
||||
const Vector &c0_,
|
||||
const int NE,
|
||||
const DenseTensor &j_,
|
||||
const Array<double> &w_,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bld_,
|
||||
Vector &h0_,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 2;
|
||||
constexpr int NBZ = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const bool const_c0 = c0_.Size() == 1;
|
||||
const auto C0 = const_c0 ?
|
||||
Reshape(c0_.Read(), 1, 1, 1) :
|
||||
Reshape(c0_.Read(), Q1D, Q1D, NE);
|
||||
const auto LD = Reshape(lim_dist.Read(), D1D, D1D, NE);
|
||||
const auto J = Reshape(j_.Read(), DIM, DIM, Q1D, Q1D, NE);
|
||||
const auto W = Reshape(w_.Read(), Q1D, Q1D);
|
||||
const auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
const auto bld = Reshape(bld_.Read(), Q1D, D1D);
|
||||
|
||||
auto H0 = Reshape(h0_.Write(), DIM, DIM, Q1D, Q1D, NE);
|
||||
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : T_MAX;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : T_MAX;
|
||||
|
||||
MFEM_SHARED double B[MQ1*MD1];
|
||||
MFEM_SHARED double BLD[MQ1*MD1];
|
||||
|
||||
MFEM_SHARED double XY[NBZ][MD1*MD1];
|
||||
MFEM_SHARED double DQ[NBZ][MD1*MQ1];
|
||||
MFEM_SHARED double QQ[NBZ][MQ1*MQ1];
|
||||
|
||||
kernels::LoadX<MD1,NBZ>(e,D1D,LD,XY);
|
||||
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,b,B);
|
||||
kernels::LoadB<MD1,MQ1>(D1D,Q1D,bld,BLD);
|
||||
|
||||
kernels::EvalX<MD1,MQ1,NBZ>(D1D,Q1D,BLD,XY,DQ);
|
||||
kernels::EvalY<MD1,MQ1,NBZ>(D1D,Q1D,BLD,DQ,QQ);
|
||||
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,e);
|
||||
const double detJtr = kernels::Det<2>(Jtr);
|
||||
const double weight = W(qx,qy) * detJtr;
|
||||
const double coeff0 = const_c0 ? C0(0,0,0) : C0(qx,qy,e);
|
||||
const double weight_m = weight * lim_normal * coeff0;
|
||||
|
||||
double D;
|
||||
kernels::PullEval<MQ1,NBZ>(qx,qy,QQ,D);
|
||||
const double dist = D; // GetValues, default comp set to 0
|
||||
|
||||
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
|
||||
// d2.Diag(1.0 / (dist * dist), x.Size());
|
||||
const double c = 1.0 / (dist * dist);
|
||||
double grad_grad[4];
|
||||
kernels::Diag<2>(c, grad_grad);
|
||||
ConstDeviceMatrix gg(grad_grad,DIM,DIM);
|
||||
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
H0(i,j,qx,qy,e) = weight_m * gg(i,j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &X) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const double ln = lim_normal;
|
||||
const Vector &LD = PA.LD;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &W = PA.ir->GetWeights();
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &BLD = PA.maps_lim->B;
|
||||
const Vector &C0 = PA.C0;
|
||||
Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(SetupGradPA_C0_2D,id,ln,LD,C0,N,J,W,B,BLD,H0);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,150 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_3D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const DenseTensor &j,
|
||||
const Vector &h,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
const auto J = Reshape(j.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
const auto H = Reshape(h.Read(), DIM, DIM, DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
|
||||
MFEM_SHARED double qdd[MQ1*MD1*MD1];
|
||||
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
|
||||
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; ++v)
|
||||
{
|
||||
for (int i = 0; i < DIM; i++)
|
||||
{
|
||||
for (int j = 0; j < DIM; j++)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD(qx,qy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double *Jtr = &J(0,0,qx,qy,qz,e);
|
||||
double jrt[9];
|
||||
ConstDeviceMatrix Jrt(jrt,3,3);
|
||||
kernels::CalcInverse<3>(Jtr, jrt);
|
||||
const double Bz = B(qz,dz);
|
||||
const double Gz = G(qz,dz);
|
||||
const double L = i==2 ? Gz : Bz;
|
||||
const double R = j==2 ? Gz : Bz;
|
||||
const double Jij = Jrt(i,i) * Jrt(j,j);
|
||||
const double h = H(v,i,v,j,qx,qy,qz,e);
|
||||
QQD(qx,qy,dz) += L * Jij * h * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// second tensor contraction, along y direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD(qx,dy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
const double Gy = G(qy,dy);
|
||||
const double L = i==1 ? Gy : By;
|
||||
const double R = j==1 ? Gy : By;
|
||||
QDD(qx,dy,dz) += L * QQD(qx,qy,dz) * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// third tensor contraction, along x direction
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
const double Gx = G(qx,dx);
|
||||
const double L = i==0 ? Gx : Bx;
|
||||
const double R = j==0 ? Gx : Bx;
|
||||
d += L * QDD(qx,dy,dz) * R;
|
||||
}
|
||||
D(dx,dy,dz,v,e) += d;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const DenseTensor &J = PA.Jtr;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Array<double> &G = PA.maps->G;
|
||||
const Vector &H = PA.H;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_3D,id,N,B,G,J,H,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,123 +0,0 @@
|
||||
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "tmop.hpp"
|
||||
#include "tmop_pa.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_REGISTER_TMOP_KERNELS(void, AssembleDiagonalPA_Kernel_C0_3D,
|
||||
const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &h0,
|
||||
Vector &diagonal,
|
||||
const int d1d,
|
||||
const int q1d)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
const auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
const auto H0 = Reshape(h0.Read(), DIM, DIM, Q1D, Q1D, Q1D, NE);
|
||||
|
||||
auto D = Reshape(diagonal.ReadWrite(), D1D, D1D, D1D, DIM, NE);
|
||||
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
MFEM_SHARED double qqd[MQ1*MQ1*MD1];
|
||||
MFEM_SHARED double qdd[MQ1*MD1*MD1];
|
||||
DeviceTensor<3,double> QQD(qqd, MQ1, MQ1, MD1);
|
||||
DeviceTensor<3,double> QDD(qdd, MQ1, MD1, MD1);
|
||||
|
||||
for (int v = 0; v < DIM; ++v)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD(qx,qy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const double Bz = B(qz,dz);
|
||||
QQD(qx,qy,dz) += Bz * H0(v,v,qx,qy,qz,e) * Bz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// second tensor contraction, along y direction
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD(qx,dy,dz) = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
QDD(qx,dy,dz) += By * QQD(qx,qy,dz) * By;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// third tensor contraction, along x direction
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double d = 0.0;
|
||||
MFEM_UNROLL(MQ1);
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
d += Bx * QDD(qx,dy,dz) * Bx;
|
||||
}
|
||||
D(dx,dy,dz, v, e) += d;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &D) const
|
||||
{
|
||||
const int N = PA.ne;
|
||||
const int D1D = PA.maps->ndof;
|
||||
const int Q1D = PA.maps->nqpt;
|
||||
const int id = (D1D << 4 ) | Q1D;
|
||||
const Array<double> &B = PA.maps->B;
|
||||
const Vector &H0 = PA.H0;
|
||||
|
||||
MFEM_LAUNCH_TMOP_KERNEL(AssembleDiagonalPA_Kernel_C0_3D,id,N,B,H0,D);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user